Skip to main content

docling_pdf/
textparse.rs

1//! Pure-Rust PDF text extraction (replacing pdfium's glyph layer).
2//!
3//! pdfium reports *rendered* glyph boxes, which diverge from docling's
4//! `docling-parse` C++ parser at exactly the points that drive conformance:
5//! generated spaces get a zero-width box, combining diacritics get a real-width
6//! box, and ligature/fraction glyphs land at different x. This module instead
7//! reconstructs each glyph's box from the **font's own advance widths** and the
8//! PDF text/graphics matrices — the same information docling-parse uses — so a
9//! space is as wide as the font says and a combining mark has zero advance.
10//!
11//! The output is the same [`Glyph`] stream pdfium produces (native PDF
12//! coordinates, y-up), fed straight into the existing docling-parse line
13//! sanitizer ([`crate::dp_lines`]). Only the digital text layer is handled here;
14//! pages without one still fall back to OCR upstream.
15
16use std::collections::HashMap;
17use std::sync::Arc;
18
19use lopdf::{Dictionary, Document, Object};
20
21use crate::pdfium_backend::Glyph;
22
23/// Per-document caches for the content-stream interpreter. Fonts are indirect
24/// objects shared by many pages, but were fully re-parsed — ToUnicode CMap
25/// decompression + tokenization, embedded Type1 program scan, width tables —
26/// for **every page and every Form XObject invocation**; decoded form content
27/// streams were likewise re-inflated on every `Do`. Cached per document,
28/// keyed by the referenced object id (fonts also by resource name, which
29/// feeds the docling-parse font hash). Inline (non-reference) dicts are rare
30/// and stay uncached.
31#[derive(Default)]
32struct DocCaches {
33    fonts: HashMap<(lopdf::ObjectId, Vec<u8>), Arc<Font>>,
34    forms: HashMap<lopdf::ObjectId, Arc<lopdf::content::Content>>,
35}
36
37/// A 2×3 affine matrix `[a b c d e f]`: maps `(x,y)` → `(a·x+c·y+e, b·x+d·y+f)`.
38#[derive(Clone, Copy)]
39struct Mat {
40    a: f64,
41    b: f64,
42    c: f64,
43    d: f64,
44    e: f64,
45    f: f64,
46}
47
48impl Mat {
49    const ID: Mat = Mat {
50        a: 1.0,
51        b: 0.0,
52        c: 0.0,
53        d: 1.0,
54        e: 0.0,
55        f: 0.0,
56    };
57
58    /// `self ∘ m`: the matrix that applies `self` first, then `m`.
59    fn then(self, m: Mat) -> Mat {
60        Mat {
61            a: self.a * m.a + self.b * m.c,
62            b: self.a * m.b + self.b * m.d,
63            c: self.c * m.a + self.d * m.c,
64            d: self.c * m.b + self.d * m.d,
65            e: self.e * m.a + self.f * m.c + m.e,
66            f: self.e * m.b + self.f * m.d + m.f,
67        }
68    }
69
70    fn apply(self, x: f64, y: f64) -> (f64, f64) {
71        (
72            self.a * x + self.c * y + self.e,
73            self.b * x + self.d * y + self.f,
74        )
75    }
76}
77
78/// A parsed font: how to turn raw string bytes into (unicode, advance) pairs.
79struct Font {
80    /// 2-byte codes (Type0 / Identity-H) vs 1-byte (simple fonts).
81    two_byte: bool,
82    /// code → Unicode string (from ToUnicode; may be multi-char, e.g. ligatures).
83    to_unicode: HashMap<u32, String>,
84    /// code → glyph advance, in 1000-unit glyph space.
85    widths: HashMap<u32, f64>,
86    default_width: f64,
87    /// 1-byte fallback decoding when ToUnicode lacks a code (WinAnsi-ish).
88    simple_encoding: Option<HashMap<u8, char>>,
89    /// code → raw `/Differences` glyph name, for GID-style names (`g115`) that
90    /// have no Unicode mapping. docling-parse emits these verbatim as `/g115`
91    /// (see the redp5110 bulleted list); matching it keeps no text skipped.
92    fallback_names: HashMap<u8, String>,
93    /// code → char from the embedded Type1 font program's own `/Encoding` vector,
94    /// used only as a last resort for glyphs the base encoding leaves unmapped
95    /// (standard TeX math fonts: `λ`, `≤`, …).
96    program_encoding: HashMap<u8, char>,
97    ascent: f64,
98    descent: f64,
99    hash: u64,
100    /// Weight and slant read from the `/BaseFont` name (the heading
101    /// hierarchy's style signal, #302).
102    style: crate::font_style::FontStyle,
103}
104
105impl Font {
106    fn decode_code(&self, code: u32) -> (Option<String>, f64) {
107        let w = self
108            .widths
109            .get(&code)
110            .copied()
111            .unwrap_or(self.default_width);
112        if let Some(s) = self.to_unicode.get(&code) {
113            return (Some(decompose_ligatures(s)), w);
114        }
115        if !self.two_byte {
116            // A GID-style `/Differences` name (no Unicode) overrides the base
117            // encoding, matching docling's verbatim `/g115` fallback.
118            if let Some(name) = self.fallback_names.get(&(code as u8)) {
119                return (Some(format!("/{name}")), w);
120            }
121            if let Some(enc) = &self.simple_encoding {
122                if let Some(&ch) = enc.get(&(code as u8)) {
123                    return (Some(decompose_ligatures(&ch.to_string())), w);
124                }
125            }
126            // Last resort: the embedded Type1 font program's own `/Encoding`
127            // vector (`dup N /glyphname put`). Standard TeX math fonts (CMMI, CMSY,
128            // …) ship no PDF `/Encoding` and no ToUnicode, so a glyph like `λ`
129            // (CMMI code 21 → `/lambda`) or `≤` (CMSY code 20 → `/lessequal`) has
130            // no other mapping and would otherwise be silently dropped. docling
131            // recovers these from the same font program. This only fills codes the
132            // base encoding left unmapped, so it never changes an existing decode.
133            if let Some(&ch) = self.program_encoding.get(&(code as u8)) {
134                return (Some(decompose_ligatures(&ch.to_string())), w);
135            }
136        }
137        (None, w)
138    }
139}
140
141/// Spell out Latin presentation-form ligatures (`fi`→`fi`, `ffi`→`ffi`, …) the way
142/// docling does, so `configuration`/`difficult` don't keep the ligature glyph.
143/// The chars share the ligature's box, so the line sanitizer recomposes them.
144fn decompose_ligatures(s: &str) -> String {
145    if !s.chars().any(|c| ('\u{FB00}'..='\u{FB06}').contains(&c)) {
146        return s.to_string();
147    }
148    s.chars()
149        .map(|c| {
150            match c {
151                '\u{FB00}' => "ff",
152                '\u{FB01}' => "fi",
153                '\u{FB02}' => "fl",
154                '\u{FB03}' => "ffi",
155                '\u{FB04}' => "ffl",
156                '\u{FB05}' => "ft",
157                '\u{FB06}' => "st",
158                _ => return c.to_string(),
159            }
160            .to_string()
161        })
162        .collect()
163}
164
165fn hash_name(name: &[u8]) -> u64 {
166    use std::hash::{Hash, Hasher};
167    let mut h = std::collections::hash_map::DefaultHasher::new();
168    name.hash(&mut h);
169    h.finish()
170}
171
172/// Resolve a possibly-indirect object to a dictionary.
173fn as_dict<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Dictionary> {
174    match obj {
175        Object::Dictionary(d) => Some(d),
176        Object::Reference(id) => doc.get_object(*id).ok().and_then(|o| o.as_dict().ok()),
177        _ => None,
178    }
179}
180
181fn deref<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Object> {
182    match obj {
183        Object::Reference(id) => doc.get_object(*id).ok(),
184        other => Some(other),
185    }
186}
187
188/// Parse one font dictionary into a [`Font`].
189fn parse_font(doc: &Document, name: &[u8], fdict: &Dictionary) -> Font {
190    let subtype: &[u8] = fdict
191        .get(b"Subtype")
192        .ok()
193        .and_then(|o| o.as_name().ok())
194        .unwrap_or(&[]);
195    let two_byte = subtype == b"Type0".as_slice();
196
197    let to_unicode = fdict
198        .get(b"ToUnicode")
199        .ok()
200        .and_then(|o| deref(doc, o))
201        .and_then(|o| o.as_stream().ok())
202        .and_then(|s| s.decompressed_content().ok())
203        .map(|data| parse_tounicode(&data))
204        .unwrap_or_default();
205
206    let (mut widths, mut default_width) = if two_byte {
207        cid_widths(doc, fdict)
208    } else {
209        simple_widths(doc, fdict)
210    };
211
212    let simple_encoding = if two_byte {
213        None
214    } else {
215        Some(simple_encoding_table(doc, fdict))
216    };
217
218    // A standard-14 font referenced without an embedded program usually ships
219    // no `/Widths` and no `/FontDescriptor` either (ReportLab's default, #187).
220    // Every advance then resolved to 0, the cells collapsed to zero width, and
221    // the page's whole text layer was silently dropped — while pdfium, with
222    // its built-in metrics, reads the same file fine. Fill the widths from the
223    // built-in Adobe Core 14 AFM tables via the font's own code→char decode
224    // (base encoding + `/Differences`), so an explicit `/Widths` always wins.
225    if !two_byte && widths.is_empty() && default_width == 0.0 {
226        if let Some(std14) = base_font_name(fdict).and_then(|n| crate::std14::widths_for(&n)) {
227            if let Some(enc) = &simple_encoding {
228                for (&code, &ch) in enc {
229                    if let Some(w) = std14.width(ch) {
230                        widths.insert(u32::from(code), w);
231                    }
232                }
233            }
234            // Codes the table misses still advance a typical width instead of
235            // stacking at x=0 (the failure mode this whole branch fixes).
236            default_width = 500.0;
237        }
238    }
239    let fallback_names = if two_byte {
240        HashMap::new()
241    } else {
242        differences_gid_names(doc, fdict)
243    };
244    let program_encoding = if two_byte {
245        HashMap::new()
246    } else {
247        type1_program_encoding(doc, fdict)
248    };
249
250    let (ascent, descent) = font_ascent_descent(doc, fdict, two_byte);
251
252    Font {
253        two_byte,
254        to_unicode,
255        widths,
256        default_width,
257        simple_encoding,
258        fallback_names,
259        program_encoding,
260        ascent,
261        descent,
262        hash: hash_name(name),
263        style: crate::font_style::parse_font_style(&String::from_utf8_lossy(
264            &base_font_name(fdict).unwrap_or_else(|| name.to_vec()),
265        )),
266    }
267}
268
269/// Collect `/Differences` entries whose glyph name is a GID placeholder
270/// (`g115`, `cid42`, `glyph7`, `index9`) with no Unicode mapping. docling-parse
271/// emits such glyphs as the literal name `/g115`; mapping them here keeps the
272/// text from being silently dropped (subsetted fonts with no ToUnicode). The
273/// GID-name restriction keeps real Adobe glyph names on the normal path so this
274/// never invents garbage on the clean files.
275fn differences_gid_names(doc: &Document, fdict: &Dictionary) -> HashMap<u8, String> {
276    let mut map = HashMap::new();
277    let Some(Object::Dictionary(enc)) = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o))
278    else {
279        return map;
280    };
281    let Some(Object::Array(diffs)) = enc.get(b"Differences").ok().and_then(|o| deref(doc, o))
282    else {
283        return map;
284    };
285    let mut code = 0u8;
286    for el in diffs {
287        match el {
288            Object::Integer(i) => code = *i as u8,
289            Object::Name(name) => {
290                if glyph_name_to_char(name).is_none() && is_gid_name(name) {
291                    map.insert(code, String::from_utf8_lossy(name).into_owned());
292                }
293                code = code.wrapping_add(1);
294            }
295            _ => {}
296        }
297    }
298    map
299}
300
301/// Parse the embedded Type1 font program's built-in `/Encoding` vector
302/// (`dup <code> /<glyphname> put` entries in the clear-text header before
303/// `eexec`) into `code → char`. This is how docling recovers glyphs from
304/// standard TeX math fonts (CMMI/CMSY/…) that carry no PDF `/Encoding` and no
305/// ToUnicode — e.g. CMMI's `dup 21 /lambda` or CMSY's `dup 20 /lessequal`.
306/// Only `FontFile` (Type1) is parsed; CFF (`FontFile3`) and TrueType
307/// (`FontFile2`) store their encoding in a binary table and are left alone.
308fn type1_program_encoding(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
309    let mut map = HashMap::new();
310    let Some(desc) = fdict
311        .get(b"FontDescriptor")
312        .ok()
313        .and_then(|o| deref(doc, o))
314        .and_then(|o| o.as_dict().ok())
315    else {
316        return map;
317    };
318    let Some(data) = desc
319        .get(b"FontFile")
320        .ok()
321        .and_then(|o| deref(doc, o))
322        .and_then(|o| o.as_stream().ok())
323        .and_then(|s| s.decompressed_content().ok())
324    else {
325        return map;
326    };
327    // The clear-text header (PostScript) ends at `eexec`; the rest is encrypted.
328    let head_end = data
329        .windows(5)
330        .position(|w| w == b"eexec")
331        .unwrap_or(data.len());
332    let head = String::from_utf8_lossy(&data[..head_end]);
333    // Scan for `dup <code> /<name> put` tokens.
334    let toks: Vec<&str> = head.split_whitespace().collect();
335    for w in toks.windows(4) {
336        if w[0] == "dup" && w[3] == "put" {
337            if let (Ok(code), Some(name)) = (w[1].parse::<u32>(), w[2].strip_prefix('/')) {
338                if code <= 255 {
339                    if let Some(ch) = glyph_name_to_char(name.as_bytes()) {
340                        map.insert(code as u8, ch);
341                    }
342                }
343            }
344        }
345    }
346    map
347}
348
349/// A glyph name that is a synthetic placeholder, not a real Adobe name:
350/// `g115`, `cid42`, `glyph7`, `index9`, `G12`, or a short-prefix code name like
351/// `SM590000` (IBM BookMaster). These carry no Unicode meaning, and docling-parse
352/// emits them verbatim (`/SM590000`). `afii####` / `uni####` are real Adobe names
353/// and excluded. The restriction keeps genuine glyph names on the Unicode path.
354fn is_gid_name(name: &[u8]) -> bool {
355    let Ok(s) = std::str::from_utf8(name) else {
356        return false;
357    };
358    if s.starts_with("afii") || s.starts_with("uni") {
359        return false;
360    }
361    for prefix in ["g", "G", "cid", "CID", "glyph", "index"] {
362        if let Some(rest) = s.strip_prefix(prefix) {
363            if !rest.is_empty() && rest.bytes().all(|b| b.is_ascii_digit()) {
364                return true;
365            }
366        }
367    }
368    // Short alpha prefix (≤3 letters) followed by a run of ≥3 digits — synthetic
369    // code names like `SM590000`, distinct from real Adobe names (whole words or
370    // letter+`.suffix` variants).
371    let alpha = s.bytes().take_while(|b| b.is_ascii_alphabetic()).count();
372    let digits = s.len() - alpha;
373    (1..=3).contains(&alpha)
374        && digits >= 3
375        && s.as_bytes()[alpha..].iter().all(|b| b.is_ascii_digit())
376}
377
378fn font_ascent_descent(doc: &Document, fdict: &Dictionary, two_byte: bool) -> (f64, f64) {
379    // For Type0, the descriptor lives on the descendant CIDFont.
380    let descr_owner = if two_byte {
381        fdict
382            .get(b"DescendantFonts")
383            .ok()
384            .and_then(|o| deref(doc, o))
385            .and_then(|o| match o {
386                Object::Array(a) => a.first(),
387                _ => None,
388            })
389            .and_then(|o| as_dict(doc, o))
390    } else {
391        Some(fdict)
392    };
393    let fd = descr_owner
394        .and_then(|d| d.get(b"FontDescriptor").ok())
395        .and_then(|o| as_dict(doc, o));
396    let asc = fd
397        .and_then(|d| d.get(b"Ascent").ok())
398        .and_then(|o| {
399            o.as_float()
400                .ok()
401                .or_else(|| o.as_i64().ok().map(|i| i as f32))
402        })
403        .unwrap_or(750.0) as f64;
404    let desc = fd
405        .and_then(|d| d.get(b"Descent").ok())
406        .and_then(|o| {
407            o.as_float()
408                .ok()
409                .or_else(|| o.as_i64().ok().map(|i| i as f32))
410        })
411        .unwrap_or(-250.0) as f64;
412    // Some subsetted fonts carry a degenerate FontDescriptor (`/Ascent 0
413    // /Descent 0`) — the real metrics live in the font program. That collapses
414    // the loose box to zero height, so the line cells get zero area and the
415    // layout's region/text assignment drops them (2305's References list lost
416    // every prose line, keeping only the URLs). Fall back to typical text metrics
417    // so the box has height.
418    if asc - desc <= 1.0 {
419        return (750.0, -250.0);
420    }
421    (asc, desc)
422}
423
424/// The `/BaseFont` name with any `ABCDEF+` subset prefix stripped (#187).
425fn base_font_name(fdict: &Dictionary) -> Option<Vec<u8>> {
426    let name = fdict.get(b"BaseFont").ok()?.as_name().ok()?;
427    let stripped = match name.iter().position(|&b| b == b'+') {
428        Some(i) if i == 6 => &name[i + 1..],
429        _ => name,
430    };
431    Some(stripped.to_vec())
432}
433
434/// Simple-font widths: `/FirstChar` + `/Widths` array, `/MissingWidth` default.
435fn simple_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
436    let mut map = HashMap::new();
437    let first = fdict
438        .get(b"FirstChar")
439        .ok()
440        .and_then(|o| o.as_i64().ok())
441        .unwrap_or(0) as u32;
442    if let Some(Object::Array(arr)) = fdict.get(b"Widths").ok().and_then(|o| deref(doc, o)) {
443        for (i, w) in arr.iter().enumerate() {
444            if let Some(w) = num(w) {
445                map.insert(first + i as u32, w);
446            }
447        }
448    }
449    let dw = fdict
450        .get(b"FontDescriptor")
451        .ok()
452        .and_then(|o| as_dict(doc, o))
453        .and_then(|d| d.get(b"MissingWidth").ok())
454        .and_then(num)
455        .unwrap_or(0.0);
456    (map, dw)
457}
458
459/// CIDFont widths: the `/W` array on the descendant font (`/DW` default = 1000).
460fn cid_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
461    let mut map = HashMap::new();
462    let Some(desc) = fdict
463        .get(b"DescendantFonts")
464        .ok()
465        .and_then(|o| deref(doc, o))
466        .and_then(|o| match o {
467            Object::Array(a) => a.first(),
468            _ => None,
469        })
470        .and_then(|o| as_dict(doc, o))
471    else {
472        return (map, 1000.0);
473    };
474    let dw = desc.get(b"DW").ok().and_then(num).unwrap_or(1000.0);
475    if let Some(Object::Array(w)) = desc.get(b"W").ok().and_then(|o| deref(doc, o)) {
476        let mut i = 0;
477        while i < w.len() {
478            let c = w.get(i).and_then(num);
479            match (c, w.get(i + 1)) {
480                // `c [w1 w2 ...]`: consecutive CIDs starting at c.
481                (Some(c), Some(Object::Array(list))) => {
482                    for (k, wv) in list.iter().enumerate() {
483                        if let Some(wv) = num(wv) {
484                            map.insert(c as u32 + k as u32, wv);
485                        }
486                    }
487                    i += 2;
488                }
489                // `c_first c_last w`: a run all of width w.
490                (Some(c1), Some(o2)) => {
491                    if let (Some(c2), Some(wv)) = (num(o2), w.get(i + 2).and_then(num)) {
492                        for cid in c1 as u32..=c2 as u32 {
493                            map.insert(cid, wv);
494                        }
495                    }
496                    i += 3;
497                }
498                _ => break,
499            }
500        }
501    }
502    (map, dw)
503}
504
505fn num(o: &Object) -> Option<f64> {
506    match o {
507        Object::Integer(i) => Some(*i as f64),
508        Object::Real(r) => Some(*r as f64),
509        _ => None,
510    }
511}
512
513/// Parse a ToUnicode CMap's `bfchar` / `bfrange` sections into code→string.
514pub(crate) fn parse_tounicode(data: &[u8]) -> HashMap<u32, String> {
515    let text = String::from_utf8_lossy(data);
516    let mut map = HashMap::new();
517    let hex = |s: &str| -> Option<Vec<u16>> {
518        let s = s.trim();
519        if !s.starts_with('<') || !s.ends_with('>') {
520            return None;
521        }
522        let h = &s[1..s.len() - 1];
523        let bytes: Vec<u8> = (0..h.len())
524            .step_by(2)
525            .filter_map(|i| u8::from_str_radix(h.get(i..i + 2)?, 16).ok())
526            .collect();
527        Some(
528            bytes
529                .chunks(2)
530                .map(|c| {
531                    if c.len() == 2 {
532                        u16::from_be_bytes([c[0], c[1]])
533                    } else {
534                        c[0] as u16
535                    }
536                })
537                .collect(),
538        )
539    };
540    let u16s_to_string = |u: &[u16]| String::from_utf16_lossy(u);
541    let code_of = |u: &[u16]| u.iter().fold(0u32, |acc, &x| (acc << 16) | x as u32);
542
543    // Tokenize by structure, not whitespace: CMap hex groups are often written
544    // back-to-back with no separators (`<21><21><0054>`), so scan for `<…>`
545    // groups, `[`/`]` brackets, and bareword keywords.
546    let tokens: Vec<String> = {
547        let bytes = text.as_bytes();
548        let mut toks = Vec::new();
549        let mut i = 0;
550        while i < bytes.len() {
551            let c = bytes[i];
552            if c.is_ascii_whitespace() {
553                i += 1;
554            } else if c == b'<' {
555                let start = i;
556                while i < bytes.len() && bytes[i] != b'>' {
557                    i += 1;
558                }
559                i += 1; // include '>'
560                toks.push(String::from_utf8_lossy(&bytes[start..i.min(bytes.len())]).into_owned());
561            } else if c == b'[' || c == b']' {
562                toks.push((c as char).to_string());
563                i += 1;
564            } else {
565                let start = i;
566                while i < bytes.len()
567                    && !bytes[i].is_ascii_whitespace()
568                    && bytes[i] != b'<'
569                    && bytes[i] != b'['
570                    && bytes[i] != b']'
571                {
572                    i += 1;
573                }
574                toks.push(String::from_utf8_lossy(&bytes[start..i]).into_owned());
575            }
576        }
577        toks
578    };
579    let tokens: Vec<&str> = tokens.iter().map(|s| s.as_str()).collect();
580    let mut i = 0;
581    while i < tokens.len() {
582        match tokens[i] {
583            "beginbfchar" => {
584                i += 1;
585                while i + 1 < tokens.len() && tokens[i] != "endbfchar" {
586                    if let (Some(src), Some(dst)) = (hex(tokens[i]), hex(tokens[i + 1])) {
587                        map.insert(code_of(&src), u16s_to_string(&dst));
588                    }
589                    i += 2;
590                }
591            }
592            "beginbfrange" => {
593                i += 1;
594                while i + 2 < tokens.len() && tokens[i] != "endbfrange" {
595                    let (Some(lo), Some(hi)) = (hex(tokens[i]), hex(tokens[i + 1])) else {
596                        i += 1;
597                        continue;
598                    };
599                    let lo = code_of(&lo);
600                    let hi = code_of(&hi);
601                    if tokens[i + 2] == "[" {
602                        // `<lo> <hi> [ <d0> <d1> ... ]`: one dst per code in the range.
603                        let mut j = i + 3;
604                        let mut code = lo;
605                        while j < tokens.len() && tokens[j] != "]" {
606                            if let Some(dst) = hex(tokens[j]) {
607                                map.insert(code, u16s_to_string(&dst));
608                            }
609                            code += 1;
610                            j += 1;
611                        }
612                        i = j + 1;
613                    } else if let Some(dst) = hex(tokens[i + 2]) {
614                        // `<lo> <hi> <dst>`: consecutive Unicode from a base.
615                        let base = code_of(&dst);
616                        for (k, code) in (lo..=hi).enumerate() {
617                            if let Some(ch) = char::from_u32(base + k as u32) {
618                                map.insert(code, ch.to_string());
619                            }
620                        }
621                        i += 3;
622                    } else {
623                        i += 1;
624                    }
625                }
626            }
627            _ => i += 1,
628        }
629    }
630    map
631}
632
633/// Decode a PDF string literal in a Tj/TJ operand into raw code units.
634fn codes(font: &Font, bytes: &[u8]) -> Vec<u32> {
635    if font.two_byte {
636        bytes
637            .chunks(2)
638            .map(|c| {
639                if c.len() == 2 {
640                    ((c[0] as u32) << 8) | c[1] as u32
641                } else {
642                    c[0] as u32
643                }
644            })
645            .collect()
646    } else {
647        bytes.iter().map(|&b| b as u32).collect()
648    }
649}
650
651/// A page's display box, in PDF user space: the `/CropBox` clipped to the
652/// `/MediaBox`, both inherited through the page tree and normalized — a
653/// missing or empty MediaBox is US Letter, an empty CropBox is the MediaBox
654/// (pdfium's `CPDF_Page::UpdateDimensions`). pdfium reports the page size from
655/// this box, renders exactly it, and translates every content coordinate so
656/// its lower-left corner is the origin (`m_PageMatrix`); docling's backends
657/// inherit that frame, so text cells, `prov` boxes and destinations all count
658/// from the CropBox corner, not the MediaBox one. The parser used to flip
659/// glyphs with the MediaBox *height* and no translation at all, so a page
660/// whose boxes do not start at (0, 0) — a trimmed book page with
661/// `MediaBox [-56 -58 576 723]` / `CropBox [1 -0.6 519 666]`, or a LaTeX
662/// figure cropped to `[156 147 637 391]` — had its text displaced against the
663/// rendered bitmap by the box offset, the bottom lines pushed past the page
664/// edge and clamped to `t = b`.
665#[derive(Debug, Clone, Copy, PartialEq)]
666pub(crate) struct PageBox {
667    /// Left edge, user space.
668    pub l: f32,
669    /// Bottom edge, user space.
670    pub b: f32,
671    pub w: f32,
672    pub h: f32,
673}
674
675impl PageBox {
676    /// Top edge, user space — the y that becomes `0` in the y-down frame.
677    pub fn top(&self) -> f32 {
678        self.b + self.h
679    }
680}
681
682/// A page-tree rect attribute (`/MediaBox`, `/CropBox`), inherited from the
683/// nearest ancestor that sets it, as normalized `(l, b, r, t)`.
684fn inherited_rect(
685    doc: &Document,
686    page_id: lopdf::ObjectId,
687    key: &[u8],
688) -> Option<(f32, f32, f32, f32)> {
689    let mut id = page_id;
690    for _ in 0..32 {
691        let dict = doc.get_object(id).ok()?.as_dict().ok()?;
692        if let Some(Object::Array(a)) = dict.get(key).ok().and_then(|o| deref(doc, o)) {
693            let v: Vec<f32> = a.iter().filter_map(|o| num(o).map(|x| x as f32)).collect();
694            if v.len() == 4 && v.iter().all(|x| x.is_finite()) {
695                return Some((
696                    v[0].min(v[2]),
697                    v[1].min(v[3]),
698                    v[0].max(v[2]),
699                    v[1].max(v[3]),
700                ));
701            }
702            return None;
703        }
704        id = dict.get(b"Parent").ok()?.as_reference().ok()?;
705    }
706    None
707}
708
709pub(crate) fn page_box(doc: &Document, page_id: lopdf::ObjectId) -> PageBox {
710    let nonempty = |r: &(f32, f32, f32, f32)| r.2 > r.0 && r.3 > r.1;
711    let media = inherited_rect(doc, page_id, b"MediaBox")
712        .filter(nonempty)
713        .unwrap_or((0.0, 0.0, 612.0, 792.0));
714    let crop = inherited_rect(doc, page_id, b"CropBox")
715        .map(|c| {
716            (
717                c.0.max(media.0),
718                c.1.max(media.1),
719                c.2.min(media.2),
720                c.3.min(media.3),
721            )
722        })
723        .filter(nonempty)
724        .unwrap_or(media);
725    PageBox {
726        l: crop.0,
727        b: crop.1,
728        w: crop.2 - crop.0,
729        h: crop.3 - crop.1,
730    }
731}
732
733/// Page size (width, height) in PDF points — the display box's, like pdfium's
734/// `FPDF_GetPageWidthF/HeightF`.
735fn page_size(doc: &Document, page_id: lopdf::ObjectId) -> (f32, f32) {
736    let pb = page_box(doc, page_id);
737    (pb.w, pb.h)
738}
739
740/// Localize where a page's text is lost, for the `text_layer` diagnostic.
741/// Extraction can come up empty at three different points — no content stream
742/// reached the parser, the stream did not decode into operators, or it ran but
743/// produced no glyphs (fonts/encodings) — and from the outside all three look
744/// the same. Report them per page.
745pub fn content_diagnosis(bytes: &[u8]) -> String {
746    let Some(doc) = load_document(bytes) else {
747        return "document does not load".into();
748    };
749    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
750    pages.sort_by_key(|(n, _)| *n);
751    let mut out = String::new();
752    let mut caches = DocCaches::default();
753    for (n, pid) in pages.into_iter().take(4) {
754        let content_bytes = doc.get_page_content(pid);
755        let ops = lopdf::content::Content::decode(&content_bytes)
756            .map(|c| c.operations.len())
757            .ok();
758        let res = page_res(&doc, pid);
759        let fonts = res.map(|r| fonts_from_res(&doc, r, &mut caches).len());
760        let glyphs = page_glyphs_cached(&doc, pid, &mut caches).len();
761        out.push_str(&format!(
762            "\n   page {n}: content {} B, ops {}, resources {}, fonts {}, glyphs {}",
763            content_bytes.len(),
764            ops.map_or("UNDECODABLE".to_string(), |n| n.to_string()),
765            if res.is_some() { "ok" } else { "MISSING" },
766            fonts.map_or("-".to_string(), |n| n.to_string()),
767            glyphs,
768        ));
769    }
770    out
771}
772
773/// Is this "text layer" a vestige rather than the document's text?
774///
775/// Scanned forms often carry a handful of typed-in strings — a date filled
776/// into three form fields, say — on top of pages that are otherwise images.
777/// Treating that as a real text layer is the worst of both worlds: the text
778/// path proudly extracts thirteen characters, and no OCR ever runs on the
779/// letter the pages actually show. The reported form did exactly this (3
780/// lines, 13 chars, 3 pages).
781///
782/// The rule is deliberately tight so genuinely sparse *digital* documents are
783/// not misrouted into OCR: only a document averaging at most one line per page
784/// **and** totalling fewer than 32 characters is called vestigial.
785pub fn text_layer_is_vestigial(pages: &[crate::pdfium_backend::PdfPage]) -> bool {
786    let lines: usize = pages.iter().map(|p| p.cells.len()).sum();
787    if lines == 0 {
788        return true;
789    }
790    let chars: usize = pages
791        .iter()
792        .flat_map(|p| &p.cells)
793        .map(|c| c.text.chars().count())
794        .sum();
795    lines <= pages.len() && chars < 32
796}
797
798/// Why the cross-reference repair did or did not fire, for the `text_layer`
799/// diagnostic. A PDF that will not load is indistinguishable from a scan in
800/// production (both convert to nothing), so the reason has to be askable.
801pub fn xref_repair_status(bytes: &[u8]) -> String {
802    if Document::load_mem(bytes).is_ok() {
803        return "loads unaided; no repair needed".into();
804    }
805    match pad_short_xref_entries(bytes) {
806        Ok(fixed) => match Document::load_mem(&fixed) {
807            Ok(_) => "repaired: cross-reference entries padded to 20 bytes".into(),
808            Err(e) => format!("padded the entries, but it still will not load: {e}"),
809        },
810        Err(why) => format!("repair declined — {why}"),
811    }
812}
813
814/// Load a PDF, repairing the one malformation that otherwise costs us the whole
815/// document: **19-byte cross-reference entries**.
816///
817/// The spec fixes an xref entry at 20 bytes — `nnnnnnnnnn ggggg n` plus a
818/// *two*-byte EOL. Some generators (an Austrian telecom's invoices, for one)
819/// emit a bare LF instead, making each entry 19 bytes. lopdf rejects the file
820/// outright (`invalid file trailer`) where pdfium reads it happily, so a
821/// perfectly good text layer looked to the browser exactly like a scan and cost
822/// ten seconds of OCR.
823///
824/// Padding is only attempted when it cannot move anything the xref points at:
825/// a single `xref` section that begins after the last object. The repair then
826/// has to prove itself — the padded bytes are used only if they load — so a
827/// mis-repair degrades to today's behaviour rather than to silent garbage.
828pub(crate) fn load_document(bytes: &[u8]) -> Option<Document> {
829    open_document(bytes, None).ok()
830}
831
832/// Why [`open_document`] could not hand back a readable document.
833#[derive(Debug, Clone, Copy, PartialEq, Eq)]
834pub(crate) enum OpenError {
835    /// lopdf cannot read the file even after the repairs.
836    Unreadable,
837    /// The file is encrypted and the password given (or none) does not open it.
838    Password,
839}
840
841/// Load `bytes` with the document's password. lopdf decrypts while reading —
842/// with `password`, or the empty user password most "protected" PDFs carry
843/// (what every viewer opens silently) — so every reader downstream sees plain
844/// streams; a file the password does not open loads with its streams still
845/// encrypted, which is reported as [`OpenError::Password`] rather than handed
846/// on as a document whose every stream decodes to nothing.
847pub(crate) fn open_document(bytes: &[u8], password: Option<&str>) -> Result<Document, OpenError> {
848    let Some(doc) = load_document_raw(bytes, password) else {
849        // lopdf refuses to load at all under a *wrong* password (a missing
850        // one loads the file with its streams still encrypted); tell the two
851        // apart by loading without it.
852        if password.is_some() && load_document_raw(bytes, None).is_some_and(|d| d.is_encrypted()) {
853            return Err(OpenError::Password);
854        }
855        return Err(OpenError::Unreadable);
856    };
857    if doc.is_encrypted() {
858        return Err(OpenError::Password);
859    }
860    Ok(doc)
861}
862
863fn load_options(password: Option<&str>) -> lopdf::LoadOptions {
864    lopdf::LoadOptions {
865        password: password.map(str::to_string),
866        ..lopdf::LoadOptions::default()
867    }
868}
869
870fn load_document_raw(bytes: &[u8], password: Option<&str>) -> Option<Document> {
871    // Try progressively more repair, and accept a candidate only once the pages
872    // actually carry content — a document whose streams were dropped still
873    // "loads", so loading alone is not evidence the repair helped. A
874    // well-formed file returns on the first attempt and pays for nothing.
875    let mut fallback = None;
876    if let Some(doc) = best_effort_load(bytes, password, &mut fallback) {
877        return Some(doc);
878    }
879    let xref_fixed = pad_short_xref_entries(bytes).ok();
880    if let Some(fixed) = &xref_fixed {
881        if let Some(doc) = best_effort_load(fixed, password, &mut fallback) {
882            return Some(doc);
883        }
884    }
885    // Both defects can coexist, and the second only becomes visible once the
886    // first is repaired, so build on whatever the previous step produced.
887    let lengths_fixed = fix_stream_lengths(xref_fixed.as_deref().unwrap_or(bytes));
888    if let Some(doc) = best_effort_load(&lengths_fixed, password, &mut fallback) {
889        return Some(doc);
890    }
891    fallback
892}
893
894/// Load `data`, returning it only when its pages carry content; a document that
895/// merely parses is remembered as the fallback for when nothing does better.
896fn best_effort_load(
897    data: &[u8],
898    password: Option<&str>,
899    fallback: &mut Option<Document>,
900) -> Option<Document> {
901    match Document::load_mem_with_options(data, load_options(password)) {
902        Ok(doc) if has_page_content(&doc) => Some(doc),
903        Ok(doc) => {
904            fallback.get_or_insert(doc);
905            None
906        }
907        Err(_) => None,
908    }
909}
910
911/// Does any page actually hand us a content stream? A document whose streams
912/// were dropped still parses — it simply has nothing to read — so this is what
913/// tells a successful repair from a pointless one.
914fn has_page_content(doc: &Document) -> bool {
915    doc.get_pages()
916        .into_values()
917        .take(4)
918        .any(|pid| !doc.get_page_content(pid).is_empty())
919}
920
921/// Correct `/Length` values that disagree with where `endstream` actually is.
922///
923/// The same generator that writes short xref entries also overstates its
924/// content-stream lengths by a byte or two. lopdf trusts `/Length`, reads past
925/// the data, fails to find `endstream` there and drops the stream — the object
926/// comes back as a bare dictionary, so the page has no content at all and the
927/// document looks like a scan. pdfium instead trusts `endstream`, which is what
928/// this does.
929///
930/// The rewrite is length-preserving: the corrected number is written over the
931/// old digits and padded with spaces, so every byte offset in the file — and
932/// therefore the whole cross-reference table — stays valid.
933fn fix_stream_lengths(bytes: &[u8]) -> Vec<u8> {
934    let mut out = bytes.to_vec();
935    let mut i = 0;
936    while let Some(rel) = find(&out[i..], b"stream") {
937        let kw = i + rel;
938        i = kw + 6;
939        // Skip `endstream` (the keyword we are measuring *to*).
940        if kw >= 3 && &out[kw - 3..kw] == b"end" {
941            continue;
942        }
943        // The stream data starts after the EOL that follows the keyword.
944        let mut data = kw + 6;
945        if out.get(data..data + 2) == Some(b"\r\n".as_slice()) {
946            data += 2;
947        } else if matches!(out.get(data), Some(b'\n' | b'\r')) {
948            data += 1;
949        }
950        let Some(end) = find(&out[data..], b"endstream").map(|r| data + r) else {
951            continue;
952        };
953        // `/Length <digits>` in the dictionary just before the keyword.
954        let dict_start = out[..kw].iter().rposition(|&c| c == b'<').unwrap_or(0);
955        let Some(lrel) = find(&out[dict_start..kw], b"/Length") else {
956            continue;
957        };
958        let mut d = dict_start + lrel + 7;
959        while matches!(out.get(d), Some(b' ')) {
960            d += 1;
961        }
962        let digits = out[d..].iter().take_while(|c| c.is_ascii_digit()).count();
963        if digits == 0 {
964            continue;
965        }
966        let declared: usize = match std::str::from_utf8(&out[d..d + digits])
967            .ok()
968            .and_then(|s| s.parse().ok())
969        {
970            Some(v) => v,
971            None => continue,
972        };
973        let actual = end - data;
974        // Only shrink, and only when the new value fits the space the old one
975        // occupied — growing the number would move every following byte.
976        let replacement = actual.to_string();
977        if actual == declared || replacement.len() > digits {
978            continue;
979        }
980        out[d..d + digits].fill(b' ');
981        out[d..d + replacement.len()].copy_from_slice(replacement.as_bytes());
982    }
983    out
984}
985
986fn find(haystack: &[u8], needle: &[u8]) -> Option<usize> {
987    haystack.windows(needle.len()).position(|w| w == needle)
988}
989
990/// Rewrite a classic cross-reference table's entries to the spec's 20 bytes,
991/// or `None` when the file's shape makes that unsafe (see [`load_document`]).
992fn pad_short_xref_entries(bytes: &[u8]) -> Result<Vec<u8>, &'static str> {
993    // Exactly one xref section, and it must start after every object, so that
994    // growing it shifts nothing the table's offsets refer to.
995    let is_boundary = |i: usize| i == 0 || matches!(bytes[i - 1], b'\n' | b'\r');
996    let mut starts = (0..bytes.len().saturating_sub(4))
997        .filter(|&i| &bytes[i..i + 4] == b"xref" && is_boundary(i));
998    let xref_at = starts
999        .next()
1000        .ok_or("no classic `xref` section (an xref stream?)")?;
1001    if starts.next().is_some() {
1002        return Err("more than one xref section (incremental update)");
1003    }
1004    let last_obj = bytes
1005        .windows(3)
1006        .rposition(|w| w == b"obj")
1007        .ok_or("no objects found")?;
1008    if last_obj > xref_at {
1009        return Err("an object follows the xref — padding would move it");
1010    }
1011
1012    let mut out = bytes[..xref_at].to_vec();
1013    out.extend_from_slice(b"xref\n");
1014    let mut i = xref_at + 4;
1015    let skip_ws = |i: &mut usize| {
1016        while matches!(bytes.get(*i), Some(b'\r' | b'\n' | b' ')) {
1017            *i += 1;
1018        }
1019    };
1020    loop {
1021        skip_ws(&mut i);
1022        // Either the next subsection header ("first count") or the trailer.
1023        if bytes[i..].starts_with(b"trailer") {
1024            out.extend_from_slice(&bytes[i..]);
1025            return Ok(out);
1026        }
1027        let header_end = i + bytes[i..]
1028            .iter()
1029            .position(|c| matches!(c, b'\n' | b'\r'))
1030            .ok_or("subsection header runs off the end")?;
1031        let header = std::str::from_utf8(&bytes[i..header_end])
1032            .map_err(|_| "subsection header is not text")?
1033            .trim();
1034        let mut parts = header.split_whitespace();
1035        let count: usize = parts
1036            .nth(1)
1037            .and_then(|c| c.parse().ok())
1038            .ok_or("unparseable subsection header")?;
1039        if parts.next().is_some() || count == 0 {
1040            return Err("unexpected subsection header shape");
1041        }
1042        out.extend_from_slice(header.as_bytes());
1043        out.push(b'\n');
1044        i = header_end;
1045        for _ in 0..count {
1046            skip_ws(&mut i);
1047            // `nnnnnnnnnn ggggg n` — the 18 bytes before whatever EOL follows.
1048            let entry = bytes.get(i..i + 18).ok_or("xref entry runs off the end")?;
1049            let well_formed = entry[..10].iter().all(u8::is_ascii_digit)
1050                && entry[10] == b' '
1051                && entry[11..16].iter().all(u8::is_ascii_digit)
1052                && entry[16] == b' '
1053                && matches!(entry[17], b'n' | b'f');
1054            if !well_formed {
1055                return Err("xref entry is not `nnnnnnnnnn ggggg n`");
1056            }
1057            out.extend_from_slice(entry);
1058            out.extend_from_slice(b" \n"); // the spec's 2-byte EOL -> 20 bytes
1059            i += 18;
1060        }
1061    }
1062}
1063
1064/// Debug: raw glyph stream `(ch, ll, lr, lb, lt)` (native coords) for page
1065/// `index`, before the sanitizer. For comparing char cells to docling-parse.
1066pub fn debug_glyphs(bytes: &[u8], index: usize) -> Vec<(char, f32, f32, f32, f32)> {
1067    let Some(doc) = load_document(bytes) else {
1068        return Vec::new();
1069    };
1070    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1071    pages.sort_by_key(|(n, _)| *n);
1072    let Some((_, pid)) = pages.get(index) else {
1073        return Vec::new();
1074    };
1075    page_glyphs(&doc, *pid)
1076        .into_iter()
1077        .map(|g| (g.ch, g.ll, g.lr, g.lb, g.lt))
1078        .collect()
1079}
1080
1081/// Public entry: per-page (width, height, line cells) for a PDF, via the Rust
1082/// text parser + the docling-parse line sanitizer. Used by the pipeline and the
1083/// `textparse_dump` example.
1084pub fn pdf_textlines(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
1085    let Some(doc) = load_document(bytes) else {
1086        return Vec::new();
1087    };
1088    let mut caches = DocCaches::default();
1089    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1090    pages.sort_by_key(|(n, _)| *n);
1091    pages
1092        .into_iter()
1093        .map(|(_, pid)| {
1094            let (w, h) = page_size(&doc, pid);
1095            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1096            let cells = crate::dp_lines::line_cells(&glyphs, h, true);
1097            (w, h, cells)
1098        })
1099        .collect()
1100}
1101
1102/// Debug/diagnostic entry: per-page (width, height, word cells) for a PDF, via
1103/// the Rust parser glyphs run through the docling-parse word grouping. Used to
1104/// compare parser word cells against docling-parse's `word_cells` oracle (roadmap
1105/// item 6).
1106pub fn pdf_words(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
1107    let Some(doc) = load_document(bytes) else {
1108        return Vec::new();
1109    };
1110    let mut caches = DocCaches::default();
1111    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1112    pages.sort_by_key(|(n, _)| *n);
1113    pages
1114        .into_iter()
1115        .map(|(_, pid)| {
1116            let (w, h) = page_size(&doc, pid);
1117            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1118            let cells = crate::dp_lines::word_cells(&glyphs, h, true);
1119            (w, h, cells)
1120        })
1121        .collect()
1122}
1123
1124/// One page's text cells from the pure-Rust parser: prose line cells, per-word
1125/// cells, and code line cells — all from a single glyph parse. Replaces the
1126/// pdfium text path (roadmap item 6) when the parser drop is enabled.
1127#[derive(Default)]
1128pub struct PageParserCells {
1129    pub prose: Vec<crate::pdfium_backend::TextCell>,
1130    pub words: Vec<crate::pdfium_backend::TextCell>,
1131    pub code: Vec<crate::pdfium_backend::TextCell>,
1132}
1133
1134/// The parser text layer, driven one page at a time: the document is loaded
1135/// (and repaired, see [`load_document`]) once, the font/form caches persist
1136/// across pages, and each page's glyphs are parsed only when asked for.
1137///
1138/// The eager whole-document walk this replaces ran *before* the first page
1139/// was rendered, so on a long PDF it was a serial prefix the page-worker pool
1140/// sat idle through — 6.2 s on the 1913-page .NET reference, in front of a
1141/// pipeline that otherwise overlaps parsing with inference — and a `--pages`
1142/// window still paid for every page in the file. Pulling pages on demand
1143/// keeps the parse on the producer thread but interleaved with rendering,
1144/// and skips unselected pages entirely. Output per page is unchanged: same
1145/// glyph walk, same shared caches, same contraction.
1146pub struct PageTextParser {
1147    doc: Document,
1148    caches: DocCaches,
1149    /// Page object ids in document order (page 1 first).
1150    pages: Vec<lopdf::ObjectId>,
1151}
1152
1153impl PageTextParser {
1154    /// Load the document; `None` when lopdf cannot read it at all.
1155    pub fn open(bytes: &[u8]) -> Option<Self> {
1156        Self::open_with_password(bytes, None)
1157    }
1158
1159    /// [`open`](Self::open) with the document's password (an encrypted file
1160    /// the password does not open reads as `None` here; the pipeline reports
1161    /// it through [`crate::pdf_meta::PdfMeta::open_with_password`] first).
1162    pub fn open_with_password(bytes: &[u8], password: Option<&str>) -> Option<Self> {
1163        let doc = open_document(bytes, password).ok()?;
1164        let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1165        pages.sort_by_key(|(n, _)| *n);
1166        Some(Self {
1167            doc,
1168            caches: DocCaches::default(),
1169            pages: pages.into_iter().map(|(_, pid)| pid).collect(),
1170        })
1171    }
1172
1173    /// The glyph boxes and font styles of the 0-based page `index` — the
1174    /// heading-hierarchy stage's style signal (#302): every non-space glyph's
1175    /// box (font ascent + descent at its size — the font-size proxy pdfium's
1176    /// loose char box also gave) in top-left coordinates, with the weight
1177    /// class and slant its `/BaseFont` name declares. Empty for a page
1178    /// without a text layer (a scan), and the stage falls back to its other
1179    /// signals.
1180    pub(crate) fn glyph_styles(
1181        &mut self,
1182        index: usize,
1183    ) -> Vec<crate::heading_hierarchy::GlyphStyle> {
1184        let Some(&pid) = self.pages.get(index) else {
1185            return Vec::new();
1186        };
1187        let (_w, h) = page_size(&self.doc, pid);
1188        let glyphs = page_glyphs_cached(&self.doc, pid, &mut self.caches);
1189        // Font hash → style, over the fonts the walk just parsed (an inline,
1190        // uncached font dictionary reads as unstyled).
1191        let styles: HashMap<u64, crate::font_style::FontStyle> = self
1192            .caches
1193            .fonts
1194            .values()
1195            .map(|f| (f.hash, f.style))
1196            .collect();
1197        glyphs
1198            .iter()
1199            .filter(|g| !g.ch.is_whitespace() && g.ll.is_finite())
1200            .map(|g| {
1201                let st = styles.get(&g.font).copied().unwrap_or_default();
1202                crate::heading_hierarchy::GlyphStyle {
1203                    l: g.ll,
1204                    t: h - g.lt,
1205                    r: g.lr,
1206                    b: h - g.lb,
1207                    height: g.height(),
1208                    weight_cls: crate::font_style::weight_class(st.weight),
1209                    italic: st.italic,
1210                    styled: st.known,
1211                }
1212            })
1213            .collect()
1214    }
1215
1216    /// Prose, word and code cells of the 0-based page `index` — empty for an
1217    /// index the parser's page tree doesn't have (a damaged file whose page
1218    /// tree disagrees with the object model's count).
1219    pub fn cells(&mut self, index: usize) -> PageParserCells {
1220        let Some(&pid) = self.pages.get(index) else {
1221            return PageParserCells::default();
1222        };
1223        let (_w, h) = page_size(&self.doc, pid);
1224        let glyphs = page_glyphs_cached(&self.doc, pid, &mut self.caches);
1225        let (prose, words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1226        PageParserCells {
1227            prose,
1228            words,
1229            code: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1230        }
1231    }
1232}
1233
1234/// Full parser text layer: prose + word + code cells per page, glyphs parsed once.
1235/// `prose`/`words` come from the docling-parse contraction ([`crate::dp_lines`]);
1236/// `code` splits only at the parser's own space glyphs (monospace keeps its
1237/// source spacing). The eager form of [`PageTextParser`].
1238pub fn pdf_all_cells(bytes: &[u8]) -> Vec<PageParserCells> {
1239    let Some(mut parser) = PageTextParser::open(bytes) else {
1240        return Vec::new();
1241    };
1242    (0..parser.pages.len()).map(|i| parser.cells(i)).collect()
1243}
1244
1245/// Whole pages for the text-layer-only conversion ([`crate::convert_text_layer`]):
1246/// the parser's prose/word/code cells plus page geometry, assembled into
1247/// [`PdfPage`]s with no rendered image and no link annotations. Everything here
1248/// is pure Rust (lopdf), so it compiles without the `ml` feature — including on
1249/// wasm32. A page the parser can't read (no text layer) comes back with empty
1250/// cells; there is no pdfium fallback on this path.
1251pub fn pdf_text_pages(bytes: &[u8]) -> Vec<crate::pdfium_backend::PdfPage> {
1252    let Some(doc) = load_document(bytes) else {
1253        return Vec::new();
1254    };
1255    let mut caches = DocCaches::default();
1256    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1257    pages.sort_by_key(|(n, _)| *n);
1258    pages
1259        .into_iter()
1260        .map(|(_, pid)| {
1261            let (w, h) = page_size(&doc, pid);
1262            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1263            let (mut prose, mut words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1264            drop_overpainted_cells(&mut prose);
1265            drop_overpainted_cells(&mut words);
1266            crate::pdfium_backend::PdfPage {
1267                #[cfg(feature = "ocr-prep")]
1268                image_layout: None,
1269                width: w,
1270                height: h,
1271                // Cells are native PDF points; there is no rendered bitmap.
1272                scale: 1.0,
1273                cells: prose,
1274                code_cells: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1275                word_cells: words,
1276                #[cfg(feature = "ocr-prep")]
1277                image: image::RgbImage::new(1, 1),
1278                links: Vec::new(),
1279                rotation: 0,
1280            }
1281        })
1282        .collect()
1283}
1284
1285/// Drop line cells that are *painted over each other* — glyphs used as artwork.
1286///
1287/// Some generators draw their logo with a symbol font: on the reporting
1288/// invoice, a `TeleLogo` Type1 paints the T-Mobile mark by stacking the glyphs
1289/// encoded as `"` and `==` on top of one another, and the flat text-layer
1290/// output opened with that garbage. Nothing in the font metadata gives it away
1291/// (the *text* fonts in the same file are also flagged symbolic, and the logo
1292/// font names its glyphs `quotedbl` &c.), but the geometry does: two cells with
1293/// different text where one lies inside the other on the same line is
1294/// physically impossible for prose — ink from two words never occupies the
1295/// same box. Both cells of such a pair are paint, not text.
1296///
1297/// Containment (not mere overlap) keeps this narrow: adjacent words touch but
1298/// never contain each other, and a same-text near-duplicate (double-draw faux
1299/// bold) is left alone for the sanitizer's usual handling. Applied on the
1300/// flat/browser path only — the ML pipeline's text layer is byte-pinned by the
1301/// PDF corpus, and there the layout model already sinks logo marks into
1302/// `picture` regions.
1303fn drop_overpainted_cells(cells: &mut Vec<crate::pdfium_backend::TextCell>) {
1304    let mut paint = vec![false; cells.len()];
1305    for i in 0..cells.len() {
1306        for j in 0..cells.len() {
1307            if i == j || cells[i].text == cells[j].text {
1308                continue;
1309            }
1310            let (a, b) = (&cells[i], &cells[j]);
1311            // Same line band: the vertical overlap covers most of the shorter.
1312            let vo = (a.b.min(b.b) - a.t.max(b.t)).max(0.0);
1313            if vo < 0.6 * (a.b - a.t).min(b.b - b.t) {
1314                continue;
1315            }
1316            // `a` horizontally inside `b` (with a small tolerance).
1317            let ho = (a.r.min(b.r) - a.l.max(b.l)).max(0.0);
1318            if ho >= 0.8 * (a.r - a.l) && (a.r - a.l) <= (b.r - b.l) {
1319                paint[i] = true;
1320                paint[j] = true;
1321            }
1322        }
1323    }
1324    let mut keep = paint.iter().map(|p| !p);
1325    cells.retain(|_| keep.next().unwrap());
1326}
1327
1328/// The text-state scalars inherited by a Form XObject when it is invoked via
1329/// `Do` (the PDF graphics state includes the text parameters, but not the text
1330/// matrices, which a form re-establishes inside its own `BT`/`ET`).
1331#[derive(Clone, Copy)]
1332struct TextState {
1333    tc: f64,
1334    tw: f64,
1335    th: f64,
1336    tl: f64,
1337    trise: f64,
1338    fsize: f64,
1339}
1340
1341impl TextState {
1342    const INIT: TextState = TextState {
1343        tc: 0.0,
1344        tw: 0.0,
1345        th: 1.0,
1346        tl: 0.0,
1347        trise: 0.0,
1348        fsize: 0.0,
1349    };
1350}
1351
1352/// The effective `/Resources` dictionary for a page (inline or via reference,
1353/// falling back to an inherited one from a `/Parent`).
1354fn page_res(doc: &Document, page_id: lopdf::ObjectId) -> Option<&Dictionary> {
1355    let (inline, ids) = doc.get_page_resources(page_id).ok()?;
1356    if let Some(d) = inline {
1357        return Some(d);
1358    }
1359    ids.into_iter().find_map(|id| doc.get_dictionary(id).ok())
1360}
1361
1362/// Build the code→[`Font`] map for a resources dictionary's `/Font` sub-dict,
1363/// reusing the per-document cache for fonts referenced indirectly (the common
1364/// case — the same font objects recur on every page).
1365fn fonts_from_res(
1366    doc: &Document,
1367    res: &Dictionary,
1368    caches: &mut DocCaches,
1369) -> HashMap<Vec<u8>, Arc<Font>> {
1370    let mut map = HashMap::new();
1371    let font_dict = res
1372        .get(b"Font")
1373        .ok()
1374        .and_then(|o| deref(doc, o))
1375        .and_then(|o| o.as_dict().ok());
1376    if let Some(fd) = font_dict {
1377        for (name, value) in fd.iter() {
1378            let font = match value {
1379                Object::Reference(id) => {
1380                    let key = (*id, name.clone());
1381                    if let Some(f) = caches.fonts.get(&key) {
1382                        Arc::clone(f)
1383                    } else if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1384                        let f = Arc::new(parse_font(doc, name, fdict));
1385                        caches.fonts.insert(key, Arc::clone(&f));
1386                        f
1387                    } else {
1388                        continue;
1389                    }
1390                }
1391                _ => {
1392                    if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1393                        Arc::new(parse_font(doc, name, fdict))
1394                    } else {
1395                        continue;
1396                    }
1397                }
1398            };
1399            map.insert(name.clone(), font);
1400        }
1401    }
1402    map
1403}
1404
1405/// Extract every glyph on a page as a native-coordinate [`Glyph`].
1406/// [`PageTextParser::glyph_styles`] for the given **1-based** pages of a
1407/// document, keyed by page number — a separate, on-demand pass (no
1408/// rendering), so the extraction pipeline stays byte-identical whether or not
1409/// the heading-hierarchy stage runs. Empty when lopdf cannot open the file.
1410pub(crate) fn glyph_styles(
1411    bytes: &[u8],
1412    pages: &[usize],
1413) -> HashMap<usize, Vec<crate::heading_hierarchy::GlyphStyle>> {
1414    let mut out = HashMap::new();
1415    let Some(mut parser) = PageTextParser::open(bytes) else {
1416        return out;
1417    };
1418    for &page_no in pages {
1419        if page_no == 0 {
1420            continue;
1421        }
1422        let styles = parser.glyph_styles(page_no - 1);
1423        if !styles.is_empty() {
1424            out.insert(page_no, styles);
1425        }
1426    }
1427    out
1428}
1429
1430pub(crate) fn page_glyphs(doc: &Document, page_id: lopdf::ObjectId) -> Vec<Glyph> {
1431    page_glyphs_cached(doc, page_id, &mut DocCaches::default())
1432}
1433
1434/// [`page_glyphs`] with an explicit per-document cache, so a multi-page walk
1435/// parses each font / decodes each form once instead of once per page.
1436fn page_glyphs_cached(
1437    doc: &Document,
1438    page_id: lopdf::ObjectId,
1439    caches: &mut DocCaches,
1440) -> Vec<Glyph> {
1441    let mut out = Vec::new();
1442    // lopdf 0.44: get_page_content returns the assembled content-stream bytes
1443    // directly (an empty Vec when the page has none).
1444    let content_bytes = doc.get_page_content(page_id);
1445    let Ok(content) = lopdf::content::Content::decode(&content_bytes) else {
1446        return out;
1447    };
1448    if let Some(res) = page_res(doc, page_id) {
1449        // pdfium's page matrix: user space translated so the display box's
1450        // lower-left corner is the origin (see [`PageBox`]).
1451        let pb = page_box(doc, page_id);
1452        let base = Mat {
1453            e: -(pb.l as f64),
1454            f: -(pb.b as f64),
1455            ..Mat::ID
1456        };
1457        run_content(
1458            doc,
1459            res,
1460            &content,
1461            base,
1462            TextState::INIT,
1463            0,
1464            caches,
1465            &mut out,
1466        );
1467        out.retain(|g| on_page(g, pb.w, pb.h));
1468    }
1469    out
1470}
1471
1472/// Whether a glyph is on the page (#529): docling-parse keeps a character
1473/// only when its whole box lies inside the display box — the CropBox, or
1474/// the MediaBox without one — edges included, so a FrameMaker print slug
1475/// drawn beside the CropBox or a tiled page's neighbouring text beyond the
1476/// MediaBox never becomes a cell, and a line crossing the edge is cut at the
1477/// last glyph that fits (`Crossing the rig`). The box is docling-parse's
1478/// char box — the advance by the font's ascent/descent, the loose box here
1479/// (its axis-aligned extent for rotated text) — in the frame `page_glyphs`
1480/// already moved to the display box's corner, so the page is `[0, w] × [0,
1481/// h]`. A glyph without a finite box is kept, as before. (A standard-14 font
1482/// without a FontDescriptor gets the 750 / −250 default ascent / descent
1483/// here where docling-parse reads the AFM's — Helvetica's 718 / −207 — so
1484/// for those a glyph within ~0.04 em of the top or bottom edge can fall the
1485/// other way; horizontally the advance is the same.)
1486fn on_page(g: &Glyph, w: f32, h: f32) -> bool {
1487    // f32 noise only: docling-parse already drops a glyph 0.01 pt over.
1488    const EPS: f32 = 1e-3;
1489    let (l, b, r, t) = if [g.ll, g.lb, g.lr, g.lt].iter().all(|v| v.is_finite()) {
1490        (g.ll, g.lb, g.lr, g.lt)
1491    } else {
1492        (g.l, g.b, g.r, g.t)
1493    };
1494    if ![l, b, r, t].iter().all(|v| v.is_finite()) {
1495        return true;
1496    }
1497    l >= -EPS && b >= -EPS && r <= w + EPS && t <= h + EPS
1498}
1499
1500/// Run a content stream's operators, emitting glyphs into `out`. Recurses into
1501/// Form XObjects on `Do` (bulk body text in heavy PDFs lives inside a form, not
1502/// the page content stream). `res` is the resources dict in scope (the page's,
1503/// or the form's own); `base_ctm` is the CTM at the point of invocation.
1504#[allow(clippy::too_many_arguments)]
1505fn run_content(
1506    doc: &Document,
1507    res: &Dictionary,
1508    content: &lopdf::content::Content,
1509    base_ctm: Mat,
1510    init: TextState,
1511    depth: u32,
1512    caches: &mut DocCaches,
1513    out: &mut Vec<Glyph>,
1514) {
1515    let fonts = fonts_from_res(doc, res, caches);
1516    let xobjects = res
1517        .get(b"XObject")
1518        .ok()
1519        .and_then(|o| deref(doc, o))
1520        .and_then(|o| o.as_dict().ok());
1521
1522    // Graphics + text state. `q`/`Q` save and restore the whole graphics state,
1523    // which includes the text parameters (Tc, Tw, Tz, TL, Tfs, Trise, font) —
1524    // *not* the text matrix (that is reset by BT). Saving only the CTM let a Tc
1525    // set inside a `q…Q` block leak out and drift every later glyph.
1526    #[allow(clippy::type_complexity)]
1527    let mut gstate_stack: Vec<(Mat, f64, f64, f64, f64, f64, f64, Option<&Arc<Font>>)> = Vec::new();
1528    let mut ctm = base_ctm;
1529    let mut tm = Mat::ID;
1530    let mut tlm = Mat::ID;
1531    let mut font: Option<&Arc<Font>> = None;
1532    let mut fsize = init.fsize;
1533    let mut tc = init.tc; // char spacing
1534    let mut tw = init.tw; // word spacing
1535    let mut th = init.th; // horizontal scale (Tz/100)
1536    let mut tl = init.tl; // leading
1537    let mut trise = init.trise;
1538
1539    let op_f = |operands: &[Object], i: usize| operands.get(i).and_then(num).unwrap_or(0.0);
1540
1541    for op in &content.operations {
1542        let operands = &op.operands;
1543        match op.operator.as_str() {
1544            "q" => gstate_stack.push((ctm, tc, tw, th, tl, trise, fsize, font)),
1545            "Q" => {
1546                if let Some((c, a, b, h, l, r, fs, f)) = gstate_stack.pop() {
1547                    ctm = c;
1548                    tc = a;
1549                    tw = b;
1550                    th = h;
1551                    tl = l;
1552                    trise = r;
1553                    fsize = fs;
1554                    font = f;
1555                }
1556            }
1557            "cm" => {
1558                let m = Mat {
1559                    a: op_f(operands, 0),
1560                    b: op_f(operands, 1),
1561                    c: op_f(operands, 2),
1562                    d: op_f(operands, 3),
1563                    e: op_f(operands, 4),
1564                    f: op_f(operands, 5),
1565                };
1566                ctm = m.then(ctm);
1567            }
1568            "BT" => {
1569                tm = Mat::ID;
1570                tlm = Mat::ID;
1571            }
1572            "ET" => {}
1573            "Tf" => {
1574                if let Some(Object::Name(n)) = operands.first() {
1575                    font = fonts.get(n.as_slice());
1576                }
1577                fsize = op_f(operands, 1);
1578            }
1579            "Td" => {
1580                tlm = Mat {
1581                    a: 1.0,
1582                    b: 0.0,
1583                    c: 0.0,
1584                    d: 1.0,
1585                    e: op_f(operands, 0),
1586                    f: op_f(operands, 1),
1587                }
1588                .then(tlm);
1589                tm = tlm;
1590            }
1591            "TD" => {
1592                tl = -op_f(operands, 1);
1593                tlm = Mat {
1594                    a: 1.0,
1595                    b: 0.0,
1596                    c: 0.0,
1597                    d: 1.0,
1598                    e: op_f(operands, 0),
1599                    f: op_f(operands, 1),
1600                }
1601                .then(tlm);
1602                tm = tlm;
1603            }
1604            "Tm" => {
1605                tlm = Mat {
1606                    a: op_f(operands, 0),
1607                    b: op_f(operands, 1),
1608                    c: op_f(operands, 2),
1609                    d: op_f(operands, 3),
1610                    e: op_f(operands, 4),
1611                    f: op_f(operands, 5),
1612                };
1613                tm = tlm;
1614            }
1615            "T*" => {
1616                tlm = Mat {
1617                    a: 1.0,
1618                    b: 0.0,
1619                    c: 0.0,
1620                    d: 1.0,
1621                    e: 0.0,
1622                    f: -tl,
1623                }
1624                .then(tlm);
1625                tm = tlm;
1626            }
1627            "Tc" => tc = op_f(operands, 0),
1628            "Tw" => tw = op_f(operands, 0),
1629            "Tz" => th = op_f(operands, 0) / 100.0,
1630            "TL" => tl = op_f(operands, 0),
1631            "Ts" => trise = op_f(operands, 0),
1632            "Tj" | "'" | "\"" => {
1633                if op.operator == "'" || op.operator == "\"" {
1634                    // move to next line first
1635                    tlm = Mat {
1636                        a: 1.0,
1637                        b: 0.0,
1638                        c: 0.0,
1639                        d: 1.0,
1640                        e: 0.0,
1641                        f: -tl,
1642                    }
1643                    .then(tlm);
1644                    tm = tlm;
1645                }
1646                if op.operator == "\"" {
1647                    // `aw ac string "` sets word- and char-spacing before
1648                    // showing the string (PDF 32000-1 §9.4.3), persisting after.
1649                    tw = op_f(operands, 0);
1650                    tc = op_f(operands, 1);
1651                }
1652                if let (Some(f), Some(Object::String(s, _))) = (font, operands.last()) {
1653                    show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out);
1654                }
1655            }
1656            "TJ" => {
1657                if let (Some(f), Some(Object::Array(arr))) = (font, operands.first()) {
1658                    for el in arr {
1659                        match el {
1660                            Object::String(s, _) => {
1661                                show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out)
1662                            }
1663                            other => {
1664                                if let Some(adj) = num(other) {
1665                                    // negative number moves text right (PDF: subtract)
1666                                    let tx = -adj / 1000.0 * fsize * th;
1667                                    tm = Mat {
1668                                        a: 1.0,
1669                                        b: 0.0,
1670                                        c: 0.0,
1671                                        d: 1.0,
1672                                        e: tx,
1673                                        f: 0.0,
1674                                    }
1675                                    .then(tm);
1676                                }
1677                            }
1678                        }
1679                    }
1680                }
1681            }
1682            "Do" => {
1683                // Invoke a Form XObject: bulk body text in many PDFs lives inside
1684                // a form, reached only here. Image XObjects are skipped (no text).
1685                if depth >= 8 {
1686                    continue;
1687                }
1688                let Some(Object::Name(n)) = operands.first() else {
1689                    continue;
1690                };
1691                let obj = xobjects.and_then(|d| d.get(n.as_slice()).ok());
1692                let form_id = match obj {
1693                    Some(Object::Reference(id)) => Some(*id),
1694                    _ => None,
1695                };
1696                let stream = obj
1697                    .and_then(|o| deref(doc, o))
1698                    .and_then(|o| o.as_stream().ok());
1699                let Some(stream) = stream else { continue };
1700                let is_form = stream
1701                    .dict
1702                    .get(b"Subtype")
1703                    .ok()
1704                    .and_then(|o| o.as_name().ok())
1705                    == Some(b"Form".as_slice());
1706                if !is_form {
1707                    continue;
1708                }
1709                // Decode the form's content once per document (headers/footers
1710                // and bulk body text invoke the same form on every page).
1711                let cached = form_id.and_then(|id| caches.forms.get(&id).cloned());
1712                let form_content = match cached {
1713                    Some(c) => c,
1714                    None => {
1715                        let Ok(data) = stream.decompressed_content() else {
1716                            continue;
1717                        };
1718                        let Ok(c) = lopdf::content::Content::decode(&data) else {
1719                            continue;
1720                        };
1721                        let c = Arc::new(c);
1722                        if let Some(id) = form_id {
1723                            caches.forms.insert(id, Arc::clone(&c));
1724                        }
1725                        c
1726                    }
1727                };
1728                // The form's /Matrix maps form space into the CTM at invocation.
1729                let form_mat = match stream.dict.get(b"Matrix").ok() {
1730                    Some(Object::Array(a)) if a.len() == 6 => {
1731                        let v: Vec<f64> = a.iter().filter_map(num).collect();
1732                        if v.len() == 6 {
1733                            Mat {
1734                                a: v[0],
1735                                b: v[1],
1736                                c: v[2],
1737                                d: v[3],
1738                                e: v[4],
1739                                f: v[5],
1740                            }
1741                        } else {
1742                            Mat::ID
1743                        }
1744                    }
1745                    _ => Mat::ID,
1746                };
1747                // The form's own /Resources, falling back to the inherited ones.
1748                let form_res = stream
1749                    .dict
1750                    .get(b"Resources")
1751                    .ok()
1752                    .and_then(|o| deref(doc, o))
1753                    .and_then(|o| o.as_dict().ok())
1754                    .unwrap_or(res);
1755                let state = TextState {
1756                    tc,
1757                    tw,
1758                    th,
1759                    tl,
1760                    trise,
1761                    fsize,
1762                };
1763                run_content(
1764                    doc,
1765                    form_res,
1766                    &form_content,
1767                    form_mat.then(ctm),
1768                    state,
1769                    depth + 1,
1770                    caches,
1771                    out,
1772                );
1773            }
1774            _ => {}
1775        }
1776    }
1777}
1778
1779#[allow(clippy::too_many_arguments)]
1780fn show_text(
1781    font: &Font,
1782    bytes: &[u8],
1783    fsize: f64,
1784    tc: f64,
1785    tw: f64,
1786    th: f64,
1787    trise: f64,
1788    tm: &mut Mat,
1789    ctm: Mat,
1790    out: &mut Vec<Glyph>,
1791) {
1792    for code in codes(font, bytes) {
1793        let (text, w) = font.decode_code(code);
1794        let w0 = w / 1000.0; // advance in text-space (em) units
1795                             // The glyph→user transform: scale glyph space by font size, then Tm, CTM.
1796        let scale = Mat {
1797            a: fsize * th,
1798            b: 0.0,
1799            c: 0.0,
1800            d: fsize,
1801            e: 0.0,
1802            f: trise,
1803        };
1804        let trm = scale.then(*tm).then(ctm);
1805        // Box in glyph space (1000-unit em): x 0..w, y descent..ascent.
1806        let (desc, asc) = (font.descent / 1000.0, font.ascent / 1000.0);
1807        let (x0, y0) = trm.apply(0.0, desc);
1808        let (x1, y1) = trm.apply(w0, desc);
1809        let (x2, y2) = trm.apply(w0, asc);
1810        let (x3, y3) = trm.apply(0.0, asc);
1811        // Upright = the baseline runs left-to-right along +x. Only then is the
1812        // loose box the rectangle spanned by the baseline's x and the em's y
1813        // (a synthetic oblique's skew is ignored, as it always was). A rotated
1814        // matrix — the `0 s -s 0 tx ty Tm` landscape pages are built with
1815        // (#528) — collapsed that rectangle to zero width, and every such run
1816        // vanished; those glyphs keep their real quad (docling-parse's char
1817        // rect) and its extent instead.
1818        let upright = trm.a > 0.0 && trm.b.abs() <= 1e-6 * trm.a;
1819        let (left, bot, right, top, quad) = if upright {
1820            (x0.min(x1), y0.min(y3), x0.max(x1), y0.max(y3), None)
1821        } else {
1822            let (xs, ys) = ([x0, x1, x2, x3], [y0, y1, y2, y3]);
1823            let fold = |v: [f64; 4], f: fn(f64, f64) -> f64| v.into_iter().reduce(f).unwrap();
1824            let q = [x0, y0, x1, y1, x2, y2, x3, y3].map(|v| v as f32);
1825            (
1826                fold(xs, f64::min),
1827                fold(ys, f64::min),
1828                fold(xs, f64::max),
1829                fold(ys, f64::max),
1830                Some(q),
1831            )
1832        };
1833        if let Some(s) = text {
1834            // A run may map one code to multiple chars (ligature/fraction); share box.
1835            for ch in s.chars() {
1836                if ch != '\u{0}' {
1837                    out.push(Glyph {
1838                        ch,
1839                        l: left as f32,
1840                        b: bot as f32,
1841                        r: right as f32,
1842                        t: top as f32,
1843                        ll: left as f32,
1844                        lb: bot as f32,
1845                        lr: right as f32,
1846                        lt: top as f32,
1847                        font: font.hash,
1848                        quad,
1849                    });
1850                }
1851            }
1852        }
1853        // Advance the text matrix. Word spacing applies to single-byte code 32.
1854        let is_space = !font.two_byte && code == 32;
1855        let tx = (w0 * fsize + tc + if is_space { tw } else { 0.0 }) * th;
1856        *tm = Mat {
1857            a: 1.0,
1858            b: 0.0,
1859            c: 0.0,
1860            d: 1.0,
1861            e: tx,
1862            f: 0.0,
1863        }
1864        .then(*tm);
1865    }
1866}
1867
1868/// Build a simple font's code→char table from its `/Encoding`: the base
1869/// encoding (WinAnsi / MacRoman) plus any `/Differences` overrides (glyph names
1870/// resolved through a small Adobe-glyph-name subset).
1871fn simple_encoding_table(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
1872    let enc = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o));
1873    let base_name = match enc {
1874        Some(Object::Name(n)) => n.clone(),
1875        Some(Object::Dictionary(d)) => d
1876            .get(b"BaseEncoding")
1877            .ok()
1878            .and_then(|o| o.as_name().ok())
1879            .map(|n| n.to_vec())
1880            .unwrap_or_default(),
1881        _ => Vec::new(),
1882    };
1883    let mut m = if base_name == b"MacRomanEncoding" {
1884        macroman_table()
1885    } else if base_name.is_empty() {
1886        // No PDF /Encoding at all: the font's *built-in* encoding applies. For
1887        // the standard TeX math fonts that is their fixed TeX layout — falling
1888        // back to StandardEncoding read CMSY's braces as `f`/`g`, `→` as `!`,
1889        // `∈` as `2` (2203's `{ahn,…}` author line). The font program (often
1890        // CFF, which this parser does not read) carries the same mapping;
1891        // docling-parse decodes it from there.
1892        tex_math_builtin(fdict).unwrap_or_else(winansi_table)
1893    } else {
1894        winansi_table()
1895    };
1896    // Apply /Differences: `code /glyphname /glyphname ... code ...`.
1897    if let Some(Object::Dictionary(d)) = enc {
1898        if let Some(Object::Array(diffs)) = d.get(b"Differences").ok().and_then(|o| deref(doc, o)) {
1899            let mut code = 0u8;
1900            for el in diffs {
1901                match el {
1902                    Object::Integer(i) => code = *i as u8,
1903                    Object::Name(name) => {
1904                        if let Some(ch) = glyph_name_to_char(name) {
1905                            m.insert(code, ch);
1906                        }
1907                        code = code.wrapping_add(1);
1908                    }
1909                    _ => {}
1910                }
1911            }
1912        }
1913    }
1914    m
1915}
1916
1917/// The fixed built-in encodings of the standard TeX math fonts (TeXbook
1918/// Appendix F), keyed off the base font name: `CMSY*` (symbols; `CMBSY` is its
1919/// bold) and `CMMI*` (math italic). These fonts ship no PDF `/Encoding` and no
1920/// ToUnicode, and their program is usually CFF — without this table the codes
1921/// fell through to StandardEncoding and rendered as the wrong ASCII.
1922fn tex_math_builtin(fdict: &Dictionary) -> Option<HashMap<u8, char>> {
1923    const CMSY: [char; 128] = [
1924        '−', '·', '×', '∗', '÷', '⋄', '±', '∓', '⊕', '⊖', '⊗', '⊘', '⊙', '◯', '∘', '•', '≍', '≡',
1925        '⊆', '⊇', '≤', '≥', '≼', '≽', '∼', '≈', '⊂', '⊃', '≪', '≫', '≺', '≻', '←', '→', '↑', '↓',
1926        '↔', '↗', '↘', '≃', '⇐', '⇒', '⇑', '⇓', '⇔', '↖', '↙', '∝', '′', '∞', '∈', '∋', '△', '▽',
1927        '\u{338}', '↦', '∀', '∃', '¬', '∅', 'ℜ', 'ℑ', '⊤', '⊥', 'ℵ', 'A', 'B', 'C', 'D', 'E', 'F',
1928        'G', 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X',
1929        'Y', 'Z', '∪', '∩', '⊎', '∧', '∨', '⊢', '⊣', '⌊', '⌋', '⌈', '⌉', '{', '}', '⟨', '⟩', '|',
1930        '∥', '↕', '⇕', '\\', '≀', '√', '∐', '∇', '∫', '⊔', '⊓', '⊑', '⊒', '§', '†', '‡', '¶', '♣',
1931        '♢', '♡', '♠',
1932    ];
1933    const CMMI: [char; 128] = [
1934        'Γ', 'Δ', 'Θ', 'Λ', 'Ξ', 'Π', 'Σ', 'Υ', 'Φ', 'Ψ', 'Ω', 'α', 'β', 'γ', 'δ', 'ε', 'ζ', 'η',
1935        'θ', 'ι', 'κ', 'λ', 'μ', 'ν', 'ξ', 'π', 'ρ', 'σ', 'τ', 'υ', 'φ', 'χ', 'ψ', 'ω', 'ϵ', 'ϑ',
1936        'ϖ', 'ϱ', 'ς', 'ϕ', '↼', '↽', '⇀', '⇁', '↩', '↪', '▷', '◁', '0', '1', '2', '3', '4', '5',
1937        '6', '7', '8', '9', '.', ',', '<', '/', '>', '⋆', '∂', 'A', 'B', 'C', 'D', 'E', 'F', 'G',
1938        'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y',
1939        'Z', '♭', '♮', '♯', '⌣', '⌢', 'ℓ', 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k',
1940        'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', 'ı', 'ȷ', '℘',
1941        '\u{20d7}', '⁀',
1942    ];
1943    let name = base_font_name(fdict)?;
1944    let up = name.to_ascii_uppercase();
1945    let table: &[char; 128] = if up.starts_with(b"CMSY") || up.starts_with(b"CMBSY") {
1946        &CMSY
1947    } else if up.starts_with(b"CMMI") {
1948        &CMMI
1949    } else {
1950        return None;
1951    };
1952    Some(
1953        table
1954            .iter()
1955            .enumerate()
1956            .map(|(i, &c)| (i as u8, c))
1957            .collect(),
1958    )
1959}
1960
1961/// Resolve an Adobe glyph name to Unicode: `uniXXXX`, single ASCII letters, the
1962/// digit/punctuation names from the Adobe Glyph List, and common typographic
1963/// names. A `.suffix` (`one.taboldstyle`, `a.sc`) is stripped and the base name
1964/// retried — docling renders these as the base character.
1965pub(crate) fn glyph_name_to_char(name: &[u8]) -> Option<char> {
1966    let s = std::str::from_utf8(name).ok()?;
1967    if let Some(hex) = s.strip_prefix("uni") {
1968        if let Ok(cp) = u32::from_str_radix(hex.get(0..4)?, 16) {
1969            return char::from_u32(cp);
1970        }
1971    }
1972    // Single ASCII letter names (`A`, `m`) map to themselves.
1973    if s.len() == 1 {
1974        let b = s.as_bytes()[0];
1975        if b.is_ascii_alphabetic() {
1976            return Some(b as char);
1977        }
1978    }
1979    let resolved = match s {
1980        "space" => ' ',
1981        "exclam" => '!',
1982        "quotedbl" => '"',
1983        "numbersign" => '#',
1984        "dollar" => '$',
1985        "percent" => '%',
1986        "ampersand" => '&',
1987        "quotesingle" => '\'',
1988        "parenleft" => '(',
1989        "parenright" => ')',
1990        "asterisk" => '*',
1991        "plus" => '+',
1992        "comma" => ',',
1993        "hyphen" => '-',
1994        "period" => '.',
1995        "slash" => '/',
1996        "zero" => '0',
1997        "one" => '1',
1998        "two" => '2',
1999        "three" => '3',
2000        "four" => '4',
2001        "five" => '5',
2002        "six" => '6',
2003        "seven" => '7',
2004        "eight" => '8',
2005        "nine" => '9',
2006        "colon" => ':',
2007        "semicolon" => ';',
2008        "less" => '<',
2009        "equal" => '=',
2010        "greater" => '>',
2011        "question" => '?',
2012        "at" => '@',
2013        "bracketleft" => '[',
2014        "backslash" => '\\',
2015        "bracketright" => ']',
2016        "asciicircum" => '^',
2017        "underscore" => '_',
2018        "grave" => '`',
2019        "braceleft" => '{',
2020        "bar" => '|',
2021        "braceright" => '}',
2022        "asciitilde" => '~',
2023        "bullet" => '\u{2022}',
2024        "periodcentered" => '\u{00B7}',
2025        "endash" => '\u{2013}',
2026        "emdash" => '\u{2014}',
2027        "quoteright" => '\u{2019}',
2028        "quoteleft" => '\u{2018}',
2029        "quotedblleft" => '\u{201C}',
2030        "quotedblright" => '\u{201D}',
2031        "quotedblbase" => '\u{201E}',
2032        "quotesinglbase" => '\u{201A}',
2033        // Latin f-ligatures named in `/Differences` (e.g. 2305's body font). These
2034        // map to the presentation-form code points, which `decompose_ligatures`
2035        // then spells back out (`ff`→"ff") — without them the glyph decodes to
2036        // nothing and the sanitizer fills the gap with a space (`di erences`).
2037        "ff" => '\u{FB00}',
2038        "fi" => '\u{FB01}',
2039        "fl" => '\u{FB02}',
2040        "ffi" => '\u{FB03}',
2041        "ffl" => '\u{FB04}',
2042        "ft" => '\u{FB05}',
2043        "st" => '\u{FB06}',
2044        "degree" => '\u{00B0}',
2045        "trademark" => '\u{2122}',
2046        "registered" => '\u{00AE}',
2047        "copyright" => '\u{00A9}',
2048        "ellipsis" => '\u{2026}',
2049        "minus" => '\u{2212}',
2050        "fraction" => '\u{2044}',
2051        "nbspace" => '\u{00A0}',
2052        // Greek + math glyph names (standard Adobe Glyph List). Standard TeX math
2053        // fonts (CMMI/CMSY/…) name their glyphs this way in the embedded font
2054        // program's `/Encoding`; without these a `λ`/`≤` decodes to nothing and is
2055        // dropped from body text (`and λ set to 0.5` → `and set to 0.5`).
2056        "alpha" => '\u{03B1}',
2057        "beta" => '\u{03B2}',
2058        "gamma" => '\u{03B3}',
2059        "delta" => '\u{03B4}',
2060        "epsilon" | "epsilon1" => '\u{03B5}',
2061        "zeta" => '\u{03B6}',
2062        "eta" => '\u{03B7}',
2063        "theta" | "theta1" => '\u{03B8}',
2064        "iota" => '\u{03B9}',
2065        "kappa" => '\u{03BA}',
2066        "lambda" => '\u{03BB}',
2067        "mu" => '\u{03BC}',
2068        "nu" => '\u{03BD}',
2069        "xi" => '\u{03BE}',
2070        "omicron" => '\u{03BF}',
2071        "pi" | "pi1" => '\u{03C0}',
2072        "rho" | "rho1" => '\u{03C1}',
2073        "sigma" => '\u{03C3}',
2074        "sigma1" => '\u{03C2}',
2075        "tau" => '\u{03C4}',
2076        "upsilon" => '\u{03C5}',
2077        "phi" | "phi1" => '\u{03C6}',
2078        "chi" => '\u{03C7}',
2079        "psi" => '\u{03C8}',
2080        "omega" | "omega1" => '\u{03C9}',
2081        "Gamma" => '\u{0393}',
2082        "Delta" => '\u{0394}',
2083        "Theta" => '\u{0398}',
2084        "Lambda" => '\u{039B}',
2085        "Xi" => '\u{039E}',
2086        "Pi" => '\u{03A0}',
2087        "Sigma" => '\u{03A3}',
2088        "Upsilon" => '\u{03A5}',
2089        "Phi" => '\u{03A6}',
2090        "Psi" => '\u{03A8}',
2091        "Omega" => '\u{03A9}',
2092        "lessequal" => '\u{2264}',
2093        "greaterequal" => '\u{2265}',
2094        "notequal" => '\u{2260}',
2095        "approxequal" => '\u{2248}',
2096        "equivalence" => '\u{2261}',
2097        "element" => '\u{2208}',
2098        "plusminus" => '\u{00B1}',
2099        "multiply" => '\u{00D7}',
2100        "divide" => '\u{00F7}',
2101        "infinity" => '\u{221E}',
2102        "partialdiff" => '\u{2202}',
2103        "gradient" => '\u{2207}',
2104        "summation" => '\u{2211}',
2105        "product" => '\u{220F}',
2106        "integral" => '\u{222B}',
2107        "radical" => '\u{221A}',
2108        "proportional" => '\u{221D}',
2109        "arrowright" => '\u{2192}',
2110        "arrowleft" => '\u{2190}',
2111        "arrowup" => '\u{2191}',
2112        "arrowdown" => '\u{2193}',
2113        "arrowboth" => '\u{2194}',
2114        "arrowdblright" => '\u{21D2}',
2115        "logicaland" => '\u{2227}',
2116        "logicalor" => '\u{2228}',
2117        "intersection" => '\u{2229}',
2118        "union" => '\u{222A}',
2119        "similar" => '\u{223C}',
2120        "congruent" => '\u{2245}',
2121        "dotmath" => '\u{22C5}',
2122        "asteriskmath" => '\u{2217}',
2123        _ => {
2124            // Strip an AGL `.suffix` (oldstyle/small-cap variant) and retry.
2125            if let Some((base, _)) = s.split_once('.') {
2126                if !base.is_empty() {
2127                    return glyph_name_to_char(base.as_bytes());
2128                }
2129            }
2130            return None;
2131        }
2132    };
2133    Some(resolved)
2134}
2135
2136/// Minimal WinAnsiEncoding (Latin-1-ish) for simple fonts lacking ToUnicode.
2137fn winansi_table() -> HashMap<u8, char> {
2138    let mut m = HashMap::new();
2139    for b in 0x20u8..=0x7e {
2140        m.insert(b, b as char);
2141    }
2142    // High range: Windows-1252 printable points that differ from Latin-1.
2143    let extra: &[(u8, char)] = &[
2144        (0x91, '\u{2018}'),
2145        (0x92, '\u{2019}'),
2146        (0x93, '\u{201C}'),
2147        (0x94, '\u{201D}'),
2148        (0x95, '\u{2022}'),
2149        (0x96, '\u{2013}'),
2150        (0x97, '\u{2014}'),
2151        (0x85, '\u{2026}'),
2152        (0xA0, '\u{00A0}'),
2153    ];
2154    for &(b, c) in extra {
2155        m.insert(b, c);
2156    }
2157    for b in 0xA1u8..=0xFF {
2158        m.entry(b).or_insert(b as char);
2159    }
2160    m
2161}
2162
2163/// Minimal MacRomanEncoding: ASCII plus the high-range points our corpus hits
2164/// (notably 0xA5 = bullet, used as a list marker).
2165fn macroman_table() -> HashMap<u8, char> {
2166    let mut m = HashMap::new();
2167    for b in 0x20u8..=0x7e {
2168        m.insert(b, b as char);
2169    }
2170    let high: &[(u8, char)] = &[
2171        (0xA5, '\u{2022}'), // bullet
2172        (0xD0, '\u{2013}'), // endash
2173        (0xD1, '\u{2014}'), // emdash
2174        (0xD2, '\u{201C}'),
2175        (0xD3, '\u{201D}'),
2176        (0xD4, '\u{2018}'),
2177        (0xD5, '\u{2019}'),
2178        (0xCA, '\u{00A0}'),
2179        (0xC9, '\u{2026}'),
2180        (0xDE, '\u{FB01}'),
2181        (0xDF, '\u{FB02}'),
2182    ];
2183    for &(b, c) in high {
2184        m.insert(b, c);
2185    }
2186    m
2187}
2188
2189#[cfg(test)]
2190mod page_box_frame {
2191    use super::*;
2192
2193    /// One page, `boxes` spliced into the page dictionary verbatim, one text
2194    /// run at user-space `(x, y)`.
2195    fn pdf(boxes: &str, x: f32, y: f32) -> Vec<u8> {
2196        pdf_content(
2197            boxes,
2198            &format!("BT /F1 12 Tf {x} {y} Td (First printing) Tj ET\n"),
2199        )
2200    }
2201
2202    /// One page, `boxes` spliced into the page dictionary, `content` as its
2203    /// content stream (font `/F1` = Helvetica).
2204    fn pdf_content(boxes: &str, content: &str) -> Vec<u8> {
2205        let objs: Vec<String> = vec![
2206            "<</Type/Catalog/Pages 2 0 R>>".into(),
2207            format!("<</Type/Pages/Kids[3 0 R]/Count 1{boxes}>>"),
2208            "<</Type/Page/Parent 2 0 R/Contents 4 0 R/Resources<</Font<</F1 5 0 R>>>>>>".into(),
2209            format!("<</Length {}>>stream\n{content}endstream", content.len()),
2210            "<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".into(),
2211        ];
2212        let mut out = b"%PDF-1.4\n".to_vec();
2213        let mut offsets = Vec::new();
2214        for (i, body) in objs.iter().enumerate() {
2215            offsets.push(out.len());
2216            out.extend_from_slice(format!("{} 0 obj{body}endobj\n", i + 1).as_bytes());
2217        }
2218        let xref_at = out.len();
2219        out.extend_from_slice(
2220            format!("xref\n0 {}\n0000000000 65535 f \n", objs.len() + 1).as_bytes(),
2221        );
2222        for off in &offsets {
2223            out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2224        }
2225        out.extend_from_slice(
2226            format!(
2227                "trailer<</Size {}/Root 1 0 R>>\nstartxref\n{xref_at}\n%%EOF\n",
2228                objs.len() + 1
2229            )
2230            .as_bytes(),
2231        );
2232        out
2233    }
2234
2235    fn only_page(bytes: &[u8]) -> (PageBox, Vec<Glyph>) {
2236        let doc = load_document(bytes).expect("loads");
2237        let pid = *doc.get_pages().values().next().expect("one page");
2238        (page_box(&doc, pid), page_glyphs(&doc, pid))
2239    }
2240
2241    /// The trimmed-book-page shape: MediaBox and CropBox both away from the
2242    /// origin (inherited from `/Pages`). The display box is the CropBox, and a
2243    /// glyph's coordinates count from its lower-left corner — the same numbers
2244    /// the same text gets on a page whose CropBox *is* `[0 0 w h]`.
2245    #[test]
2246    fn glyphs_count_from_the_cropbox_corner_like_pdfium() {
2247        let (pb, shifted) = only_page(&pdf(
2248            "/MediaBox[-56.505 -58.25 576.303 723.31]/CropBox[1.095 -0.65 518.703 665.71]",
2249            37.0 + 1.095,
2250            58.0 - 0.65,
2251        ));
2252        assert!((pb.l - 1.095).abs() < 1e-3 && (pb.b + 0.65).abs() < 1e-3);
2253        assert!(
2254            (pb.w - 517.608).abs() < 1e-3 && (pb.h - 666.36).abs() < 1e-3,
2255            "{pb:?}"
2256        );
2257        let (pb0, plain) = only_page(&pdf("/MediaBox[0 0 517.608 666.36]", 37.0, 58.0));
2258        assert!((pb0.w - pb.w).abs() < 1e-3 && (pb0.h - pb.h).abs() < 1e-3);
2259        assert_eq!(shifted.len(), plain.len());
2260        assert!(!plain.is_empty());
2261        for (a, b) in shifted.iter().zip(&plain) {
2262            assert!(
2263                (a.l - b.l).abs() < 1e-3 && (a.b - b.b).abs() < 1e-3,
2264                "{:?} vs {:?}",
2265                (a.l, a.b),
2266                (b.l, b.b)
2267            );
2268        }
2269        assert!((plain[0].l - 37.0).abs() < 1e-3, "{}", plain[0].l);
2270    }
2271
2272    fn text(glyphs: &[Glyph]) -> String {
2273        glyphs.iter().map(|g| g.ch).collect()
2274    }
2275
2276    /// #529: text drawn outside the display box — a print slug beside the
2277    /// CropBox, a tiled page's neighbour beyond the MediaBox — is dropped,
2278    /// glyph by glyph like docling-parse, instead of being clamped onto the
2279    /// page edge; a line crossing the edge keeps the glyphs that fit.
2280    #[test]
2281    fn glyphs_outside_the_display_box_are_dropped() {
2282        let (_, g) = only_page(&pdf_content(
2283            "/MediaBox[0 0 400 400]/CropBox[100 100 400 400]",
2284            "BT /F1 14 Tf 120 300 Td (Visible.) Tj ET\nBT /F1 8 Tf 5 40 Td (Slug) Tj ET\n",
2285        ));
2286        assert_eq!(text(&g), "Visible.");
2287        let (_, g) = only_page(&pdf_content(
2288            "/MediaBox[0 0 300 400]",
2289            "BT /F1 14 Tf 40 300 Td (On page) Tj ET\nBT /F1 14 Tf 400 250 Td (Beyond) Tj ET\n",
2290        ));
2291        assert_eq!(text(&g), "On page");
2292        // docling-parse: "Crossing the right edge" on a 300 pt page → "Crossing the rig".
2293        let (_, g) = only_page(&pdf_content(
2294            "/MediaBox[0 0 300 400]",
2295            "BT /F1 14 Tf 200 250 Td (Crossing the right edge) Tj ET\n",
2296        ));
2297        assert_eq!(text(&g), "Crossing the rig");
2298    }
2299
2300    /// The containment test is docling-parse's: the whole char box (advance
2301    /// × the font's ascent / descent) inside the display box, edges
2302    /// included — a Helvetica `W` at 50 pt (47.2 wide) ending exactly on the
2303    /// right edge stays, 0.01 pt further it goes; likewise on the left.
2304    #[test]
2305    fn a_glyph_on_the_edge_stays_one_over_it_goes() {
2306        let w_at = |x: f32, y: f32| {
2307            let (_, g) = only_page(&pdf_content(
2308                "/MediaBox[0 0 300 400]",
2309                &format!("BT /F1 50 Tf {x} {y} Td (W) Tj ET\n"),
2310            ));
2311            text(&g)
2312        };
2313        assert_eq!(w_at(252.8, 200.0), "W");
2314        assert_eq!(w_at(252.81, 200.0), "");
2315        assert_eq!(w_at(0.0, 200.0), "W");
2316        assert_eq!(w_at(-0.01, 200.0), "");
2317        // Vertically the box is the font's ascent / descent: a baseline far
2318        // enough up keeps the descender on the page, one at 0 does not.
2319        assert_eq!(w_at(100.0, 20.0), "W");
2320        assert_eq!(w_at(100.0, 0.0), "");
2321        assert_eq!(w_at(100.0, 380.0), "");
2322    }
2323
2324    /// pdfium's fallbacks: no MediaBox → Letter; a CropBox is clipped to the
2325    /// MediaBox, and one that misses it entirely is ignored.
2326    #[test]
2327    fn page_box_follows_pdfium_fallbacks() {
2328        let (pb, _) = only_page(&pdf("", 10.0, 10.0));
2329        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 612.0, 792.0));
2330        let (pb, _) = only_page(&pdf(
2331            "/MediaBox[0 0 500 700]/CropBox[-100 100 600 900]",
2332            10.0,
2333            10.0,
2334        ));
2335        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 100.0, 500.0, 600.0));
2336        let (pb, _) = only_page(&pdf(
2337            "/MediaBox[0 0 500 700]/CropBox[800 800 900 900]",
2338            10.0,
2339            10.0,
2340        ));
2341        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 500.0, 700.0));
2342        // Reversed corners normalize.
2343        let (pb, _) = only_page(&pdf("/MediaBox[500 700 0 0]", 10.0, 10.0));
2344        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 500.0, 700.0));
2345    }
2346}
2347
2348#[cfg(test)]
2349mod xref_repair {
2350    /// Build a tiny one-page PDF whose cross-reference entries are either the
2351    /// spec's 20 bytes (`two_byte_eol`) or the 19-byte form some generators
2352    /// emit — everything else about the two files is identical.
2353    fn pdf_with_xref(two_byte_eol: bool) -> Vec<u8> {
2354        let content = b"BT /F1 12 Tf 72 700 Td (Invoice 922769430725) Tj ET\n";
2355        let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2356        let objs: Vec<Vec<u8>> = vec![
2357            b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2358            b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2359            b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2360               /Resources<</Font<</F1 5 0 R>>>>>>"
2361                .to_vec(),
2362            [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
2363            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".to_vec(),
2364        ];
2365
2366        let mut out = b"%PDF-1.4\n".to_vec();
2367        let mut offsets = Vec::new();
2368        for (i, body) in objs.iter().enumerate() {
2369            offsets.push(out.len());
2370            out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2371            out.extend_from_slice(body);
2372            out.extend_from_slice(b"endobj\n");
2373        }
2374        let xref_at = out.len();
2375        let eol: &[u8] = if two_byte_eol { b" \n" } else { b"\n" };
2376        out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2377        out.extend_from_slice(b"0000000000 65535 f");
2378        out.extend_from_slice(eol);
2379        for off in &offsets {
2380            out.extend_from_slice(format!("{off:010} 00000 n").as_bytes());
2381            out.extend_from_slice(eol);
2382        }
2383        out.extend_from_slice(
2384            format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2385        );
2386        out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2387        out
2388    }
2389
2390    /// A 19-byte cross-reference entry (a bare LF where the spec wants a
2391    /// two-byte EOL) makes lopdf reject the whole file, so a readable text
2392    /// layer used to look exactly like a scan — in the browser that meant ten
2393    /// seconds of OCR for nothing. The repair must recover *the same* parse the
2394    /// well-formed file gives.
2395    #[test]
2396    fn short_xref_entries_still_parse() {
2397        let good = pdf_with_xref(true);
2398        let broken = pdf_with_xref(false);
2399        assert!(
2400            broken.len() < good.len(),
2401            "the broken file is the shorter one"
2402        );
2403        assert!(
2404            lopdf::Document::load_mem(&good).is_ok(),
2405            "the control file must load unaided"
2406        );
2407        assert!(
2408            lopdf::Document::load_mem(&broken).is_err(),
2409            "lopdf rejects 19-byte entries — if this ever passes, drop the repair"
2410        );
2411
2412        let cells = |b: &[u8]| -> Vec<String> {
2413            super::pdf_textlines(b)
2414                .into_iter()
2415                .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2416                .collect()
2417        };
2418        let from_good = cells(&good);
2419        assert!(
2420            from_good.iter().any(|t| t.contains("922769430725")),
2421            "control text: {from_good:?}"
2422        );
2423        assert_eq!(
2424            cells(&broken),
2425            from_good,
2426            "repair must match the good parse"
2427        );
2428    }
2429
2430    /// The same generator overstates `/Length`, so lopdf reads past the data,
2431    /// misses `endstream` and drops the stream — the object comes back as a
2432    /// bare dictionary and the page has no content at all. Trust `endstream`
2433    /// instead, and do it without moving a single byte.
2434    #[test]
2435    fn overstated_stream_length_still_yields_content() {
2436        let good = pdf_with_xref(true);
2437        // Inflate the content stream's /Length by one, exactly as the invoice
2438        // that prompted this does.
2439        let broken = {
2440            let at = good
2441                .windows(8)
2442                .position(|w| w == b"/Length ")
2443                .expect("a /Length")
2444                + 8;
2445            let digits = good[at..].iter().take_while(|c| c.is_ascii_digit()).count();
2446            let n: usize = std::str::from_utf8(&good[at..at + digits])
2447                .unwrap()
2448                .parse()
2449                .unwrap();
2450            let inflated = (n + 1).to_string();
2451            assert_eq!(inflated.len(), digits, "keep the digit count");
2452            let mut b = good.clone();
2453            b[at..at + digits].copy_from_slice(inflated.as_bytes());
2454            b
2455        };
2456        assert_eq!(broken.len(), good.len(), "the defect must not move bytes");
2457        // lopdf alone loses the stream: the page parses but carries no content.
2458        let raw = lopdf::Document::load_mem(&broken).expect("still loads");
2459        assert!(
2460            raw.get_pages()
2461                .into_values()
2462                .all(|p| raw.get_page_content(p).is_empty()),
2463            "lopdf should drop the stream — if it stops, drop this repair"
2464        );
2465        // Ours recovers the same text the well-formed file gives.
2466        let text = |b: &[u8]| -> Vec<String> {
2467            super::pdf_textlines(b)
2468                .into_iter()
2469                .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2470                .collect()
2471        };
2472        let expected = text(&good);
2473        assert!(!expected.is_empty(), "control must produce text");
2474        assert_eq!(text(&broken), expected);
2475    }
2476
2477    /// The repair only fires where padding cannot move an object: it declines a
2478    /// file whose xref precedes an object (an incremental update), rather than
2479    /// shifting every offset the table records.
2480    #[test]
2481    fn repair_declines_when_padding_would_move_objects() {
2482        let mut incremental = pdf_with_xref(false);
2483        incremental.extend_from_slice(b"6 0 obj<</Type/Whatever>>endobj\n");
2484        let declined = super::pad_short_xref_entries(&incremental).unwrap_err();
2485        assert!(
2486            declined.contains("object follows the xref"),
2487            "reason: {declined}"
2488        );
2489    }
2490}
2491
2492/// #187: standard-14 fonts referenced without an embedded program (and thus
2493/// usually without `/Widths` or a `/FontDescriptor`) must decode with the
2494/// built-in Adobe Core 14 metrics instead of collapsing every cell to zero
2495/// width — the failure mode where a valid text layer was silently dropped
2496/// while pdfium read the same file fine.
2497#[cfg(test)]
2498mod base14_fonts {
2499    /// A one-page PDF whose single `Tj` uses `fontdict` (no embedded program).
2500    fn pdf_with_font(fontdict: &[u8], text: &[u8]) -> Vec<u8> {
2501        let content = [b"BT /F1 12 Tf 72 700 Td (".as_slice(), text, b") Tj ET\n"].concat();
2502        pdf_with_content(fontdict, &content)
2503    }
2504
2505    /// A one-page A4 PDF drawing `content` with `fontdict` as `/F1`.
2506    fn pdf_with_content(fontdict: &[u8], content: &[u8]) -> Vec<u8> {
2507        let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2508        let objs: Vec<Vec<u8>> = vec![
2509            b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2510            b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2511            b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2512               /Resources<</Font<</F1 5 0 R>>>>>>"
2513                .to_vec(),
2514            [stream.as_slice(), content, b"endstream"].concat(),
2515            fontdict.to_vec(),
2516        ];
2517        let mut out = b"%PDF-1.4\n".to_vec();
2518        let mut offsets = Vec::new();
2519        for (i, body) in objs.iter().enumerate() {
2520            offsets.push(out.len());
2521            out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2522            out.extend_from_slice(body);
2523            out.extend_from_slice(b"endobj\n");
2524        }
2525        let xref_at = out.len();
2526        out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2527        out.extend_from_slice(b"0000000000 65535 f \n");
2528        for off in &offsets {
2529            out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2530        }
2531        out.extend_from_slice(
2532            format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2533        );
2534        out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2535        out
2536    }
2537
2538    /// The parsed cells of the only page.
2539    fn cells(pdf: &[u8]) -> Vec<crate::pdfium_backend::TextCell> {
2540        super::pdf_textlines(pdf)
2541            .into_iter()
2542            .flat_map(|(_, _, c)| c)
2543            .collect()
2544    }
2545
2546    /// Text drawn with a rotated text matrix reads as one line in its own
2547    /// reading order, boxed by the run's axis-aligned extent (#528). The
2548    /// landscape case is `0 s -s 0 tx ty Tm` on a `/Rotate 90` page; every
2549    /// such run used to collapse to zero width and vanish (90°/270°/tilted)
2550    /// or read backwards (180°). Expected boxes are docling-parse 7.22's line
2551    /// cells for the same runs (y-up, PDF points).
2552    #[test]
2553    fn rotated_text_matrix_reads_in_order() {
2554        const HELV: &[u8] = b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>";
2555        const TEXT: &str = "Upright text on a rotated page.";
2556        for (tm, [l, b, r, t]) in [
2557            ("0 14 -14 0 200 100", [189.9, 100.0, 202.9, 289.1]), // 90°
2558            ("-14 0 0 -14 350 300", [160.9, 289.9, 350.0, 302.9]), // 180°
2559            ("0 -14 14 0 200 500", [197.1, 310.9, 210.1, 500.0]), // 270°
2560            (
2561                "9.8995 9.8995 -9.8995 9.8995 100 100",
2562                [92.9, 98.0, 235.8, 240.8],
2563            ), // 45°
2564        ] {
2565            let content = format!("BT /F1 1 Tf {tm} Tm ({TEXT}) Tj ET\n");
2566            let pdf = pdf_with_content(HELV, content.as_bytes());
2567            let cs = cells(&pdf);
2568            assert_eq!(cs.len(), 1, "{tm}: {cs:?}");
2569            let c = &cs[0];
2570            assert_eq!(c.text, TEXT, "{tm}");
2571            // `cells` is top-left-origin on the 842 pt page. The descriptor-
2572            // less base-14 face gets the parser's 1-em box where docling-parse
2573            // reads Helvetica's AFM (718/−207) — the same ≤ 0.6 pt
2574            // across-the-baseline offset upright text has — so allow 1 pt.
2575            let got = [c.l, 842.0 - c.b, c.r, 842.0 - c.t];
2576            for (g, w) in got.iter().zip([l, b, r, t]) {
2577                assert!(
2578                    (g - w).abs() < 1.0,
2579                    "{tm}: box {got:?} vs docling-parse {:?}",
2580                    [l, b, r, t]
2581                );
2582            }
2583            let words: Vec<String> = super::pdf_words(&pdf)
2584                .into_iter()
2585                .flat_map(|(_, _, c)| c)
2586                .map(|c| c.text)
2587                .collect();
2588            assert_eq!(words, TEXT.split(' ').collect::<Vec<_>>(), "{tm}");
2589        }
2590    }
2591
2592    /// A rotated run past the page edge (a cover's spine title) is dropped
2593    /// by the on-page test (#529) — measured on the quad's extent, not the
2594    /// zero-width box rotated glyphs used to get — instead of clamping onto
2595    /// the edge as a zero-width cell. One inside the page is kept.
2596    #[test]
2597    fn rotated_text_off_the_page_is_dropped() {
2598        const HELV: &[u8] = b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>";
2599        let off = pdf_with_content(
2600            HELV,
2601            b"BT /F1 1 Tf 0 14 -14 0 -4 200 Tm (Spine title) Tj ET\n",
2602        );
2603        assert!(cells(&off).is_empty(), "{:?}", cells(&off));
2604        let on = pdf_with_content(
2605            HELV,
2606            b"BT /F1 1 Tf 0 14 -14 0 30 200 Tm (Margin note) Tj ET\n",
2607        );
2608        assert_eq!(cells(&on).len(), 1);
2609    }
2610
2611    /// Upright text keeps the plain loose rectangle — no quad, and the
2612    /// sanitizer path is the one every pinned PDF baseline was made with.
2613    #[test]
2614    fn upright_glyphs_carry_no_quad() {
2615        let only_page = |pdf: &[u8]| {
2616            let doc = super::load_document(pdf).expect("loads");
2617            let pid = *doc.get_pages().values().next().expect("one page");
2618            super::page_glyphs(&doc, pid)
2619        };
2620        let pdf = pdf_with_font(b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>", b"Plain");
2621        let glyphs = only_page(&pdf);
2622        assert!(!glyphs.is_empty());
2623        assert!(glyphs.iter().all(|g| g.quad.is_none() && g.lr > g.ll));
2624        // A 90° glyph's style height is its font size, not its advance.
2625        let pdf = pdf_with_content(
2626            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>",
2627            b"BT /F1 1 Tf 0 14 -14 0 200 100 Tm (W) Tj ET\n",
2628        );
2629        let glyphs = only_page(&pdf);
2630        let g = &glyphs[0];
2631        assert!(g.quad.is_some());
2632        // 14 pt × the face's 1-em box (no descriptor: ascent − descent = 1).
2633        assert!((g.height() - 14.0).abs() < 0.05, "{}", g.height());
2634    }
2635
2636    /// Every standard-14 alias/style decodes with real (positive-width) boxes.
2637    #[test]
2638    fn standard14_faces_get_builtin_widths() {
2639        for fontdict in [
2640            // ReportLab's default: base-14 Helvetica, WinAnsi, nothing else.
2641            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>".as_slice(),
2642            // No /Encoding at all (StandardEncoding-ish default).
2643            b"<</Type/Font/Subtype/Type1/BaseFont/Times-BoldItalic>>",
2644            // Substitution aliases + a subset prefix.
2645            b"<</Type/Font/Subtype/TrueType/BaseFont/Arial,Bold>>",
2646            b"<</Type/Font/Subtype/Type1/BaseFont/ABCDEF+Courier-Oblique>>",
2647        ] {
2648            let pdf = pdf_with_font(fontdict, b"Words have width now");
2649            let cs = cells(&pdf);
2650            let text: String = cs
2651                .iter()
2652                .map(|c| c.text.as_str())
2653                .collect::<Vec<_>>()
2654                .join(" ");
2655            assert!(
2656                text.contains("Words have width now"),
2657                "{}: text lost: {text:?}",
2658                String::from_utf8_lossy(fontdict)
2659            );
2660            assert!(
2661                cs.iter().all(|c| c.r > c.l),
2662                "{}: zero-width cells: {cs:?}",
2663                String::from_utf8_lossy(fontdict)
2664            );
2665        }
2666    }
2667
2668    /// An explicit `/Widths` array always wins over the built-in metrics, and a
2669    /// non-standard face without `/Widths` stays as before (no invented boxes).
2670    #[test]
2671    fn explicit_widths_win_and_unknown_faces_are_untouched() {
2672        // Helvetica with explicit 100/1000-em widths: the word's box must be
2673        // ~4×100 units at 12pt = 4.8pt wide — far narrower than the ~2.7×
2674        // wider built-in Helvetica advances would make it.
2675        let explicit = pdf_with_font(
2676            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/FirstChar 65\
2677               /Widths[100 100 100 100]/Encoding/WinAnsiEncoding>>",
2678            b"ABBA",
2679        );
2680        let builtin = pdf_with_font(
2681            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>",
2682            b"ABBA",
2683        );
2684        let w = |pdf: &[u8]| {
2685            let cs = cells(pdf);
2686            assert_eq!(cs.len(), 1, "one word cell: {cs:?}");
2687            cs[0].r - cs[0].l
2688        };
2689        let (we, wb) = (w(&explicit), w(&builtin));
2690        assert!(
2691            (we - 4.8).abs() < 0.1,
2692            "explicit widths must win: got {we}, want 4×100×12/1000"
2693        );
2694        assert!(
2695            wb > 2.0 * we,
2696            "built-in Helvetica is much wider: {wb} vs {we}"
2697        );
2698
2699        // An unknown face with no /Widths: still parses (text kept), but no
2700        // built-in table applies — the old zero-width behavior is preserved
2701        // rather than inventing Helvetica metrics for an arbitrary font.
2702        let unknown = pdf_with_font(
2703            b"<</Type/Font/Subtype/Type1/BaseFont/FancyCorp-Display>>",
2704            b"Mystery",
2705        );
2706        let cs = cells(&unknown);
2707        let text: String = cs.iter().map(|c| c.text.as_str()).collect();
2708        assert!(text.contains("Mystery"), "text still decodes: {cs:?}");
2709    }
2710}
2711
2712#[cfg(test)]
2713mod overpainted {
2714    use crate::pdfium_backend::TextCell;
2715
2716    fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
2717        TextCell {
2718            text: text.into(),
2719            l,
2720            t,
2721            r,
2722            b,
2723        }
2724    }
2725
2726    /// The reporting invoice's logo: a `"` painted inside a `==` on one band —
2727    /// artwork drawn with glyphs. Both cells go; the real text on the next
2728    /// band stays.
2729    #[test]
2730    fn stacked_logo_glyphs_are_dropped() {
2731        let mut cells = vec![
2732            cell("\"", 72.7, 21.5, 86.4, 31.5),
2733            cell("==", 59.4, 21.5, 99.6, 31.5),
2734            cell("Herr", 65.2, 151.3, 81.7, 161.3),
2735        ];
2736        super::drop_overpainted_cells(&mut cells);
2737        assert_eq!(cells.len(), 1, "cells: {cells:?}");
2738        assert_eq!(cells[0].text, "Herr");
2739    }
2740
2741    /// Adjacent words on a line touch but never contain each other — prose is
2742    /// untouched, and so is a same-text near-duplicate (double-drawn faux
2743    /// bold), which is not evidence of artwork.
2744    #[test]
2745    fn prose_and_double_draw_are_kept() {
2746        let mut cells = vec![
2747            cell("Telefon", 354.3, 133.2, 381.5, 143.2),
2748            cell("0676/2000", 387.3, 133.2, 428.7, 143.2),
2749            cell("Bold", 100.0, 50.0, 130.0, 60.0),
2750            cell("Bold", 100.3, 50.0, 130.3, 60.0),
2751        ];
2752        super::drop_overpainted_cells(&mut cells);
2753        assert_eq!(cells.len(), 4);
2754    }
2755}
2756
2757#[cfg(test)]
2758mod vestigial_layer {
2759    use crate::pdfium_backend::{PdfPage, TextCell};
2760
2761    fn page_with(texts: &[&str]) -> PdfPage {
2762        let cells = texts
2763            .iter()
2764            .enumerate()
2765            .map(|(i, t)| TextCell {
2766                text: t.to_string(),
2767                l: 10.0,
2768                t: 10.0 + 12.0 * i as f32,
2769                r: 90.0,
2770                b: 20.0 + 12.0 * i as f32,
2771            })
2772            .collect();
2773        PdfPage::from_cells(595.0, 842.0, 1.0, cells)
2774    }
2775
2776    /// The reported scanned form: three typed-in field values ("03", "05",
2777    /// "2025") over three image pages. That must read as *no usable layer*,
2778    /// so the browser routes the document to OCR instead of extracting
2779    /// thirteen characters and skipping the letter entirely.
2780    #[test]
2781    fn typed_in_form_fields_are_not_a_text_layer() {
2782        let pages = vec![
2783            page_with(&["03", "05", "2025"]),
2784            page_with(&[]),
2785            page_with(&[]),
2786        ];
2787        assert!(super::text_layer_is_vestigial(&pages));
2788        assert!(super::text_layer_is_vestigial(&[page_with(&[])]));
2789    }
2790
2791    /// A short but genuine digital document — one page, a few real lines —
2792    /// keeps the fast text path.
2793    #[test]
2794    fn sparse_but_real_documents_pass() {
2795        let one_pager = vec![page_with(&[
2796            "Confidential briefing",
2797            "Prepared for the board meeting",
2798            "Do not distribute",
2799        ])];
2800        assert!(!super::text_layer_is_vestigial(&one_pager));
2801    }
2802}