Skip to main content

docling_pdf/
textparse.rs

1//! Pure-Rust PDF text extraction (replacing pdfium's glyph layer).
2//!
3//! pdfium reports *rendered* glyph boxes, which diverge from docling's
4//! `docling-parse` C++ parser at exactly the points that drive conformance:
5//! generated spaces get a zero-width box, combining diacritics get a real-width
6//! box, and ligature/fraction glyphs land at different x. This module instead
7//! reconstructs each glyph's box from the **font's own advance widths** and the
8//! PDF text/graphics matrices — the same information docling-parse uses — so a
9//! space is as wide as the font says and a combining mark has zero advance.
10//!
11//! The output is the same [`Glyph`] stream pdfium produces (native PDF
12//! coordinates, y-up), fed straight into the existing docling-parse line
13//! sanitizer ([`crate::dp_lines`]). Only the digital text layer is handled here;
14//! pages without one still fall back to OCR upstream.
15
16use std::collections::HashMap;
17use std::sync::Arc;
18
19use lopdf::{Dictionary, Document, Object};
20
21use crate::pdfium_backend::Glyph;
22
23/// Per-document caches for the content-stream interpreter. Fonts are indirect
24/// objects shared by many pages, but were fully re-parsed — ToUnicode CMap
25/// decompression + tokenization, embedded Type1 program scan, width tables —
26/// for **every page and every Form XObject invocation**; decoded form content
27/// streams were likewise re-inflated on every `Do`. Cached per document,
28/// keyed by the referenced object id (fonts also by resource name, which
29/// feeds the docling-parse font hash). Inline (non-reference) dicts are rare
30/// and stay uncached.
31#[derive(Default)]
32struct DocCaches {
33    fonts: HashMap<(lopdf::ObjectId, Vec<u8>), Arc<Font>>,
34    forms: HashMap<lopdf::ObjectId, Arc<lopdf::content::Content>>,
35}
36
37/// A 2×3 affine matrix `[a b c d e f]`: maps `(x,y)` → `(a·x+c·y+e, b·x+d·y+f)`.
38#[derive(Clone, Copy)]
39struct Mat {
40    a: f64,
41    b: f64,
42    c: f64,
43    d: f64,
44    e: f64,
45    f: f64,
46}
47
48impl Mat {
49    const ID: Mat = Mat {
50        a: 1.0,
51        b: 0.0,
52        c: 0.0,
53        d: 1.0,
54        e: 0.0,
55        f: 0.0,
56    };
57
58    /// `self ∘ m`: the matrix that applies `self` first, then `m`.
59    fn then(self, m: Mat) -> Mat {
60        Mat {
61            a: self.a * m.a + self.b * m.c,
62            b: self.a * m.b + self.b * m.d,
63            c: self.c * m.a + self.d * m.c,
64            d: self.c * m.b + self.d * m.d,
65            e: self.e * m.a + self.f * m.c + m.e,
66            f: self.e * m.b + self.f * m.d + m.f,
67        }
68    }
69
70    fn apply(self, x: f64, y: f64) -> (f64, f64) {
71        (
72            self.a * x + self.c * y + self.e,
73            self.b * x + self.d * y + self.f,
74        )
75    }
76}
77
78/// A parsed font: how to turn raw string bytes into (unicode, advance) pairs.
79struct Font {
80    /// 2-byte codes (Type0 / Identity-H) vs 1-byte (simple fonts).
81    two_byte: bool,
82    /// code → Unicode string (from ToUnicode; may be multi-char, e.g. ligatures).
83    to_unicode: HashMap<u32, String>,
84    /// code → glyph advance, in 1000-unit glyph space.
85    widths: HashMap<u32, f64>,
86    default_width: f64,
87    /// 1-byte fallback decoding when ToUnicode lacks a code (WinAnsi-ish).
88    simple_encoding: Option<HashMap<u8, char>>,
89    /// code → raw `/Differences` glyph name, for GID-style names (`g115`) that
90    /// have no Unicode mapping. docling-parse emits these verbatim as `/g115`
91    /// (see the redp5110 bulleted list); matching it keeps no text skipped.
92    fallback_names: HashMap<u8, String>,
93    /// code → char from the embedded Type1 font program's own `/Encoding` vector,
94    /// used only as a last resort for glyphs the base encoding leaves unmapped
95    /// (standard TeX math fonts: `λ`, `≤`, …).
96    program_encoding: HashMap<u8, char>,
97    ascent: f64,
98    descent: f64,
99    hash: u64,
100    /// Weight and slant read from the `/BaseFont` name (the heading
101    /// hierarchy's style signal, #302).
102    style: crate::font_style::FontStyle,
103}
104
105impl Font {
106    fn decode_code(&self, code: u32) -> (Option<String>, f64) {
107        let w = self
108            .widths
109            .get(&code)
110            .copied()
111            .unwrap_or(self.default_width);
112        if let Some(s) = self.to_unicode.get(&code) {
113            return (Some(decompose_ligatures(s)), w);
114        }
115        if !self.two_byte {
116            // A GID-style `/Differences` name (no Unicode) overrides the base
117            // encoding, matching docling's verbatim `/g115` fallback.
118            if let Some(name) = self.fallback_names.get(&(code as u8)) {
119                return (Some(format!("/{name}")), w);
120            }
121            if let Some(enc) = &self.simple_encoding {
122                if let Some(&ch) = enc.get(&(code as u8)) {
123                    return (Some(decompose_ligatures(&ch.to_string())), w);
124                }
125            }
126            // Last resort: the embedded Type1 font program's own `/Encoding`
127            // vector (`dup N /glyphname put`). Standard TeX math fonts (CMMI, CMSY,
128            // …) ship no PDF `/Encoding` and no ToUnicode, so a glyph like `λ`
129            // (CMMI code 21 → `/lambda`) or `≤` (CMSY code 20 → `/lessequal`) has
130            // no other mapping and would otherwise be silently dropped. docling
131            // recovers these from the same font program. This only fills codes the
132            // base encoding left unmapped, so it never changes an existing decode.
133            if let Some(&ch) = self.program_encoding.get(&(code as u8)) {
134                return (Some(decompose_ligatures(&ch.to_string())), w);
135            }
136        }
137        (None, w)
138    }
139}
140
141/// Spell out Latin presentation-form ligatures (`fi`→`fi`, `ffi`→`ffi`, …) the way
142/// docling does, so `configuration`/`difficult` don't keep the ligature glyph.
143/// The chars share the ligature's box, so the line sanitizer recomposes them.
144fn decompose_ligatures(s: &str) -> String {
145    if !s.chars().any(|c| ('\u{FB00}'..='\u{FB06}').contains(&c)) {
146        return s.to_string();
147    }
148    s.chars()
149        .map(|c| {
150            match c {
151                '\u{FB00}' => "ff",
152                '\u{FB01}' => "fi",
153                '\u{FB02}' => "fl",
154                '\u{FB03}' => "ffi",
155                '\u{FB04}' => "ffl",
156                '\u{FB05}' => "ft",
157                '\u{FB06}' => "st",
158                _ => return c.to_string(),
159            }
160            .to_string()
161        })
162        .collect()
163}
164
165fn hash_name(name: &[u8]) -> u64 {
166    use std::hash::{Hash, Hasher};
167    let mut h = std::collections::hash_map::DefaultHasher::new();
168    name.hash(&mut h);
169    h.finish()
170}
171
172/// Resolve a possibly-indirect object to a dictionary.
173fn as_dict<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Dictionary> {
174    match obj {
175        Object::Dictionary(d) => Some(d),
176        Object::Reference(id) => doc.get_object(*id).ok().and_then(|o| o.as_dict().ok()),
177        _ => None,
178    }
179}
180
181fn deref<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Object> {
182    match obj {
183        Object::Reference(id) => doc.get_object(*id).ok(),
184        other => Some(other),
185    }
186}
187
188/// Parse one font dictionary into a [`Font`].
189fn parse_font(doc: &Document, name: &[u8], fdict: &Dictionary) -> Font {
190    let subtype: &[u8] = fdict
191        .get(b"Subtype")
192        .ok()
193        .and_then(|o| o.as_name().ok())
194        .unwrap_or(&[]);
195    let two_byte = subtype == b"Type0".as_slice();
196
197    let to_unicode = fdict
198        .get(b"ToUnicode")
199        .ok()
200        .and_then(|o| deref(doc, o))
201        .and_then(|o| o.as_stream().ok())
202        .and_then(|s| s.decompressed_content().ok())
203        .map(|data| parse_tounicode(&data))
204        .unwrap_or_default();
205
206    let (mut widths, mut default_width) = if two_byte {
207        cid_widths(doc, fdict)
208    } else {
209        simple_widths(doc, fdict)
210    };
211
212    let simple_encoding = if two_byte {
213        None
214    } else {
215        Some(simple_encoding_table(doc, fdict))
216    };
217
218    // A standard-14 font referenced without an embedded program usually ships
219    // no `/Widths` and no `/FontDescriptor` either (ReportLab's default, #187).
220    // Every advance then resolved to 0, the cells collapsed to zero width, and
221    // the page's whole text layer was silently dropped — while pdfium, with
222    // its built-in metrics, reads the same file fine. Fill the widths from the
223    // built-in Adobe Core 14 AFM tables via the font's own code→char decode
224    // (base encoding + `/Differences`), so an explicit `/Widths` always wins.
225    if !two_byte && widths.is_empty() && default_width == 0.0 {
226        if let Some(std14) = base_font_name(fdict).and_then(|n| crate::std14::widths_for(&n)) {
227            if let Some(enc) = &simple_encoding {
228                for (&code, &ch) in enc {
229                    if let Some(w) = std14.width(ch) {
230                        widths.insert(u32::from(code), w);
231                    }
232                }
233            }
234            // Codes the table misses still advance a typical width instead of
235            // stacking at x=0 (the failure mode this whole branch fixes).
236            default_width = 500.0;
237        }
238    }
239    let fallback_names = if two_byte {
240        HashMap::new()
241    } else {
242        differences_gid_names(doc, fdict)
243    };
244    let program_encoding = if two_byte {
245        HashMap::new()
246    } else {
247        type1_program_encoding(doc, fdict)
248    };
249
250    let (ascent, descent) = font_ascent_descent(doc, fdict, two_byte);
251
252    Font {
253        two_byte,
254        to_unicode,
255        widths,
256        default_width,
257        simple_encoding,
258        fallback_names,
259        program_encoding,
260        ascent,
261        descent,
262        hash: hash_name(name),
263        style: crate::font_style::parse_font_style(&String::from_utf8_lossy(
264            &base_font_name(fdict).unwrap_or_else(|| name.to_vec()),
265        )),
266    }
267}
268
269/// Collect `/Differences` entries whose glyph name is a GID placeholder
270/// (`g115`, `cid42`, `glyph7`, `index9`) with no Unicode mapping. docling-parse
271/// emits such glyphs as the literal name `/g115`; mapping them here keeps the
272/// text from being silently dropped (subsetted fonts with no ToUnicode). The
273/// GID-name restriction keeps real Adobe glyph names on the normal path so this
274/// never invents garbage on the clean files.
275fn differences_gid_names(doc: &Document, fdict: &Dictionary) -> HashMap<u8, String> {
276    let mut map = HashMap::new();
277    let Some(Object::Dictionary(enc)) = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o))
278    else {
279        return map;
280    };
281    let Some(Object::Array(diffs)) = enc.get(b"Differences").ok().and_then(|o| deref(doc, o))
282    else {
283        return map;
284    };
285    let mut code = 0u8;
286    for el in diffs {
287        match el {
288            Object::Integer(i) => code = *i as u8,
289            Object::Name(name) => {
290                if glyph_name_to_char(name).is_none() && is_gid_name(name) {
291                    map.insert(code, String::from_utf8_lossy(name).into_owned());
292                }
293                code = code.wrapping_add(1);
294            }
295            _ => {}
296        }
297    }
298    map
299}
300
301/// Parse the embedded Type1 font program's built-in `/Encoding` vector
302/// (`dup <code> /<glyphname> put` entries in the clear-text header before
303/// `eexec`) into `code → char`. This is how docling recovers glyphs from
304/// standard TeX math fonts (CMMI/CMSY/…) that carry no PDF `/Encoding` and no
305/// ToUnicode — e.g. CMMI's `dup 21 /lambda` or CMSY's `dup 20 /lessequal`.
306/// Only `FontFile` (Type1) is parsed; CFF (`FontFile3`) and TrueType
307/// (`FontFile2`) store their encoding in a binary table and are left alone.
308fn type1_program_encoding(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
309    let mut map = HashMap::new();
310    let Some(desc) = fdict
311        .get(b"FontDescriptor")
312        .ok()
313        .and_then(|o| deref(doc, o))
314        .and_then(|o| o.as_dict().ok())
315    else {
316        return map;
317    };
318    let Some(data) = desc
319        .get(b"FontFile")
320        .ok()
321        .and_then(|o| deref(doc, o))
322        .and_then(|o| o.as_stream().ok())
323        .and_then(|s| s.decompressed_content().ok())
324    else {
325        return map;
326    };
327    // The clear-text header (PostScript) ends at `eexec`; the rest is encrypted.
328    let head_end = data
329        .windows(5)
330        .position(|w| w == b"eexec")
331        .unwrap_or(data.len());
332    let head = String::from_utf8_lossy(&data[..head_end]);
333    // Scan for `dup <code> /<name> put` tokens.
334    let toks: Vec<&str> = head.split_whitespace().collect();
335    for w in toks.windows(4) {
336        if w[0] == "dup" && w[3] == "put" {
337            if let (Ok(code), Some(name)) = (w[1].parse::<u32>(), w[2].strip_prefix('/')) {
338                if code <= 255 {
339                    if let Some(ch) = glyph_name_to_char(name.as_bytes()) {
340                        map.insert(code as u8, ch);
341                    }
342                }
343            }
344        }
345    }
346    map
347}
348
349/// A glyph name that is a synthetic placeholder, not a real Adobe name:
350/// `g115`, `cid42`, `glyph7`, `index9`, `G12`, or a short-prefix code name like
351/// `SM590000` (IBM BookMaster). These carry no Unicode meaning, and docling-parse
352/// emits them verbatim (`/SM590000`). `afii####` / `uni####` are real Adobe names
353/// and excluded. The restriction keeps genuine glyph names on the Unicode path.
354fn is_gid_name(name: &[u8]) -> bool {
355    let Ok(s) = std::str::from_utf8(name) else {
356        return false;
357    };
358    if s.starts_with("afii") || s.starts_with("uni") {
359        return false;
360    }
361    for prefix in ["g", "G", "cid", "CID", "glyph", "index"] {
362        if let Some(rest) = s.strip_prefix(prefix) {
363            if !rest.is_empty() && rest.bytes().all(|b| b.is_ascii_digit()) {
364                return true;
365            }
366        }
367    }
368    // Short alpha prefix (≤3 letters) followed by a run of ≥3 digits — synthetic
369    // code names like `SM590000`, distinct from real Adobe names (whole words or
370    // letter+`.suffix` variants).
371    let alpha = s.bytes().take_while(|b| b.is_ascii_alphabetic()).count();
372    let digits = s.len() - alpha;
373    (1..=3).contains(&alpha)
374        && digits >= 3
375        && s.as_bytes()[alpha..].iter().all(|b| b.is_ascii_digit())
376}
377
378fn font_ascent_descent(doc: &Document, fdict: &Dictionary, two_byte: bool) -> (f64, f64) {
379    // For Type0, the descriptor lives on the descendant CIDFont.
380    let descr_owner = if two_byte {
381        fdict
382            .get(b"DescendantFonts")
383            .ok()
384            .and_then(|o| deref(doc, o))
385            .and_then(|o| match o {
386                Object::Array(a) => a.first(),
387                _ => None,
388            })
389            .and_then(|o| as_dict(doc, o))
390    } else {
391        Some(fdict)
392    };
393    let fd = descr_owner
394        .and_then(|d| d.get(b"FontDescriptor").ok())
395        .and_then(|o| as_dict(doc, o));
396    let asc = fd
397        .and_then(|d| d.get(b"Ascent").ok())
398        .and_then(|o| {
399            o.as_float()
400                .ok()
401                .or_else(|| o.as_i64().ok().map(|i| i as f32))
402        })
403        .unwrap_or(750.0) as f64;
404    let desc = fd
405        .and_then(|d| d.get(b"Descent").ok())
406        .and_then(|o| {
407            o.as_float()
408                .ok()
409                .or_else(|| o.as_i64().ok().map(|i| i as f32))
410        })
411        .unwrap_or(-250.0) as f64;
412    // Some subsetted fonts carry a degenerate FontDescriptor (`/Ascent 0
413    // /Descent 0`) — the real metrics live in the font program. That collapses
414    // the loose box to zero height, so the line cells get zero area and the
415    // layout's region/text assignment drops them (2305's References list lost
416    // every prose line, keeping only the URLs). Fall back to typical text metrics
417    // so the box has height.
418    if asc - desc <= 1.0 {
419        return (750.0, -250.0);
420    }
421    (asc, desc)
422}
423
424/// The `/BaseFont` name with any `ABCDEF+` subset prefix stripped (#187).
425fn base_font_name(fdict: &Dictionary) -> Option<Vec<u8>> {
426    let name = fdict.get(b"BaseFont").ok()?.as_name().ok()?;
427    let stripped = match name.iter().position(|&b| b == b'+') {
428        Some(i) if i == 6 => &name[i + 1..],
429        _ => name,
430    };
431    Some(stripped.to_vec())
432}
433
434/// Simple-font widths: `/FirstChar` + `/Widths` array, `/MissingWidth` default.
435fn simple_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
436    let mut map = HashMap::new();
437    let first = fdict
438        .get(b"FirstChar")
439        .ok()
440        .and_then(|o| o.as_i64().ok())
441        .unwrap_or(0) as u32;
442    if let Some(Object::Array(arr)) = fdict.get(b"Widths").ok().and_then(|o| deref(doc, o)) {
443        for (i, w) in arr.iter().enumerate() {
444            if let Some(w) = num(w) {
445                map.insert(first + i as u32, w);
446            }
447        }
448    }
449    let dw = fdict
450        .get(b"FontDescriptor")
451        .ok()
452        .and_then(|o| as_dict(doc, o))
453        .and_then(|d| d.get(b"MissingWidth").ok())
454        .and_then(num)
455        .unwrap_or(0.0);
456    (map, dw)
457}
458
459/// CIDFont widths: the `/W` array on the descendant font (`/DW` default = 1000).
460fn cid_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
461    let mut map = HashMap::new();
462    let Some(desc) = fdict
463        .get(b"DescendantFonts")
464        .ok()
465        .and_then(|o| deref(doc, o))
466        .and_then(|o| match o {
467            Object::Array(a) => a.first(),
468            _ => None,
469        })
470        .and_then(|o| as_dict(doc, o))
471    else {
472        return (map, 1000.0);
473    };
474    let dw = desc.get(b"DW").ok().and_then(num).unwrap_or(1000.0);
475    if let Some(Object::Array(w)) = desc.get(b"W").ok().and_then(|o| deref(doc, o)) {
476        let mut i = 0;
477        while i < w.len() {
478            let c = w.get(i).and_then(num);
479            match (c, w.get(i + 1)) {
480                // `c [w1 w2 ...]`: consecutive CIDs starting at c.
481                (Some(c), Some(Object::Array(list))) => {
482                    for (k, wv) in list.iter().enumerate() {
483                        if let Some(wv) = num(wv) {
484                            map.insert(c as u32 + k as u32, wv);
485                        }
486                    }
487                    i += 2;
488                }
489                // `c_first c_last w`: a run all of width w.
490                (Some(c1), Some(o2)) => {
491                    if let (Some(c2), Some(wv)) = (num(o2), w.get(i + 2).and_then(num)) {
492                        for cid in c1 as u32..=c2 as u32 {
493                            map.insert(cid, wv);
494                        }
495                    }
496                    i += 3;
497                }
498                _ => break,
499            }
500        }
501    }
502    (map, dw)
503}
504
505fn num(o: &Object) -> Option<f64> {
506    match o {
507        Object::Integer(i) => Some(*i as f64),
508        Object::Real(r) => Some(*r as f64),
509        _ => None,
510    }
511}
512
513/// Parse a ToUnicode CMap's `bfchar` / `bfrange` sections into code→string.
514pub(crate) fn parse_tounicode(data: &[u8]) -> HashMap<u32, String> {
515    let text = String::from_utf8_lossy(data);
516    let mut map = HashMap::new();
517    let hex = |s: &str| -> Option<Vec<u16>> {
518        let s = s.trim();
519        if !s.starts_with('<') || !s.ends_with('>') {
520            return None;
521        }
522        let h = &s[1..s.len() - 1];
523        let bytes: Vec<u8> = (0..h.len())
524            .step_by(2)
525            .filter_map(|i| u8::from_str_radix(h.get(i..i + 2)?, 16).ok())
526            .collect();
527        Some(
528            bytes
529                .chunks(2)
530                .map(|c| {
531                    if c.len() == 2 {
532                        u16::from_be_bytes([c[0], c[1]])
533                    } else {
534                        c[0] as u16
535                    }
536                })
537                .collect(),
538        )
539    };
540    let u16s_to_string = |u: &[u16]| String::from_utf16_lossy(u);
541    let code_of = |u: &[u16]| u.iter().fold(0u32, |acc, &x| (acc << 16) | x as u32);
542
543    // Tokenize by structure, not whitespace: CMap hex groups are often written
544    // back-to-back with no separators (`<21><21><0054>`), so scan for `<…>`
545    // groups, `[`/`]` brackets, and bareword keywords.
546    let tokens: Vec<String> = {
547        let bytes = text.as_bytes();
548        let mut toks = Vec::new();
549        let mut i = 0;
550        while i < bytes.len() {
551            let c = bytes[i];
552            if c.is_ascii_whitespace() {
553                i += 1;
554            } else if c == b'<' {
555                let start = i;
556                while i < bytes.len() && bytes[i] != b'>' {
557                    i += 1;
558                }
559                i += 1; // include '>'
560                toks.push(String::from_utf8_lossy(&bytes[start..i.min(bytes.len())]).into_owned());
561            } else if c == b'[' || c == b']' {
562                toks.push((c as char).to_string());
563                i += 1;
564            } else {
565                let start = i;
566                while i < bytes.len()
567                    && !bytes[i].is_ascii_whitespace()
568                    && bytes[i] != b'<'
569                    && bytes[i] != b'['
570                    && bytes[i] != b']'
571                {
572                    i += 1;
573                }
574                toks.push(String::from_utf8_lossy(&bytes[start..i]).into_owned());
575            }
576        }
577        toks
578    };
579    let tokens: Vec<&str> = tokens.iter().map(|s| s.as_str()).collect();
580    let mut i = 0;
581    while i < tokens.len() {
582        match tokens[i] {
583            "beginbfchar" => {
584                i += 1;
585                while i + 1 < tokens.len() && tokens[i] != "endbfchar" {
586                    if let (Some(src), Some(dst)) = (hex(tokens[i]), hex(tokens[i + 1])) {
587                        map.insert(code_of(&src), u16s_to_string(&dst));
588                    }
589                    i += 2;
590                }
591            }
592            "beginbfrange" => {
593                i += 1;
594                while i + 2 < tokens.len() && tokens[i] != "endbfrange" {
595                    let (Some(lo), Some(hi)) = (hex(tokens[i]), hex(tokens[i + 1])) else {
596                        i += 1;
597                        continue;
598                    };
599                    let lo = code_of(&lo);
600                    let hi = code_of(&hi);
601                    if tokens[i + 2] == "[" {
602                        // `<lo> <hi> [ <d0> <d1> ... ]`: one dst per code in the range.
603                        let mut j = i + 3;
604                        let mut code = lo;
605                        while j < tokens.len() && tokens[j] != "]" {
606                            if let Some(dst) = hex(tokens[j]) {
607                                map.insert(code, u16s_to_string(&dst));
608                            }
609                            code += 1;
610                            j += 1;
611                        }
612                        i = j + 1;
613                    } else if let Some(dst) = hex(tokens[i + 2]) {
614                        // `<lo> <hi> <dst>`: consecutive Unicode from a base.
615                        let base = code_of(&dst);
616                        for (k, code) in (lo..=hi).enumerate() {
617                            if let Some(ch) = char::from_u32(base + k as u32) {
618                                map.insert(code, ch.to_string());
619                            }
620                        }
621                        i += 3;
622                    } else {
623                        i += 1;
624                    }
625                }
626            }
627            _ => i += 1,
628        }
629    }
630    map
631}
632
633/// Decode a PDF string literal in a Tj/TJ operand into raw code units.
634fn codes(font: &Font, bytes: &[u8]) -> Vec<u32> {
635    if font.two_byte {
636        bytes
637            .chunks(2)
638            .map(|c| {
639                if c.len() == 2 {
640                    ((c[0] as u32) << 8) | c[1] as u32
641                } else {
642                    c[0] as u32
643                }
644            })
645            .collect()
646    } else {
647        bytes.iter().map(|&b| b as u32).collect()
648    }
649}
650
651/// A page's display box, in PDF user space: the `/CropBox` clipped to the
652/// `/MediaBox`, both inherited through the page tree and normalized — a
653/// missing or empty MediaBox is US Letter, an empty CropBox is the MediaBox
654/// (pdfium's `CPDF_Page::UpdateDimensions`). pdfium reports the page size from
655/// this box, renders exactly it, and translates every content coordinate so
656/// its lower-left corner is the origin (`m_PageMatrix`); docling's backends
657/// inherit that frame, so text cells, `prov` boxes and destinations all count
658/// from the CropBox corner, not the MediaBox one. The parser used to flip
659/// glyphs with the MediaBox *height* and no translation at all, so a page
660/// whose boxes do not start at (0, 0) — a trimmed book page with
661/// `MediaBox [-56 -58 576 723]` / `CropBox [1 -0.6 519 666]`, or a LaTeX
662/// figure cropped to `[156 147 637 391]` — had its text displaced against the
663/// rendered bitmap by the box offset, the bottom lines pushed past the page
664/// edge and clamped to `t = b`.
665#[derive(Debug, Clone, Copy, PartialEq)]
666pub(crate) struct PageBox {
667    /// Left edge, user space.
668    pub l: f32,
669    /// Bottom edge, user space.
670    pub b: f32,
671    pub w: f32,
672    pub h: f32,
673}
674
675impl PageBox {
676    /// Top edge, user space — the y that becomes `0` in the y-down frame.
677    pub fn top(&self) -> f32 {
678        self.b + self.h
679    }
680}
681
682/// A page-tree rect attribute (`/MediaBox`, `/CropBox`), inherited from the
683/// nearest ancestor that sets it, as normalized `(l, b, r, t)`.
684fn inherited_rect(
685    doc: &Document,
686    page_id: lopdf::ObjectId,
687    key: &[u8],
688) -> Option<(f32, f32, f32, f32)> {
689    let mut id = page_id;
690    for _ in 0..32 {
691        let dict = doc.get_object(id).ok()?.as_dict().ok()?;
692        if let Some(Object::Array(a)) = dict.get(key).ok().and_then(|o| deref(doc, o)) {
693            let v: Vec<f32> = a.iter().filter_map(|o| num(o).map(|x| x as f32)).collect();
694            if v.len() == 4 && v.iter().all(|x| x.is_finite()) {
695                return Some((
696                    v[0].min(v[2]),
697                    v[1].min(v[3]),
698                    v[0].max(v[2]),
699                    v[1].max(v[3]),
700                ));
701            }
702            return None;
703        }
704        id = dict.get(b"Parent").ok()?.as_reference().ok()?;
705    }
706    None
707}
708
709pub(crate) fn page_box(doc: &Document, page_id: lopdf::ObjectId) -> PageBox {
710    let nonempty = |r: &(f32, f32, f32, f32)| r.2 > r.0 && r.3 > r.1;
711    let media = inherited_rect(doc, page_id, b"MediaBox")
712        .filter(nonempty)
713        .unwrap_or((0.0, 0.0, 612.0, 792.0));
714    let crop = inherited_rect(doc, page_id, b"CropBox")
715        .map(|c| {
716            (
717                c.0.max(media.0),
718                c.1.max(media.1),
719                c.2.min(media.2),
720                c.3.min(media.3),
721            )
722        })
723        .filter(nonempty)
724        .unwrap_or(media);
725    PageBox {
726        l: crop.0,
727        b: crop.1,
728        w: crop.2 - crop.0,
729        h: crop.3 - crop.1,
730    }
731}
732
733/// Page size (width, height) in PDF points — the display box's, like pdfium's
734/// `FPDF_GetPageWidthF/HeightF`.
735fn page_size(doc: &Document, page_id: lopdf::ObjectId) -> (f32, f32) {
736    let pb = page_box(doc, page_id);
737    (pb.w, pb.h)
738}
739
740/// Localize where a page's text is lost, for the `text_layer` diagnostic.
741/// Extraction can come up empty at three different points — no content stream
742/// reached the parser, the stream did not decode into operators, or it ran but
743/// produced no glyphs (fonts/encodings) — and from the outside all three look
744/// the same. Report them per page.
745pub fn content_diagnosis(bytes: &[u8]) -> String {
746    let Some(doc) = load_document(bytes) else {
747        return "document does not load".into();
748    };
749    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
750    pages.sort_by_key(|(n, _)| *n);
751    let mut out = String::new();
752    let mut caches = DocCaches::default();
753    for (n, pid) in pages.into_iter().take(4) {
754        let content_bytes = doc.get_page_content(pid);
755        let ops = lopdf::content::Content::decode(&content_bytes)
756            .map(|c| c.operations.len())
757            .ok();
758        let res = page_res(&doc, pid);
759        let fonts = res.map(|r| fonts_from_res(&doc, r, &mut caches).len());
760        let glyphs = page_glyphs_cached(&doc, pid, &mut caches).len();
761        out.push_str(&format!(
762            "\n   page {n}: content {} B, ops {}, resources {}, fonts {}, glyphs {}",
763            content_bytes.len(),
764            ops.map_or("UNDECODABLE".to_string(), |n| n.to_string()),
765            if res.is_some() { "ok" } else { "MISSING" },
766            fonts.map_or("-".to_string(), |n| n.to_string()),
767            glyphs,
768        ));
769    }
770    out
771}
772
773/// Is this "text layer" a vestige rather than the document's text?
774///
775/// Scanned forms often carry a handful of typed-in strings — a date filled
776/// into three form fields, say — on top of pages that are otherwise images.
777/// Treating that as a real text layer is the worst of both worlds: the text
778/// path proudly extracts thirteen characters, and no OCR ever runs on the
779/// letter the pages actually show. The reported form did exactly this (3
780/// lines, 13 chars, 3 pages).
781///
782/// The rule is deliberately tight so genuinely sparse *digital* documents are
783/// not misrouted into OCR: only a document averaging at most one line per page
784/// **and** totalling fewer than 32 characters is called vestigial.
785pub fn text_layer_is_vestigial(pages: &[crate::pdfium_backend::PdfPage]) -> bool {
786    let lines: usize = pages.iter().map(|p| p.cells.len()).sum();
787    if lines == 0 {
788        return true;
789    }
790    let chars: usize = pages
791        .iter()
792        .flat_map(|p| &p.cells)
793        .map(|c| c.text.chars().count())
794        .sum();
795    lines <= pages.len() && chars < 32
796}
797
798/// Why the cross-reference repair did or did not fire, for the `text_layer`
799/// diagnostic. A PDF that will not load is indistinguishable from a scan in
800/// production (both convert to nothing), so the reason has to be askable.
801pub fn xref_repair_status(bytes: &[u8]) -> String {
802    if Document::load_mem(bytes).is_ok() {
803        return "loads unaided; no repair needed".into();
804    }
805    match pad_short_xref_entries(bytes) {
806        Ok(fixed) => match Document::load_mem(&fixed) {
807            Ok(_) => "repaired: cross-reference entries padded to 20 bytes".into(),
808            Err(e) => format!("padded the entries, but it still will not load: {e}"),
809        },
810        Err(why) => format!("repair declined — {why}"),
811    }
812}
813
814/// Load a PDF, repairing the one malformation that otherwise costs us the whole
815/// document: **19-byte cross-reference entries**.
816///
817/// The spec fixes an xref entry at 20 bytes — `nnnnnnnnnn ggggg n` plus a
818/// *two*-byte EOL. Some generators (an Austrian telecom's invoices, for one)
819/// emit a bare LF instead, making each entry 19 bytes. lopdf rejects the file
820/// outright (`invalid file trailer`) where pdfium reads it happily, so a
821/// perfectly good text layer looked to the browser exactly like a scan and cost
822/// ten seconds of OCR.
823///
824/// Padding is only attempted when it cannot move anything the xref points at:
825/// a single `xref` section that begins after the last object. The repair then
826/// has to prove itself — the padded bytes are used only if they load — so a
827/// mis-repair degrades to today's behaviour rather than to silent garbage.
828pub(crate) fn load_document(bytes: &[u8]) -> Option<Document> {
829    open_document(bytes, None).ok()
830}
831
832/// Why [`open_document`] could not hand back a readable document.
833#[derive(Debug, Clone, Copy, PartialEq, Eq)]
834pub(crate) enum OpenError {
835    /// lopdf cannot read the file even after the repairs.
836    Unreadable,
837    /// The file is encrypted and the password given (or none) does not open it.
838    Password,
839}
840
841/// Load `bytes` with the document's password. lopdf decrypts while reading —
842/// with `password`, or the empty user password most "protected" PDFs carry
843/// (what every viewer opens silently) — so every reader downstream sees plain
844/// streams; a file the password does not open loads with its streams still
845/// encrypted, which is reported as [`OpenError::Password`] rather than handed
846/// on as a document whose every stream decodes to nothing.
847pub(crate) fn open_document(bytes: &[u8], password: Option<&str>) -> Result<Document, OpenError> {
848    let Some(doc) = load_document_raw(bytes, password) else {
849        // lopdf refuses to load at all under a *wrong* password (a missing
850        // one loads the file with its streams still encrypted); tell the two
851        // apart by loading without it.
852        if password.is_some() && load_document_raw(bytes, None).is_some_and(|d| d.is_encrypted()) {
853            return Err(OpenError::Password);
854        }
855        return Err(OpenError::Unreadable);
856    };
857    if doc.is_encrypted() {
858        return Err(OpenError::Password);
859    }
860    Ok(doc)
861}
862
863fn load_options(password: Option<&str>) -> lopdf::LoadOptions {
864    lopdf::LoadOptions {
865        password: password.map(str::to_string),
866        ..lopdf::LoadOptions::default()
867    }
868}
869
870fn load_document_raw(bytes: &[u8], password: Option<&str>) -> Option<Document> {
871    // Try progressively more repair, and accept a candidate only once the pages
872    // actually carry content — a document whose streams were dropped still
873    // "loads", so loading alone is not evidence the repair helped. A
874    // well-formed file returns on the first attempt and pays for nothing.
875    let mut fallback = None;
876    if let Some(doc) = best_effort_load(bytes, password, &mut fallback) {
877        return Some(doc);
878    }
879    let xref_fixed = pad_short_xref_entries(bytes).ok();
880    if let Some(fixed) = &xref_fixed {
881        if let Some(doc) = best_effort_load(fixed, password, &mut fallback) {
882            return Some(doc);
883        }
884    }
885    // Both defects can coexist, and the second only becomes visible once the
886    // first is repaired, so build on whatever the previous step produced.
887    let lengths_fixed = fix_stream_lengths(xref_fixed.as_deref().unwrap_or(bytes));
888    if let Some(doc) = best_effort_load(&lengths_fixed, password, &mut fallback) {
889        return Some(doc);
890    }
891    fallback
892}
893
894/// Load `data`, returning it only when its pages carry content; a document that
895/// merely parses is remembered as the fallback for when nothing does better.
896fn best_effort_load(
897    data: &[u8],
898    password: Option<&str>,
899    fallback: &mut Option<Document>,
900) -> Option<Document> {
901    match Document::load_mem_with_options(data, load_options(password)) {
902        Ok(doc) if has_page_content(&doc) => Some(doc),
903        Ok(doc) => {
904            fallback.get_or_insert(doc);
905            None
906        }
907        Err(_) => None,
908    }
909}
910
911/// Does any page actually hand us a content stream? A document whose streams
912/// were dropped still parses — it simply has nothing to read — so this is what
913/// tells a successful repair from a pointless one.
914fn has_page_content(doc: &Document) -> bool {
915    doc.get_pages()
916        .into_values()
917        .take(4)
918        .any(|pid| !doc.get_page_content(pid).is_empty())
919}
920
921/// Correct `/Length` values that disagree with where `endstream` actually is.
922///
923/// The same generator that writes short xref entries also overstates its
924/// content-stream lengths by a byte or two. lopdf trusts `/Length`, reads past
925/// the data, fails to find `endstream` there and drops the stream — the object
926/// comes back as a bare dictionary, so the page has no content at all and the
927/// document looks like a scan. pdfium instead trusts `endstream`, which is what
928/// this does.
929///
930/// The rewrite is length-preserving: the corrected number is written over the
931/// old digits and padded with spaces, so every byte offset in the file — and
932/// therefore the whole cross-reference table — stays valid.
933fn fix_stream_lengths(bytes: &[u8]) -> Vec<u8> {
934    let mut out = bytes.to_vec();
935    let mut i = 0;
936    while let Some(rel) = find(&out[i..], b"stream") {
937        let kw = i + rel;
938        i = kw + 6;
939        // Skip `endstream` (the keyword we are measuring *to*).
940        if kw >= 3 && &out[kw - 3..kw] == b"end" {
941            continue;
942        }
943        // The stream data starts after the EOL that follows the keyword.
944        let mut data = kw + 6;
945        if out.get(data..data + 2) == Some(b"\r\n".as_slice()) {
946            data += 2;
947        } else if matches!(out.get(data), Some(b'\n' | b'\r')) {
948            data += 1;
949        }
950        let Some(end) = find(&out[data..], b"endstream").map(|r| data + r) else {
951            continue;
952        };
953        // `/Length <digits>` in the dictionary just before the keyword.
954        let dict_start = out[..kw].iter().rposition(|&c| c == b'<').unwrap_or(0);
955        let Some(lrel) = find(&out[dict_start..kw], b"/Length") else {
956            continue;
957        };
958        let mut d = dict_start + lrel + 7;
959        while matches!(out.get(d), Some(b' ')) {
960            d += 1;
961        }
962        let digits = out[d..].iter().take_while(|c| c.is_ascii_digit()).count();
963        if digits == 0 {
964            continue;
965        }
966        let declared: usize = match std::str::from_utf8(&out[d..d + digits])
967            .ok()
968            .and_then(|s| s.parse().ok())
969        {
970            Some(v) => v,
971            None => continue,
972        };
973        let actual = end - data;
974        // Only shrink, and only when the new value fits the space the old one
975        // occupied — growing the number would move every following byte.
976        let replacement = actual.to_string();
977        if actual == declared || replacement.len() > digits {
978            continue;
979        }
980        out[d..d + digits].fill(b' ');
981        out[d..d + replacement.len()].copy_from_slice(replacement.as_bytes());
982    }
983    out
984}
985
986fn find(haystack: &[u8], needle: &[u8]) -> Option<usize> {
987    haystack.windows(needle.len()).position(|w| w == needle)
988}
989
990/// Rewrite a classic cross-reference table's entries to the spec's 20 bytes,
991/// or `None` when the file's shape makes that unsafe (see [`load_document`]).
992fn pad_short_xref_entries(bytes: &[u8]) -> Result<Vec<u8>, &'static str> {
993    // Exactly one xref section, and it must start after every object, so that
994    // growing it shifts nothing the table's offsets refer to.
995    let is_boundary = |i: usize| i == 0 || matches!(bytes[i - 1], b'\n' | b'\r');
996    let mut starts = (0..bytes.len().saturating_sub(4))
997        .filter(|&i| &bytes[i..i + 4] == b"xref" && is_boundary(i));
998    let xref_at = starts
999        .next()
1000        .ok_or("no classic `xref` section (an xref stream?)")?;
1001    if starts.next().is_some() {
1002        return Err("more than one xref section (incremental update)");
1003    }
1004    let last_obj = bytes
1005        .windows(3)
1006        .rposition(|w| w == b"obj")
1007        .ok_or("no objects found")?;
1008    if last_obj > xref_at {
1009        return Err("an object follows the xref — padding would move it");
1010    }
1011
1012    let mut out = bytes[..xref_at].to_vec();
1013    out.extend_from_slice(b"xref\n");
1014    let mut i = xref_at + 4;
1015    let skip_ws = |i: &mut usize| {
1016        while matches!(bytes.get(*i), Some(b'\r' | b'\n' | b' ')) {
1017            *i += 1;
1018        }
1019    };
1020    loop {
1021        skip_ws(&mut i);
1022        // Either the next subsection header ("first count") or the trailer.
1023        if bytes[i..].starts_with(b"trailer") {
1024            out.extend_from_slice(&bytes[i..]);
1025            return Ok(out);
1026        }
1027        let header_end = i + bytes[i..]
1028            .iter()
1029            .position(|c| matches!(c, b'\n' | b'\r'))
1030            .ok_or("subsection header runs off the end")?;
1031        let header = std::str::from_utf8(&bytes[i..header_end])
1032            .map_err(|_| "subsection header is not text")?
1033            .trim();
1034        let mut parts = header.split_whitespace();
1035        let count: usize = parts
1036            .nth(1)
1037            .and_then(|c| c.parse().ok())
1038            .ok_or("unparseable subsection header")?;
1039        if parts.next().is_some() || count == 0 {
1040            return Err("unexpected subsection header shape");
1041        }
1042        out.extend_from_slice(header.as_bytes());
1043        out.push(b'\n');
1044        i = header_end;
1045        for _ in 0..count {
1046            skip_ws(&mut i);
1047            // `nnnnnnnnnn ggggg n` — the 18 bytes before whatever EOL follows.
1048            let entry = bytes.get(i..i + 18).ok_or("xref entry runs off the end")?;
1049            let well_formed = entry[..10].iter().all(u8::is_ascii_digit)
1050                && entry[10] == b' '
1051                && entry[11..16].iter().all(u8::is_ascii_digit)
1052                && entry[16] == b' '
1053                && matches!(entry[17], b'n' | b'f');
1054            if !well_formed {
1055                return Err("xref entry is not `nnnnnnnnnn ggggg n`");
1056            }
1057            out.extend_from_slice(entry);
1058            out.extend_from_slice(b" \n"); // the spec's 2-byte EOL -> 20 bytes
1059            i += 18;
1060        }
1061    }
1062}
1063
1064/// Debug: raw glyph stream `(ch, ll, lr, lb, lt)` (native coords) for page
1065/// `index`, before the sanitizer. For comparing char cells to docling-parse.
1066pub fn debug_glyphs(bytes: &[u8], index: usize) -> Vec<(char, f32, f32, f32, f32)> {
1067    let Some(doc) = load_document(bytes) else {
1068        return Vec::new();
1069    };
1070    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1071    pages.sort_by_key(|(n, _)| *n);
1072    let Some((_, pid)) = pages.get(index) else {
1073        return Vec::new();
1074    };
1075    page_glyphs(&doc, *pid)
1076        .into_iter()
1077        .map(|g| (g.ch, g.ll, g.lr, g.lb, g.lt))
1078        .collect()
1079}
1080
1081/// Public entry: per-page (width, height, line cells) for a PDF, via the Rust
1082/// text parser + the docling-parse line sanitizer. Used by the pipeline and the
1083/// `textparse_dump` example.
1084pub fn pdf_textlines(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
1085    let Some(doc) = load_document(bytes) else {
1086        return Vec::new();
1087    };
1088    let mut caches = DocCaches::default();
1089    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1090    pages.sort_by_key(|(n, _)| *n);
1091    pages
1092        .into_iter()
1093        .map(|(_, pid)| {
1094            let (w, h) = page_size(&doc, pid);
1095            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1096            let cells = crate::dp_lines::line_cells(&glyphs, h, true);
1097            (w, h, cells)
1098        })
1099        .collect()
1100}
1101
1102/// Debug/diagnostic entry: per-page (width, height, word cells) for a PDF, via
1103/// the Rust parser glyphs run through the docling-parse word grouping. Used to
1104/// compare parser word cells against docling-parse's `word_cells` oracle (roadmap
1105/// item 6).
1106pub fn pdf_words(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
1107    let Some(doc) = load_document(bytes) else {
1108        return Vec::new();
1109    };
1110    let mut caches = DocCaches::default();
1111    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1112    pages.sort_by_key(|(n, _)| *n);
1113    pages
1114        .into_iter()
1115        .map(|(_, pid)| {
1116            let (w, h) = page_size(&doc, pid);
1117            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1118            let cells = crate::dp_lines::word_cells(&glyphs, h, true);
1119            (w, h, cells)
1120        })
1121        .collect()
1122}
1123
1124/// One page's text cells from the pure-Rust parser: prose line cells, per-word
1125/// cells, and code line cells — all from a single glyph parse. Replaces the
1126/// pdfium text path (roadmap item 6) when the parser drop is enabled.
1127#[derive(Default)]
1128pub struct PageParserCells {
1129    pub prose: Vec<crate::pdfium_backend::TextCell>,
1130    pub words: Vec<crate::pdfium_backend::TextCell>,
1131    pub code: Vec<crate::pdfium_backend::TextCell>,
1132}
1133
1134/// The parser text layer, driven one page at a time: the document is loaded
1135/// (and repaired, see [`load_document`]) once, the font/form caches persist
1136/// across pages, and each page's glyphs are parsed only when asked for.
1137///
1138/// The eager whole-document walk this replaces ran *before* the first page
1139/// was rendered, so on a long PDF it was a serial prefix the page-worker pool
1140/// sat idle through — 6.2 s on the 1913-page .NET reference, in front of a
1141/// pipeline that otherwise overlaps parsing with inference — and a `--pages`
1142/// window still paid for every page in the file. Pulling pages on demand
1143/// keeps the parse on the producer thread but interleaved with rendering,
1144/// and skips unselected pages entirely. Output per page is unchanged: same
1145/// glyph walk, same shared caches, same contraction.
1146pub struct PageTextParser {
1147    doc: Document,
1148    caches: DocCaches,
1149    /// Page object ids in document order (page 1 first).
1150    pages: Vec<lopdf::ObjectId>,
1151}
1152
1153impl PageTextParser {
1154    /// Load the document; `None` when lopdf cannot read it at all.
1155    pub fn open(bytes: &[u8]) -> Option<Self> {
1156        Self::open_with_password(bytes, None)
1157    }
1158
1159    /// [`open`](Self::open) with the document's password (an encrypted file
1160    /// the password does not open reads as `None` here; the pipeline reports
1161    /// it through [`crate::pdf_meta::PdfMeta::open_with_password`] first).
1162    pub fn open_with_password(bytes: &[u8], password: Option<&str>) -> Option<Self> {
1163        let doc = open_document(bytes, password).ok()?;
1164        let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1165        pages.sort_by_key(|(n, _)| *n);
1166        Some(Self {
1167            doc,
1168            caches: DocCaches::default(),
1169            pages: pages.into_iter().map(|(_, pid)| pid).collect(),
1170        })
1171    }
1172
1173    /// The glyph boxes and font styles of the 0-based page `index` — the
1174    /// heading-hierarchy stage's style signal (#302): every non-space glyph's
1175    /// box (font ascent + descent at its size — the font-size proxy pdfium's
1176    /// loose char box also gave) in top-left coordinates, with the weight
1177    /// class and slant its `/BaseFont` name declares. Empty for a page
1178    /// without a text layer (a scan), and the stage falls back to its other
1179    /// signals.
1180    pub(crate) fn glyph_styles(
1181        &mut self,
1182        index: usize,
1183    ) -> Vec<crate::heading_hierarchy::GlyphStyle> {
1184        let Some(&pid) = self.pages.get(index) else {
1185            return Vec::new();
1186        };
1187        let (_w, h) = page_size(&self.doc, pid);
1188        let glyphs = page_glyphs_cached(&self.doc, pid, &mut self.caches);
1189        // Font hash → style, over the fonts the walk just parsed (an inline,
1190        // uncached font dictionary reads as unstyled).
1191        let styles: HashMap<u64, crate::font_style::FontStyle> = self
1192            .caches
1193            .fonts
1194            .values()
1195            .map(|f| (f.hash, f.style))
1196            .collect();
1197        glyphs
1198            .iter()
1199            .filter(|g| !g.ch.is_whitespace() && g.ll.is_finite())
1200            .map(|g| {
1201                let st = styles.get(&g.font).copied().unwrap_or_default();
1202                crate::heading_hierarchy::GlyphStyle {
1203                    l: g.ll,
1204                    t: h - g.lt,
1205                    r: g.lr,
1206                    b: h - g.lb,
1207                    height: g.lt - g.lb,
1208                    weight_cls: crate::font_style::weight_class(st.weight),
1209                    italic: st.italic,
1210                    styled: st.known,
1211                }
1212            })
1213            .collect()
1214    }
1215
1216    /// Prose, word and code cells of the 0-based page `index` — empty for an
1217    /// index the parser's page tree doesn't have (a damaged file whose page
1218    /// tree disagrees with the object model's count).
1219    pub fn cells(&mut self, index: usize) -> PageParserCells {
1220        let Some(&pid) = self.pages.get(index) else {
1221            return PageParserCells::default();
1222        };
1223        let (_w, h) = page_size(&self.doc, pid);
1224        let glyphs = page_glyphs_cached(&self.doc, pid, &mut self.caches);
1225        let (prose, words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1226        PageParserCells {
1227            prose,
1228            words,
1229            code: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1230        }
1231    }
1232}
1233
1234/// Full parser text layer: prose + word + code cells per page, glyphs parsed once.
1235/// `prose`/`words` come from the docling-parse contraction ([`crate::dp_lines`]);
1236/// `code` splits only at the parser's own space glyphs (monospace keeps its
1237/// source spacing). The eager form of [`PageTextParser`].
1238pub fn pdf_all_cells(bytes: &[u8]) -> Vec<PageParserCells> {
1239    let Some(mut parser) = PageTextParser::open(bytes) else {
1240        return Vec::new();
1241    };
1242    (0..parser.pages.len()).map(|i| parser.cells(i)).collect()
1243}
1244
1245/// Whole pages for the text-layer-only conversion ([`crate::convert_text_layer`]):
1246/// the parser's prose/word/code cells plus page geometry, assembled into
1247/// [`PdfPage`]s with no rendered image and no link annotations. Everything here
1248/// is pure Rust (lopdf), so it compiles without the `ml` feature — including on
1249/// wasm32. A page the parser can't read (no text layer) comes back with empty
1250/// cells; there is no pdfium fallback on this path.
1251pub fn pdf_text_pages(bytes: &[u8]) -> Vec<crate::pdfium_backend::PdfPage> {
1252    let Some(doc) = load_document(bytes) else {
1253        return Vec::new();
1254    };
1255    let mut caches = DocCaches::default();
1256    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1257    pages.sort_by_key(|(n, _)| *n);
1258    pages
1259        .into_iter()
1260        .map(|(_, pid)| {
1261            let (w, h) = page_size(&doc, pid);
1262            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1263            let (mut prose, mut words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1264            drop_overpainted_cells(&mut prose);
1265            drop_overpainted_cells(&mut words);
1266            crate::pdfium_backend::PdfPage {
1267                #[cfg(feature = "ocr-prep")]
1268                image_layout: None,
1269                width: w,
1270                height: h,
1271                // Cells are native PDF points; there is no rendered bitmap.
1272                scale: 1.0,
1273                cells: prose,
1274                code_cells: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1275                word_cells: words,
1276                #[cfg(feature = "ocr-prep")]
1277                image: image::RgbImage::new(1, 1),
1278                links: Vec::new(),
1279                rotation: 0,
1280            }
1281        })
1282        .collect()
1283}
1284
1285/// Drop line cells that are *painted over each other* — glyphs used as artwork.
1286///
1287/// Some generators draw their logo with a symbol font: on the reporting
1288/// invoice, a `TeleLogo` Type1 paints the T-Mobile mark by stacking the glyphs
1289/// encoded as `"` and `==` on top of one another, and the flat text-layer
1290/// output opened with that garbage. Nothing in the font metadata gives it away
1291/// (the *text* fonts in the same file are also flagged symbolic, and the logo
1292/// font names its glyphs `quotedbl` &c.), but the geometry does: two cells with
1293/// different text where one lies inside the other on the same line is
1294/// physically impossible for prose — ink from two words never occupies the
1295/// same box. Both cells of such a pair are paint, not text.
1296///
1297/// Containment (not mere overlap) keeps this narrow: adjacent words touch but
1298/// never contain each other, and a same-text near-duplicate (double-draw faux
1299/// bold) is left alone for the sanitizer's usual handling. Applied on the
1300/// flat/browser path only — the ML pipeline's text layer is byte-pinned by the
1301/// PDF corpus, and there the layout model already sinks logo marks into
1302/// `picture` regions.
1303fn drop_overpainted_cells(cells: &mut Vec<crate::pdfium_backend::TextCell>) {
1304    let mut paint = vec![false; cells.len()];
1305    for i in 0..cells.len() {
1306        for j in 0..cells.len() {
1307            if i == j || cells[i].text == cells[j].text {
1308                continue;
1309            }
1310            let (a, b) = (&cells[i], &cells[j]);
1311            // Same line band: the vertical overlap covers most of the shorter.
1312            let vo = (a.b.min(b.b) - a.t.max(b.t)).max(0.0);
1313            if vo < 0.6 * (a.b - a.t).min(b.b - b.t) {
1314                continue;
1315            }
1316            // `a` horizontally inside `b` (with a small tolerance).
1317            let ho = (a.r.min(b.r) - a.l.max(b.l)).max(0.0);
1318            if ho >= 0.8 * (a.r - a.l) && (a.r - a.l) <= (b.r - b.l) {
1319                paint[i] = true;
1320                paint[j] = true;
1321            }
1322        }
1323    }
1324    let mut keep = paint.iter().map(|p| !p);
1325    cells.retain(|_| keep.next().unwrap());
1326}
1327
1328/// The text-state scalars inherited by a Form XObject when it is invoked via
1329/// `Do` (the PDF graphics state includes the text parameters, but not the text
1330/// matrices, which a form re-establishes inside its own `BT`/`ET`).
1331#[derive(Clone, Copy)]
1332struct TextState {
1333    tc: f64,
1334    tw: f64,
1335    th: f64,
1336    tl: f64,
1337    trise: f64,
1338    fsize: f64,
1339}
1340
1341impl TextState {
1342    const INIT: TextState = TextState {
1343        tc: 0.0,
1344        tw: 0.0,
1345        th: 1.0,
1346        tl: 0.0,
1347        trise: 0.0,
1348        fsize: 0.0,
1349    };
1350}
1351
1352/// The effective `/Resources` dictionary for a page (inline or via reference,
1353/// falling back to an inherited one from a `/Parent`).
1354fn page_res(doc: &Document, page_id: lopdf::ObjectId) -> Option<&Dictionary> {
1355    let (inline, ids) = doc.get_page_resources(page_id).ok()?;
1356    if let Some(d) = inline {
1357        return Some(d);
1358    }
1359    ids.into_iter().find_map(|id| doc.get_dictionary(id).ok())
1360}
1361
1362/// Build the code→[`Font`] map for a resources dictionary's `/Font` sub-dict,
1363/// reusing the per-document cache for fonts referenced indirectly (the common
1364/// case — the same font objects recur on every page).
1365fn fonts_from_res(
1366    doc: &Document,
1367    res: &Dictionary,
1368    caches: &mut DocCaches,
1369) -> HashMap<Vec<u8>, Arc<Font>> {
1370    let mut map = HashMap::new();
1371    let font_dict = res
1372        .get(b"Font")
1373        .ok()
1374        .and_then(|o| deref(doc, o))
1375        .and_then(|o| o.as_dict().ok());
1376    if let Some(fd) = font_dict {
1377        for (name, value) in fd.iter() {
1378            let font = match value {
1379                Object::Reference(id) => {
1380                    let key = (*id, name.clone());
1381                    if let Some(f) = caches.fonts.get(&key) {
1382                        Arc::clone(f)
1383                    } else if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1384                        let f = Arc::new(parse_font(doc, name, fdict));
1385                        caches.fonts.insert(key, Arc::clone(&f));
1386                        f
1387                    } else {
1388                        continue;
1389                    }
1390                }
1391                _ => {
1392                    if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1393                        Arc::new(parse_font(doc, name, fdict))
1394                    } else {
1395                        continue;
1396                    }
1397                }
1398            };
1399            map.insert(name.clone(), font);
1400        }
1401    }
1402    map
1403}
1404
1405/// Extract every glyph on a page as a native-coordinate [`Glyph`].
1406/// [`PageTextParser::glyph_styles`] for the given **1-based** pages of a
1407/// document, keyed by page number — a separate, on-demand pass (no
1408/// rendering), so the extraction pipeline stays byte-identical whether or not
1409/// the heading-hierarchy stage runs. Empty when lopdf cannot open the file.
1410pub(crate) fn glyph_styles(
1411    bytes: &[u8],
1412    pages: &[usize],
1413) -> HashMap<usize, Vec<crate::heading_hierarchy::GlyphStyle>> {
1414    let mut out = HashMap::new();
1415    let Some(mut parser) = PageTextParser::open(bytes) else {
1416        return out;
1417    };
1418    for &page_no in pages {
1419        if page_no == 0 {
1420            continue;
1421        }
1422        let styles = parser.glyph_styles(page_no - 1);
1423        if !styles.is_empty() {
1424            out.insert(page_no, styles);
1425        }
1426    }
1427    out
1428}
1429
1430pub(crate) fn page_glyphs(doc: &Document, page_id: lopdf::ObjectId) -> Vec<Glyph> {
1431    page_glyphs_cached(doc, page_id, &mut DocCaches::default())
1432}
1433
1434/// [`page_glyphs`] with an explicit per-document cache, so a multi-page walk
1435/// parses each font / decodes each form once instead of once per page.
1436fn page_glyphs_cached(
1437    doc: &Document,
1438    page_id: lopdf::ObjectId,
1439    caches: &mut DocCaches,
1440) -> Vec<Glyph> {
1441    let mut out = Vec::new();
1442    // lopdf 0.44: get_page_content returns the assembled content-stream bytes
1443    // directly (an empty Vec when the page has none).
1444    let content_bytes = doc.get_page_content(page_id);
1445    let Ok(content) = lopdf::content::Content::decode(&content_bytes) else {
1446        return out;
1447    };
1448    if let Some(res) = page_res(doc, page_id) {
1449        // pdfium's page matrix: user space translated so the display box's
1450        // lower-left corner is the origin (see [`PageBox`]).
1451        let pb = page_box(doc, page_id);
1452        let base = Mat {
1453            e: -(pb.l as f64),
1454            f: -(pb.b as f64),
1455            ..Mat::ID
1456        };
1457        run_content(
1458            doc,
1459            res,
1460            &content,
1461            base,
1462            TextState::INIT,
1463            0,
1464            caches,
1465            &mut out,
1466        );
1467    }
1468    out
1469}
1470
1471/// Run a content stream's operators, emitting glyphs into `out`. Recurses into
1472/// Form XObjects on `Do` (bulk body text in heavy PDFs lives inside a form, not
1473/// the page content stream). `res` is the resources dict in scope (the page's,
1474/// or the form's own); `base_ctm` is the CTM at the point of invocation.
1475#[allow(clippy::too_many_arguments)]
1476fn run_content(
1477    doc: &Document,
1478    res: &Dictionary,
1479    content: &lopdf::content::Content,
1480    base_ctm: Mat,
1481    init: TextState,
1482    depth: u32,
1483    caches: &mut DocCaches,
1484    out: &mut Vec<Glyph>,
1485) {
1486    let fonts = fonts_from_res(doc, res, caches);
1487    let xobjects = res
1488        .get(b"XObject")
1489        .ok()
1490        .and_then(|o| deref(doc, o))
1491        .and_then(|o| o.as_dict().ok());
1492
1493    // Graphics + text state. `q`/`Q` save and restore the whole graphics state,
1494    // which includes the text parameters (Tc, Tw, Tz, TL, Tfs, Trise, font) —
1495    // *not* the text matrix (that is reset by BT). Saving only the CTM let a Tc
1496    // set inside a `q…Q` block leak out and drift every later glyph.
1497    #[allow(clippy::type_complexity)]
1498    let mut gstate_stack: Vec<(Mat, f64, f64, f64, f64, f64, f64, Option<&Arc<Font>>)> = Vec::new();
1499    let mut ctm = base_ctm;
1500    let mut tm = Mat::ID;
1501    let mut tlm = Mat::ID;
1502    let mut font: Option<&Arc<Font>> = None;
1503    let mut fsize = init.fsize;
1504    let mut tc = init.tc; // char spacing
1505    let mut tw = init.tw; // word spacing
1506    let mut th = init.th; // horizontal scale (Tz/100)
1507    let mut tl = init.tl; // leading
1508    let mut trise = init.trise;
1509
1510    let op_f = |operands: &[Object], i: usize| operands.get(i).and_then(num).unwrap_or(0.0);
1511
1512    for op in &content.operations {
1513        let operands = &op.operands;
1514        match op.operator.as_str() {
1515            "q" => gstate_stack.push((ctm, tc, tw, th, tl, trise, fsize, font)),
1516            "Q" => {
1517                if let Some((c, a, b, h, l, r, fs, f)) = gstate_stack.pop() {
1518                    ctm = c;
1519                    tc = a;
1520                    tw = b;
1521                    th = h;
1522                    tl = l;
1523                    trise = r;
1524                    fsize = fs;
1525                    font = f;
1526                }
1527            }
1528            "cm" => {
1529                let m = Mat {
1530                    a: op_f(operands, 0),
1531                    b: op_f(operands, 1),
1532                    c: op_f(operands, 2),
1533                    d: op_f(operands, 3),
1534                    e: op_f(operands, 4),
1535                    f: op_f(operands, 5),
1536                };
1537                ctm = m.then(ctm);
1538            }
1539            "BT" => {
1540                tm = Mat::ID;
1541                tlm = Mat::ID;
1542            }
1543            "ET" => {}
1544            "Tf" => {
1545                if let Some(Object::Name(n)) = operands.first() {
1546                    font = fonts.get(n.as_slice());
1547                }
1548                fsize = op_f(operands, 1);
1549            }
1550            "Td" => {
1551                tlm = Mat {
1552                    a: 1.0,
1553                    b: 0.0,
1554                    c: 0.0,
1555                    d: 1.0,
1556                    e: op_f(operands, 0),
1557                    f: op_f(operands, 1),
1558                }
1559                .then(tlm);
1560                tm = tlm;
1561            }
1562            "TD" => {
1563                tl = -op_f(operands, 1);
1564                tlm = Mat {
1565                    a: 1.0,
1566                    b: 0.0,
1567                    c: 0.0,
1568                    d: 1.0,
1569                    e: op_f(operands, 0),
1570                    f: op_f(operands, 1),
1571                }
1572                .then(tlm);
1573                tm = tlm;
1574            }
1575            "Tm" => {
1576                tlm = Mat {
1577                    a: op_f(operands, 0),
1578                    b: op_f(operands, 1),
1579                    c: op_f(operands, 2),
1580                    d: op_f(operands, 3),
1581                    e: op_f(operands, 4),
1582                    f: op_f(operands, 5),
1583                };
1584                tm = tlm;
1585            }
1586            "T*" => {
1587                tlm = Mat {
1588                    a: 1.0,
1589                    b: 0.0,
1590                    c: 0.0,
1591                    d: 1.0,
1592                    e: 0.0,
1593                    f: -tl,
1594                }
1595                .then(tlm);
1596                tm = tlm;
1597            }
1598            "Tc" => tc = op_f(operands, 0),
1599            "Tw" => tw = op_f(operands, 0),
1600            "Tz" => th = op_f(operands, 0) / 100.0,
1601            "TL" => tl = op_f(operands, 0),
1602            "Ts" => trise = op_f(operands, 0),
1603            "Tj" | "'" | "\"" => {
1604                if op.operator == "'" || op.operator == "\"" {
1605                    // move to next line first
1606                    tlm = Mat {
1607                        a: 1.0,
1608                        b: 0.0,
1609                        c: 0.0,
1610                        d: 1.0,
1611                        e: 0.0,
1612                        f: -tl,
1613                    }
1614                    .then(tlm);
1615                    tm = tlm;
1616                }
1617                if op.operator == "\"" {
1618                    // `aw ac string "` sets word- and char-spacing before
1619                    // showing the string (PDF 32000-1 §9.4.3), persisting after.
1620                    tw = op_f(operands, 0);
1621                    tc = op_f(operands, 1);
1622                }
1623                if let (Some(f), Some(Object::String(s, _))) = (font, operands.last()) {
1624                    show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out);
1625                }
1626            }
1627            "TJ" => {
1628                if let (Some(f), Some(Object::Array(arr))) = (font, operands.first()) {
1629                    for el in arr {
1630                        match el {
1631                            Object::String(s, _) => {
1632                                show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out)
1633                            }
1634                            other => {
1635                                if let Some(adj) = num(other) {
1636                                    // negative number moves text right (PDF: subtract)
1637                                    let tx = -adj / 1000.0 * fsize * th;
1638                                    tm = Mat {
1639                                        a: 1.0,
1640                                        b: 0.0,
1641                                        c: 0.0,
1642                                        d: 1.0,
1643                                        e: tx,
1644                                        f: 0.0,
1645                                    }
1646                                    .then(tm);
1647                                }
1648                            }
1649                        }
1650                    }
1651                }
1652            }
1653            "Do" => {
1654                // Invoke a Form XObject: bulk body text in many PDFs lives inside
1655                // a form, reached only here. Image XObjects are skipped (no text).
1656                if depth >= 8 {
1657                    continue;
1658                }
1659                let Some(Object::Name(n)) = operands.first() else {
1660                    continue;
1661                };
1662                let obj = xobjects.and_then(|d| d.get(n.as_slice()).ok());
1663                let form_id = match obj {
1664                    Some(Object::Reference(id)) => Some(*id),
1665                    _ => None,
1666                };
1667                let stream = obj
1668                    .and_then(|o| deref(doc, o))
1669                    .and_then(|o| o.as_stream().ok());
1670                let Some(stream) = stream else { continue };
1671                let is_form = stream
1672                    .dict
1673                    .get(b"Subtype")
1674                    .ok()
1675                    .and_then(|o| o.as_name().ok())
1676                    == Some(b"Form".as_slice());
1677                if !is_form {
1678                    continue;
1679                }
1680                // Decode the form's content once per document (headers/footers
1681                // and bulk body text invoke the same form on every page).
1682                let cached = form_id.and_then(|id| caches.forms.get(&id).cloned());
1683                let form_content = match cached {
1684                    Some(c) => c,
1685                    None => {
1686                        let Ok(data) = stream.decompressed_content() else {
1687                            continue;
1688                        };
1689                        let Ok(c) = lopdf::content::Content::decode(&data) else {
1690                            continue;
1691                        };
1692                        let c = Arc::new(c);
1693                        if let Some(id) = form_id {
1694                            caches.forms.insert(id, Arc::clone(&c));
1695                        }
1696                        c
1697                    }
1698                };
1699                // The form's /Matrix maps form space into the CTM at invocation.
1700                let form_mat = match stream.dict.get(b"Matrix").ok() {
1701                    Some(Object::Array(a)) if a.len() == 6 => {
1702                        let v: Vec<f64> = a.iter().filter_map(num).collect();
1703                        if v.len() == 6 {
1704                            Mat {
1705                                a: v[0],
1706                                b: v[1],
1707                                c: v[2],
1708                                d: v[3],
1709                                e: v[4],
1710                                f: v[5],
1711                            }
1712                        } else {
1713                            Mat::ID
1714                        }
1715                    }
1716                    _ => Mat::ID,
1717                };
1718                // The form's own /Resources, falling back to the inherited ones.
1719                let form_res = stream
1720                    .dict
1721                    .get(b"Resources")
1722                    .ok()
1723                    .and_then(|o| deref(doc, o))
1724                    .and_then(|o| o.as_dict().ok())
1725                    .unwrap_or(res);
1726                let state = TextState {
1727                    tc,
1728                    tw,
1729                    th,
1730                    tl,
1731                    trise,
1732                    fsize,
1733                };
1734                run_content(
1735                    doc,
1736                    form_res,
1737                    &form_content,
1738                    form_mat.then(ctm),
1739                    state,
1740                    depth + 1,
1741                    caches,
1742                    out,
1743                );
1744            }
1745            _ => {}
1746        }
1747    }
1748}
1749
1750#[allow(clippy::too_many_arguments)]
1751fn show_text(
1752    font: &Font,
1753    bytes: &[u8],
1754    fsize: f64,
1755    tc: f64,
1756    tw: f64,
1757    th: f64,
1758    trise: f64,
1759    tm: &mut Mat,
1760    ctm: Mat,
1761    out: &mut Vec<Glyph>,
1762) {
1763    for code in codes(font, bytes) {
1764        let (text, w) = font.decode_code(code);
1765        let w0 = w / 1000.0; // advance in text-space (em) units
1766                             // The glyph→user transform: scale glyph space by font size, then Tm, CTM.
1767        let scale = Mat {
1768            a: fsize * th,
1769            b: 0.0,
1770            c: 0.0,
1771            d: fsize,
1772            e: 0.0,
1773            f: trise,
1774        };
1775        let trm = scale.then(*tm).then(ctm);
1776        // Box in glyph space (1000-unit em): x 0..w, y descent..ascent.
1777        let (x0, y0) = trm.apply(0.0, font.descent / 1000.0);
1778        let (x1, _y1) = trm.apply(w0, font.descent / 1000.0);
1779        let (_x2, y2) = trm.apply(0.0, font.ascent / 1000.0);
1780        let (left, right) = (x0.min(x1), x0.max(x1));
1781        let (bot, top) = (y0.min(y2), y0.max(y2));
1782        if let Some(s) = text {
1783            // A run may map one code to multiple chars (ligature/fraction); share box.
1784            for ch in s.chars() {
1785                if ch != '\u{0}' {
1786                    out.push(Glyph {
1787                        ch,
1788                        l: left as f32,
1789                        b: bot as f32,
1790                        r: right as f32,
1791                        t: top as f32,
1792                        ll: left as f32,
1793                        lb: bot as f32,
1794                        lr: right as f32,
1795                        lt: top as f32,
1796                        font: font.hash,
1797                    });
1798                }
1799            }
1800        }
1801        // Advance the text matrix. Word spacing applies to single-byte code 32.
1802        let is_space = !font.two_byte && code == 32;
1803        let tx = (w0 * fsize + tc + if is_space { tw } else { 0.0 }) * th;
1804        *tm = Mat {
1805            a: 1.0,
1806            b: 0.0,
1807            c: 0.0,
1808            d: 1.0,
1809            e: tx,
1810            f: 0.0,
1811        }
1812        .then(*tm);
1813    }
1814}
1815
1816/// Build a simple font's code→char table from its `/Encoding`: the base
1817/// encoding (WinAnsi / MacRoman) plus any `/Differences` overrides (glyph names
1818/// resolved through a small Adobe-glyph-name subset).
1819fn simple_encoding_table(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
1820    let enc = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o));
1821    let base_name = match enc {
1822        Some(Object::Name(n)) => n.clone(),
1823        Some(Object::Dictionary(d)) => d
1824            .get(b"BaseEncoding")
1825            .ok()
1826            .and_then(|o| o.as_name().ok())
1827            .map(|n| n.to_vec())
1828            .unwrap_or_default(),
1829        _ => Vec::new(),
1830    };
1831    let mut m = if base_name == b"MacRomanEncoding" {
1832        macroman_table()
1833    } else if base_name.is_empty() {
1834        // No PDF /Encoding at all: the font's *built-in* encoding applies. For
1835        // the standard TeX math fonts that is their fixed TeX layout — falling
1836        // back to StandardEncoding read CMSY's braces as `f`/`g`, `→` as `!`,
1837        // `∈` as `2` (2203's `{ahn,…}` author line). The font program (often
1838        // CFF, which this parser does not read) carries the same mapping;
1839        // docling-parse decodes it from there.
1840        tex_math_builtin(fdict).unwrap_or_else(winansi_table)
1841    } else {
1842        winansi_table()
1843    };
1844    // Apply /Differences: `code /glyphname /glyphname ... code ...`.
1845    if let Some(Object::Dictionary(d)) = enc {
1846        if let Some(Object::Array(diffs)) = d.get(b"Differences").ok().and_then(|o| deref(doc, o)) {
1847            let mut code = 0u8;
1848            for el in diffs {
1849                match el {
1850                    Object::Integer(i) => code = *i as u8,
1851                    Object::Name(name) => {
1852                        if let Some(ch) = glyph_name_to_char(name) {
1853                            m.insert(code, ch);
1854                        }
1855                        code = code.wrapping_add(1);
1856                    }
1857                    _ => {}
1858                }
1859            }
1860        }
1861    }
1862    m
1863}
1864
1865/// The fixed built-in encodings of the standard TeX math fonts (TeXbook
1866/// Appendix F), keyed off the base font name: `CMSY*` (symbols; `CMBSY` is its
1867/// bold) and `CMMI*` (math italic). These fonts ship no PDF `/Encoding` and no
1868/// ToUnicode, and their program is usually CFF — without this table the codes
1869/// fell through to StandardEncoding and rendered as the wrong ASCII.
1870fn tex_math_builtin(fdict: &Dictionary) -> Option<HashMap<u8, char>> {
1871    const CMSY: [char; 128] = [
1872        '−', '·', '×', '∗', '÷', '⋄', '±', '∓', '⊕', '⊖', '⊗', '⊘', '⊙', '◯', '∘', '•', '≍', '≡',
1873        '⊆', '⊇', '≤', '≥', '≼', '≽', '∼', '≈', '⊂', '⊃', '≪', '≫', '≺', '≻', '←', '→', '↑', '↓',
1874        '↔', '↗', '↘', '≃', '⇐', '⇒', '⇑', '⇓', '⇔', '↖', '↙', '∝', '′', '∞', '∈', '∋', '△', '▽',
1875        '\u{338}', '↦', '∀', '∃', '¬', '∅', 'ℜ', 'ℑ', '⊤', '⊥', 'ℵ', 'A', 'B', 'C', 'D', 'E', 'F',
1876        'G', 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X',
1877        'Y', 'Z', '∪', '∩', '⊎', '∧', '∨', '⊢', '⊣', '⌊', '⌋', '⌈', '⌉', '{', '}', '⟨', '⟩', '|',
1878        '∥', '↕', '⇕', '\\', '≀', '√', '∐', '∇', '∫', '⊔', '⊓', '⊑', '⊒', '§', '†', '‡', '¶', '♣',
1879        '♢', '♡', '♠',
1880    ];
1881    const CMMI: [char; 128] = [
1882        'Γ', 'Δ', 'Θ', 'Λ', 'Ξ', 'Π', 'Σ', 'Υ', 'Φ', 'Ψ', 'Ω', 'α', 'β', 'γ', 'δ', 'ε', 'ζ', 'η',
1883        'θ', 'ι', 'κ', 'λ', 'μ', 'ν', 'ξ', 'π', 'ρ', 'σ', 'τ', 'υ', 'φ', 'χ', 'ψ', 'ω', 'ϵ', 'ϑ',
1884        'ϖ', 'ϱ', 'ς', 'ϕ', '↼', '↽', '⇀', '⇁', '↩', '↪', '▷', '◁', '0', '1', '2', '3', '4', '5',
1885        '6', '7', '8', '9', '.', ',', '<', '/', '>', '⋆', '∂', 'A', 'B', 'C', 'D', 'E', 'F', 'G',
1886        'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y',
1887        'Z', '♭', '♮', '♯', '⌣', '⌢', 'ℓ', 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k',
1888        'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', 'ı', 'ȷ', '℘',
1889        '\u{20d7}', '⁀',
1890    ];
1891    let name = base_font_name(fdict)?;
1892    let up = name.to_ascii_uppercase();
1893    let table: &[char; 128] = if up.starts_with(b"CMSY") || up.starts_with(b"CMBSY") {
1894        &CMSY
1895    } else if up.starts_with(b"CMMI") {
1896        &CMMI
1897    } else {
1898        return None;
1899    };
1900    Some(
1901        table
1902            .iter()
1903            .enumerate()
1904            .map(|(i, &c)| (i as u8, c))
1905            .collect(),
1906    )
1907}
1908
1909/// Resolve an Adobe glyph name to Unicode: `uniXXXX`, single ASCII letters, the
1910/// digit/punctuation names from the Adobe Glyph List, and common typographic
1911/// names. A `.suffix` (`one.taboldstyle`, `a.sc`) is stripped and the base name
1912/// retried — docling renders these as the base character.
1913pub(crate) fn glyph_name_to_char(name: &[u8]) -> Option<char> {
1914    let s = std::str::from_utf8(name).ok()?;
1915    if let Some(hex) = s.strip_prefix("uni") {
1916        if let Ok(cp) = u32::from_str_radix(hex.get(0..4)?, 16) {
1917            return char::from_u32(cp);
1918        }
1919    }
1920    // Single ASCII letter names (`A`, `m`) map to themselves.
1921    if s.len() == 1 {
1922        let b = s.as_bytes()[0];
1923        if b.is_ascii_alphabetic() {
1924            return Some(b as char);
1925        }
1926    }
1927    let resolved = match s {
1928        "space" => ' ',
1929        "exclam" => '!',
1930        "quotedbl" => '"',
1931        "numbersign" => '#',
1932        "dollar" => '$',
1933        "percent" => '%',
1934        "ampersand" => '&',
1935        "quotesingle" => '\'',
1936        "parenleft" => '(',
1937        "parenright" => ')',
1938        "asterisk" => '*',
1939        "plus" => '+',
1940        "comma" => ',',
1941        "hyphen" => '-',
1942        "period" => '.',
1943        "slash" => '/',
1944        "zero" => '0',
1945        "one" => '1',
1946        "two" => '2',
1947        "three" => '3',
1948        "four" => '4',
1949        "five" => '5',
1950        "six" => '6',
1951        "seven" => '7',
1952        "eight" => '8',
1953        "nine" => '9',
1954        "colon" => ':',
1955        "semicolon" => ';',
1956        "less" => '<',
1957        "equal" => '=',
1958        "greater" => '>',
1959        "question" => '?',
1960        "at" => '@',
1961        "bracketleft" => '[',
1962        "backslash" => '\\',
1963        "bracketright" => ']',
1964        "asciicircum" => '^',
1965        "underscore" => '_',
1966        "grave" => '`',
1967        "braceleft" => '{',
1968        "bar" => '|',
1969        "braceright" => '}',
1970        "asciitilde" => '~',
1971        "bullet" => '\u{2022}',
1972        "periodcentered" => '\u{00B7}',
1973        "endash" => '\u{2013}',
1974        "emdash" => '\u{2014}',
1975        "quoteright" => '\u{2019}',
1976        "quoteleft" => '\u{2018}',
1977        "quotedblleft" => '\u{201C}',
1978        "quotedblright" => '\u{201D}',
1979        "quotedblbase" => '\u{201E}',
1980        "quotesinglbase" => '\u{201A}',
1981        // Latin f-ligatures named in `/Differences` (e.g. 2305's body font). These
1982        // map to the presentation-form code points, which `decompose_ligatures`
1983        // then spells back out (`ff`→"ff") — without them the glyph decodes to
1984        // nothing and the sanitizer fills the gap with a space (`di erences`).
1985        "ff" => '\u{FB00}',
1986        "fi" => '\u{FB01}',
1987        "fl" => '\u{FB02}',
1988        "ffi" => '\u{FB03}',
1989        "ffl" => '\u{FB04}',
1990        "ft" => '\u{FB05}',
1991        "st" => '\u{FB06}',
1992        "degree" => '\u{00B0}',
1993        "trademark" => '\u{2122}',
1994        "registered" => '\u{00AE}',
1995        "copyright" => '\u{00A9}',
1996        "ellipsis" => '\u{2026}',
1997        "minus" => '\u{2212}',
1998        "fraction" => '\u{2044}',
1999        "nbspace" => '\u{00A0}',
2000        // Greek + math glyph names (standard Adobe Glyph List). Standard TeX math
2001        // fonts (CMMI/CMSY/…) name their glyphs this way in the embedded font
2002        // program's `/Encoding`; without these a `λ`/`≤` decodes to nothing and is
2003        // dropped from body text (`and λ set to 0.5` → `and set to 0.5`).
2004        "alpha" => '\u{03B1}',
2005        "beta" => '\u{03B2}',
2006        "gamma" => '\u{03B3}',
2007        "delta" => '\u{03B4}',
2008        "epsilon" | "epsilon1" => '\u{03B5}',
2009        "zeta" => '\u{03B6}',
2010        "eta" => '\u{03B7}',
2011        "theta" | "theta1" => '\u{03B8}',
2012        "iota" => '\u{03B9}',
2013        "kappa" => '\u{03BA}',
2014        "lambda" => '\u{03BB}',
2015        "mu" => '\u{03BC}',
2016        "nu" => '\u{03BD}',
2017        "xi" => '\u{03BE}',
2018        "omicron" => '\u{03BF}',
2019        "pi" | "pi1" => '\u{03C0}',
2020        "rho" | "rho1" => '\u{03C1}',
2021        "sigma" => '\u{03C3}',
2022        "sigma1" => '\u{03C2}',
2023        "tau" => '\u{03C4}',
2024        "upsilon" => '\u{03C5}',
2025        "phi" | "phi1" => '\u{03C6}',
2026        "chi" => '\u{03C7}',
2027        "psi" => '\u{03C8}',
2028        "omega" | "omega1" => '\u{03C9}',
2029        "Gamma" => '\u{0393}',
2030        "Delta" => '\u{0394}',
2031        "Theta" => '\u{0398}',
2032        "Lambda" => '\u{039B}',
2033        "Xi" => '\u{039E}',
2034        "Pi" => '\u{03A0}',
2035        "Sigma" => '\u{03A3}',
2036        "Upsilon" => '\u{03A5}',
2037        "Phi" => '\u{03A6}',
2038        "Psi" => '\u{03A8}',
2039        "Omega" => '\u{03A9}',
2040        "lessequal" => '\u{2264}',
2041        "greaterequal" => '\u{2265}',
2042        "notequal" => '\u{2260}',
2043        "approxequal" => '\u{2248}',
2044        "equivalence" => '\u{2261}',
2045        "element" => '\u{2208}',
2046        "plusminus" => '\u{00B1}',
2047        "multiply" => '\u{00D7}',
2048        "divide" => '\u{00F7}',
2049        "infinity" => '\u{221E}',
2050        "partialdiff" => '\u{2202}',
2051        "gradient" => '\u{2207}',
2052        "summation" => '\u{2211}',
2053        "product" => '\u{220F}',
2054        "integral" => '\u{222B}',
2055        "radical" => '\u{221A}',
2056        "proportional" => '\u{221D}',
2057        "arrowright" => '\u{2192}',
2058        "arrowleft" => '\u{2190}',
2059        "arrowup" => '\u{2191}',
2060        "arrowdown" => '\u{2193}',
2061        "arrowboth" => '\u{2194}',
2062        "arrowdblright" => '\u{21D2}',
2063        "logicaland" => '\u{2227}',
2064        "logicalor" => '\u{2228}',
2065        "intersection" => '\u{2229}',
2066        "union" => '\u{222A}',
2067        "similar" => '\u{223C}',
2068        "congruent" => '\u{2245}',
2069        "dotmath" => '\u{22C5}',
2070        "asteriskmath" => '\u{2217}',
2071        _ => {
2072            // Strip an AGL `.suffix` (oldstyle/small-cap variant) and retry.
2073            if let Some((base, _)) = s.split_once('.') {
2074                if !base.is_empty() {
2075                    return glyph_name_to_char(base.as_bytes());
2076                }
2077            }
2078            return None;
2079        }
2080    };
2081    Some(resolved)
2082}
2083
2084/// Minimal WinAnsiEncoding (Latin-1-ish) for simple fonts lacking ToUnicode.
2085fn winansi_table() -> HashMap<u8, char> {
2086    let mut m = HashMap::new();
2087    for b in 0x20u8..=0x7e {
2088        m.insert(b, b as char);
2089    }
2090    // High range: Windows-1252 printable points that differ from Latin-1.
2091    let extra: &[(u8, char)] = &[
2092        (0x91, '\u{2018}'),
2093        (0x92, '\u{2019}'),
2094        (0x93, '\u{201C}'),
2095        (0x94, '\u{201D}'),
2096        (0x95, '\u{2022}'),
2097        (0x96, '\u{2013}'),
2098        (0x97, '\u{2014}'),
2099        (0x85, '\u{2026}'),
2100        (0xA0, '\u{00A0}'),
2101    ];
2102    for &(b, c) in extra {
2103        m.insert(b, c);
2104    }
2105    for b in 0xA1u8..=0xFF {
2106        m.entry(b).or_insert(b as char);
2107    }
2108    m
2109}
2110
2111/// Minimal MacRomanEncoding: ASCII plus the high-range points our corpus hits
2112/// (notably 0xA5 = bullet, used as a list marker).
2113fn macroman_table() -> HashMap<u8, char> {
2114    let mut m = HashMap::new();
2115    for b in 0x20u8..=0x7e {
2116        m.insert(b, b as char);
2117    }
2118    let high: &[(u8, char)] = &[
2119        (0xA5, '\u{2022}'), // bullet
2120        (0xD0, '\u{2013}'), // endash
2121        (0xD1, '\u{2014}'), // emdash
2122        (0xD2, '\u{201C}'),
2123        (0xD3, '\u{201D}'),
2124        (0xD4, '\u{2018}'),
2125        (0xD5, '\u{2019}'),
2126        (0xCA, '\u{00A0}'),
2127        (0xC9, '\u{2026}'),
2128        (0xDE, '\u{FB01}'),
2129        (0xDF, '\u{FB02}'),
2130    ];
2131    for &(b, c) in high {
2132        m.insert(b, c);
2133    }
2134    m
2135}
2136
2137#[cfg(test)]
2138mod page_box_frame {
2139    use super::*;
2140
2141    /// One page, `boxes` spliced into the page dictionary verbatim, one text
2142    /// run at user-space `(x, y)`.
2143    fn pdf(boxes: &str, x: f32, y: f32) -> Vec<u8> {
2144        let content = format!("BT /F1 12 Tf {x} {y} Td (First printing) Tj ET\n");
2145        let objs: Vec<String> = vec![
2146            "<</Type/Catalog/Pages 2 0 R>>".into(),
2147            format!("<</Type/Pages/Kids[3 0 R]/Count 1{boxes}>>"),
2148            "<</Type/Page/Parent 2 0 R/Contents 4 0 R/Resources<</Font<</F1 5 0 R>>>>>>".into(),
2149            format!("<</Length {}>>stream\n{content}endstream", content.len()),
2150            "<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".into(),
2151        ];
2152        let mut out = b"%PDF-1.4\n".to_vec();
2153        let mut offsets = Vec::new();
2154        for (i, body) in objs.iter().enumerate() {
2155            offsets.push(out.len());
2156            out.extend_from_slice(format!("{} 0 obj{body}endobj\n", i + 1).as_bytes());
2157        }
2158        let xref_at = out.len();
2159        out.extend_from_slice(
2160            format!("xref\n0 {}\n0000000000 65535 f \n", objs.len() + 1).as_bytes(),
2161        );
2162        for off in &offsets {
2163            out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2164        }
2165        out.extend_from_slice(
2166            format!(
2167                "trailer<</Size {}/Root 1 0 R>>\nstartxref\n{xref_at}\n%%EOF\n",
2168                objs.len() + 1
2169            )
2170            .as_bytes(),
2171        );
2172        out
2173    }
2174
2175    fn only_page(bytes: &[u8]) -> (PageBox, Vec<Glyph>) {
2176        let doc = load_document(bytes).expect("loads");
2177        let pid = *doc.get_pages().values().next().expect("one page");
2178        (page_box(&doc, pid), page_glyphs(&doc, pid))
2179    }
2180
2181    /// The trimmed-book-page shape: MediaBox and CropBox both away from the
2182    /// origin (inherited from `/Pages`). The display box is the CropBox, and a
2183    /// glyph's coordinates count from its lower-left corner — the same numbers
2184    /// the same text gets on a page whose CropBox *is* `[0 0 w h]`.
2185    #[test]
2186    fn glyphs_count_from_the_cropbox_corner_like_pdfium() {
2187        let (pb, shifted) = only_page(&pdf(
2188            "/MediaBox[-56.505 -58.25 576.303 723.31]/CropBox[1.095 -0.65 518.703 665.71]",
2189            37.0 + 1.095,
2190            58.0 - 0.65,
2191        ));
2192        assert!((pb.l - 1.095).abs() < 1e-3 && (pb.b + 0.65).abs() < 1e-3);
2193        assert!(
2194            (pb.w - 517.608).abs() < 1e-3 && (pb.h - 666.36).abs() < 1e-3,
2195            "{pb:?}"
2196        );
2197        let (pb0, plain) = only_page(&pdf("/MediaBox[0 0 517.608 666.36]", 37.0, 58.0));
2198        assert!((pb0.w - pb.w).abs() < 1e-3 && (pb0.h - pb.h).abs() < 1e-3);
2199        assert_eq!(shifted.len(), plain.len());
2200        assert!(!plain.is_empty());
2201        for (a, b) in shifted.iter().zip(&plain) {
2202            assert!(
2203                (a.l - b.l).abs() < 1e-3 && (a.b - b.b).abs() < 1e-3,
2204                "{:?} vs {:?}",
2205                (a.l, a.b),
2206                (b.l, b.b)
2207            );
2208        }
2209        assert!((plain[0].l - 37.0).abs() < 1e-3, "{}", plain[0].l);
2210    }
2211
2212    /// pdfium's fallbacks: no MediaBox → Letter; a CropBox is clipped to the
2213    /// MediaBox, and one that misses it entirely is ignored.
2214    #[test]
2215    fn page_box_follows_pdfium_fallbacks() {
2216        let (pb, _) = only_page(&pdf("", 10.0, 10.0));
2217        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 612.0, 792.0));
2218        let (pb, _) = only_page(&pdf(
2219            "/MediaBox[0 0 500 700]/CropBox[-100 100 600 900]",
2220            10.0,
2221            10.0,
2222        ));
2223        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 100.0, 500.0, 600.0));
2224        let (pb, _) = only_page(&pdf(
2225            "/MediaBox[0 0 500 700]/CropBox[800 800 900 900]",
2226            10.0,
2227            10.0,
2228        ));
2229        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 500.0, 700.0));
2230        // Reversed corners normalize.
2231        let (pb, _) = only_page(&pdf("/MediaBox[500 700 0 0]", 10.0, 10.0));
2232        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 500.0, 700.0));
2233    }
2234}
2235
2236#[cfg(test)]
2237mod xref_repair {
2238    /// Build a tiny one-page PDF whose cross-reference entries are either the
2239    /// spec's 20 bytes (`two_byte_eol`) or the 19-byte form some generators
2240    /// emit — everything else about the two files is identical.
2241    fn pdf_with_xref(two_byte_eol: bool) -> Vec<u8> {
2242        let content = b"BT /F1 12 Tf 72 700 Td (Invoice 922769430725) Tj ET\n";
2243        let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2244        let objs: Vec<Vec<u8>> = vec![
2245            b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2246            b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2247            b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2248               /Resources<</Font<</F1 5 0 R>>>>>>"
2249                .to_vec(),
2250            [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
2251            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".to_vec(),
2252        ];
2253
2254        let mut out = b"%PDF-1.4\n".to_vec();
2255        let mut offsets = Vec::new();
2256        for (i, body) in objs.iter().enumerate() {
2257            offsets.push(out.len());
2258            out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2259            out.extend_from_slice(body);
2260            out.extend_from_slice(b"endobj\n");
2261        }
2262        let xref_at = out.len();
2263        let eol: &[u8] = if two_byte_eol { b" \n" } else { b"\n" };
2264        out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2265        out.extend_from_slice(b"0000000000 65535 f");
2266        out.extend_from_slice(eol);
2267        for off in &offsets {
2268            out.extend_from_slice(format!("{off:010} 00000 n").as_bytes());
2269            out.extend_from_slice(eol);
2270        }
2271        out.extend_from_slice(
2272            format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2273        );
2274        out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2275        out
2276    }
2277
2278    /// A 19-byte cross-reference entry (a bare LF where the spec wants a
2279    /// two-byte EOL) makes lopdf reject the whole file, so a readable text
2280    /// layer used to look exactly like a scan — in the browser that meant ten
2281    /// seconds of OCR for nothing. The repair must recover *the same* parse the
2282    /// well-formed file gives.
2283    #[test]
2284    fn short_xref_entries_still_parse() {
2285        let good = pdf_with_xref(true);
2286        let broken = pdf_with_xref(false);
2287        assert!(
2288            broken.len() < good.len(),
2289            "the broken file is the shorter one"
2290        );
2291        assert!(
2292            lopdf::Document::load_mem(&good).is_ok(),
2293            "the control file must load unaided"
2294        );
2295        assert!(
2296            lopdf::Document::load_mem(&broken).is_err(),
2297            "lopdf rejects 19-byte entries — if this ever passes, drop the repair"
2298        );
2299
2300        let cells = |b: &[u8]| -> Vec<String> {
2301            super::pdf_textlines(b)
2302                .into_iter()
2303                .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2304                .collect()
2305        };
2306        let from_good = cells(&good);
2307        assert!(
2308            from_good.iter().any(|t| t.contains("922769430725")),
2309            "control text: {from_good:?}"
2310        );
2311        assert_eq!(
2312            cells(&broken),
2313            from_good,
2314            "repair must match the good parse"
2315        );
2316    }
2317
2318    /// The same generator overstates `/Length`, so lopdf reads past the data,
2319    /// misses `endstream` and drops the stream — the object comes back as a
2320    /// bare dictionary and the page has no content at all. Trust `endstream`
2321    /// instead, and do it without moving a single byte.
2322    #[test]
2323    fn overstated_stream_length_still_yields_content() {
2324        let good = pdf_with_xref(true);
2325        // Inflate the content stream's /Length by one, exactly as the invoice
2326        // that prompted this does.
2327        let broken = {
2328            let at = good
2329                .windows(8)
2330                .position(|w| w == b"/Length ")
2331                .expect("a /Length")
2332                + 8;
2333            let digits = good[at..].iter().take_while(|c| c.is_ascii_digit()).count();
2334            let n: usize = std::str::from_utf8(&good[at..at + digits])
2335                .unwrap()
2336                .parse()
2337                .unwrap();
2338            let inflated = (n + 1).to_string();
2339            assert_eq!(inflated.len(), digits, "keep the digit count");
2340            let mut b = good.clone();
2341            b[at..at + digits].copy_from_slice(inflated.as_bytes());
2342            b
2343        };
2344        assert_eq!(broken.len(), good.len(), "the defect must not move bytes");
2345        // lopdf alone loses the stream: the page parses but carries no content.
2346        let raw = lopdf::Document::load_mem(&broken).expect("still loads");
2347        assert!(
2348            raw.get_pages()
2349                .into_values()
2350                .all(|p| raw.get_page_content(p).is_empty()),
2351            "lopdf should drop the stream — if it stops, drop this repair"
2352        );
2353        // Ours recovers the same text the well-formed file gives.
2354        let text = |b: &[u8]| -> Vec<String> {
2355            super::pdf_textlines(b)
2356                .into_iter()
2357                .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2358                .collect()
2359        };
2360        let expected = text(&good);
2361        assert!(!expected.is_empty(), "control must produce text");
2362        assert_eq!(text(&broken), expected);
2363    }
2364
2365    /// The repair only fires where padding cannot move an object: it declines a
2366    /// file whose xref precedes an object (an incremental update), rather than
2367    /// shifting every offset the table records.
2368    #[test]
2369    fn repair_declines_when_padding_would_move_objects() {
2370        let mut incremental = pdf_with_xref(false);
2371        incremental.extend_from_slice(b"6 0 obj<</Type/Whatever>>endobj\n");
2372        let declined = super::pad_short_xref_entries(&incremental).unwrap_err();
2373        assert!(
2374            declined.contains("object follows the xref"),
2375            "reason: {declined}"
2376        );
2377    }
2378}
2379
2380/// #187: standard-14 fonts referenced without an embedded program (and thus
2381/// usually without `/Widths` or a `/FontDescriptor`) must decode with the
2382/// built-in Adobe Core 14 metrics instead of collapsing every cell to zero
2383/// width — the failure mode where a valid text layer was silently dropped
2384/// while pdfium read the same file fine.
2385#[cfg(test)]
2386mod base14_fonts {
2387    /// A one-page PDF whose single `Tj` uses `fontdict` (no embedded program).
2388    fn pdf_with_font(fontdict: &[u8], text: &[u8]) -> Vec<u8> {
2389        let content = [b"BT /F1 12 Tf 72 700 Td (".as_slice(), text, b") Tj ET\n"].concat();
2390        let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2391        let objs: Vec<Vec<u8>> = vec![
2392            b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2393            b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2394            b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2395               /Resources<</Font<</F1 5 0 R>>>>>>"
2396                .to_vec(),
2397            [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
2398            fontdict.to_vec(),
2399        ];
2400        let mut out = b"%PDF-1.4\n".to_vec();
2401        let mut offsets = Vec::new();
2402        for (i, body) in objs.iter().enumerate() {
2403            offsets.push(out.len());
2404            out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2405            out.extend_from_slice(body);
2406            out.extend_from_slice(b"endobj\n");
2407        }
2408        let xref_at = out.len();
2409        out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2410        out.extend_from_slice(b"0000000000 65535 f \n");
2411        for off in &offsets {
2412            out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2413        }
2414        out.extend_from_slice(
2415            format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2416        );
2417        out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2418        out
2419    }
2420
2421    /// The parsed cells of the only page.
2422    fn cells(pdf: &[u8]) -> Vec<crate::pdfium_backend::TextCell> {
2423        super::pdf_textlines(pdf)
2424            .into_iter()
2425            .flat_map(|(_, _, c)| c)
2426            .collect()
2427    }
2428
2429    /// Every standard-14 alias/style decodes with real (positive-width) boxes.
2430    #[test]
2431    fn standard14_faces_get_builtin_widths() {
2432        for fontdict in [
2433            // ReportLab's default: base-14 Helvetica, WinAnsi, nothing else.
2434            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>".as_slice(),
2435            // No /Encoding at all (StandardEncoding-ish default).
2436            b"<</Type/Font/Subtype/Type1/BaseFont/Times-BoldItalic>>",
2437            // Substitution aliases + a subset prefix.
2438            b"<</Type/Font/Subtype/TrueType/BaseFont/Arial,Bold>>",
2439            b"<</Type/Font/Subtype/Type1/BaseFont/ABCDEF+Courier-Oblique>>",
2440        ] {
2441            let pdf = pdf_with_font(fontdict, b"Words have width now");
2442            let cs = cells(&pdf);
2443            let text: String = cs
2444                .iter()
2445                .map(|c| c.text.as_str())
2446                .collect::<Vec<_>>()
2447                .join(" ");
2448            assert!(
2449                text.contains("Words have width now"),
2450                "{}: text lost: {text:?}",
2451                String::from_utf8_lossy(fontdict)
2452            );
2453            assert!(
2454                cs.iter().all(|c| c.r > c.l),
2455                "{}: zero-width cells: {cs:?}",
2456                String::from_utf8_lossy(fontdict)
2457            );
2458        }
2459    }
2460
2461    /// An explicit `/Widths` array always wins over the built-in metrics, and a
2462    /// non-standard face without `/Widths` stays as before (no invented boxes).
2463    #[test]
2464    fn explicit_widths_win_and_unknown_faces_are_untouched() {
2465        // Helvetica with explicit 100/1000-em widths: the word's box must be
2466        // ~4×100 units at 12pt = 4.8pt wide — far narrower than the ~2.7×
2467        // wider built-in Helvetica advances would make it.
2468        let explicit = pdf_with_font(
2469            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/FirstChar 65\
2470               /Widths[100 100 100 100]/Encoding/WinAnsiEncoding>>",
2471            b"ABBA",
2472        );
2473        let builtin = pdf_with_font(
2474            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>",
2475            b"ABBA",
2476        );
2477        let w = |pdf: &[u8]| {
2478            let cs = cells(pdf);
2479            assert_eq!(cs.len(), 1, "one word cell: {cs:?}");
2480            cs[0].r - cs[0].l
2481        };
2482        let (we, wb) = (w(&explicit), w(&builtin));
2483        assert!(
2484            (we - 4.8).abs() < 0.1,
2485            "explicit widths must win: got {we}, want 4×100×12/1000"
2486        );
2487        assert!(
2488            wb > 2.0 * we,
2489            "built-in Helvetica is much wider: {wb} vs {we}"
2490        );
2491
2492        // An unknown face with no /Widths: still parses (text kept), but no
2493        // built-in table applies — the old zero-width behavior is preserved
2494        // rather than inventing Helvetica metrics for an arbitrary font.
2495        let unknown = pdf_with_font(
2496            b"<</Type/Font/Subtype/Type1/BaseFont/FancyCorp-Display>>",
2497            b"Mystery",
2498        );
2499        let cs = cells(&unknown);
2500        let text: String = cs.iter().map(|c| c.text.as_str()).collect();
2501        assert!(text.contains("Mystery"), "text still decodes: {cs:?}");
2502    }
2503}
2504
2505#[cfg(test)]
2506mod overpainted {
2507    use crate::pdfium_backend::TextCell;
2508
2509    fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
2510        TextCell {
2511            text: text.into(),
2512            l,
2513            t,
2514            r,
2515            b,
2516        }
2517    }
2518
2519    /// The reporting invoice's logo: a `"` painted inside a `==` on one band —
2520    /// artwork drawn with glyphs. Both cells go; the real text on the next
2521    /// band stays.
2522    #[test]
2523    fn stacked_logo_glyphs_are_dropped() {
2524        let mut cells = vec![
2525            cell("\"", 72.7, 21.5, 86.4, 31.5),
2526            cell("==", 59.4, 21.5, 99.6, 31.5),
2527            cell("Herr", 65.2, 151.3, 81.7, 161.3),
2528        ];
2529        super::drop_overpainted_cells(&mut cells);
2530        assert_eq!(cells.len(), 1, "cells: {cells:?}");
2531        assert_eq!(cells[0].text, "Herr");
2532    }
2533
2534    /// Adjacent words on a line touch but never contain each other — prose is
2535    /// untouched, and so is a same-text near-duplicate (double-drawn faux
2536    /// bold), which is not evidence of artwork.
2537    #[test]
2538    fn prose_and_double_draw_are_kept() {
2539        let mut cells = vec![
2540            cell("Telefon", 354.3, 133.2, 381.5, 143.2),
2541            cell("0676/2000", 387.3, 133.2, 428.7, 143.2),
2542            cell("Bold", 100.0, 50.0, 130.0, 60.0),
2543            cell("Bold", 100.3, 50.0, 130.3, 60.0),
2544        ];
2545        super::drop_overpainted_cells(&mut cells);
2546        assert_eq!(cells.len(), 4);
2547    }
2548}
2549
2550#[cfg(test)]
2551mod vestigial_layer {
2552    use crate::pdfium_backend::{PdfPage, TextCell};
2553
2554    fn page_with(texts: &[&str]) -> PdfPage {
2555        let cells = texts
2556            .iter()
2557            .enumerate()
2558            .map(|(i, t)| TextCell {
2559                text: t.to_string(),
2560                l: 10.0,
2561                t: 10.0 + 12.0 * i as f32,
2562                r: 90.0,
2563                b: 20.0 + 12.0 * i as f32,
2564            })
2565            .collect();
2566        PdfPage::from_cells(595.0, 842.0, 1.0, cells)
2567    }
2568
2569    /// The reported scanned form: three typed-in field values ("03", "05",
2570    /// "2025") over three image pages. That must read as *no usable layer*,
2571    /// so the browser routes the document to OCR instead of extracting
2572    /// thirteen characters and skipping the letter entirely.
2573    #[test]
2574    fn typed_in_form_fields_are_not_a_text_layer() {
2575        let pages = vec![
2576            page_with(&["03", "05", "2025"]),
2577            page_with(&[]),
2578            page_with(&[]),
2579        ];
2580        assert!(super::text_layer_is_vestigial(&pages));
2581        assert!(super::text_layer_is_vestigial(&[page_with(&[])]));
2582    }
2583
2584    /// A short but genuine digital document — one page, a few real lines —
2585    /// keeps the fast text path.
2586    #[test]
2587    fn sparse_but_real_documents_pass() {
2588        let one_pager = vec![page_with(&[
2589            "Confidential briefing",
2590            "Prepared for the board meeting",
2591            "Do not distribute",
2592        ])];
2593        assert!(!super::text_layer_is_vestigial(&one_pager));
2594    }
2595}