Skip to main content

docling_pdf/
textparse.rs

1//! Pure-Rust PDF text extraction (replacing pdfium's glyph layer).
2//!
3//! pdfium reports *rendered* glyph boxes, which diverge from docling's
4//! `docling-parse` C++ parser at exactly the points that drive conformance:
5//! generated spaces get a zero-width box, combining diacritics get a real-width
6//! box, and ligature/fraction glyphs land at different x. This module instead
7//! reconstructs each glyph's box from the **font's own advance widths** and the
8//! PDF text/graphics matrices — the same information docling-parse uses — so a
9//! space is as wide as the font says and a combining mark has zero advance.
10//!
11//! The output is the same [`Glyph`] stream pdfium produces (native PDF
12//! coordinates, y-up), fed straight into the existing docling-parse line
13//! sanitizer ([`crate::dp_lines`]). Only the digital text layer is handled here;
14//! pages without one still fall back to OCR upstream.
15
16use std::collections::HashMap;
17use std::rc::Rc;
18
19use lopdf::{Dictionary, Document, Object};
20
21use crate::pdfium_backend::Glyph;
22
23/// Per-document caches for the content-stream interpreter. Fonts are indirect
24/// objects shared by many pages, but were fully re-parsed — ToUnicode CMap
25/// decompression + tokenization, embedded Type1 program scan, width tables —
26/// for **every page and every Form XObject invocation**; decoded form content
27/// streams were likewise re-inflated on every `Do`. Cached per document,
28/// keyed by the referenced object id (fonts also by resource name, which
29/// feeds the docling-parse font hash). Inline (non-reference) dicts are rare
30/// and stay uncached.
31#[derive(Default)]
32struct DocCaches {
33    fonts: HashMap<(lopdf::ObjectId, Vec<u8>), Rc<Font>>,
34    forms: HashMap<lopdf::ObjectId, Rc<lopdf::content::Content>>,
35}
36
37/// A 2×3 affine matrix `[a b c d e f]`: maps `(x,y)` → `(a·x+c·y+e, b·x+d·y+f)`.
38#[derive(Clone, Copy)]
39struct Mat {
40    a: f64,
41    b: f64,
42    c: f64,
43    d: f64,
44    e: f64,
45    f: f64,
46}
47
48impl Mat {
49    const ID: Mat = Mat {
50        a: 1.0,
51        b: 0.0,
52        c: 0.0,
53        d: 1.0,
54        e: 0.0,
55        f: 0.0,
56    };
57
58    /// `self ∘ m`: the matrix that applies `self` first, then `m`.
59    fn then(self, m: Mat) -> Mat {
60        Mat {
61            a: self.a * m.a + self.b * m.c,
62            b: self.a * m.b + self.b * m.d,
63            c: self.c * m.a + self.d * m.c,
64            d: self.c * m.b + self.d * m.d,
65            e: self.e * m.a + self.f * m.c + m.e,
66            f: self.e * m.b + self.f * m.d + m.f,
67        }
68    }
69
70    fn apply(self, x: f64, y: f64) -> (f64, f64) {
71        (
72            self.a * x + self.c * y + self.e,
73            self.b * x + self.d * y + self.f,
74        )
75    }
76}
77
78/// A parsed font: how to turn raw string bytes into (unicode, advance) pairs.
79struct Font {
80    /// 2-byte codes (Type0 / Identity-H) vs 1-byte (simple fonts).
81    two_byte: bool,
82    /// code → Unicode string (from ToUnicode; may be multi-char, e.g. ligatures).
83    to_unicode: HashMap<u32, String>,
84    /// code → glyph advance, in 1000-unit glyph space.
85    widths: HashMap<u32, f64>,
86    default_width: f64,
87    /// 1-byte fallback decoding when ToUnicode lacks a code (WinAnsi-ish).
88    simple_encoding: Option<HashMap<u8, char>>,
89    /// code → raw `/Differences` glyph name, for GID-style names (`g115`) that
90    /// have no Unicode mapping. docling-parse emits these verbatim as `/g115`
91    /// (see the redp5110 bulleted list); matching it keeps no text skipped.
92    fallback_names: HashMap<u8, String>,
93    /// code → char from the embedded Type1 font program's own `/Encoding` vector,
94    /// used only as a last resort for glyphs the base encoding leaves unmapped
95    /// (standard TeX math fonts: `λ`, `≤`, …).
96    program_encoding: HashMap<u8, char>,
97    ascent: f64,
98    descent: f64,
99    hash: u64,
100}
101
102impl Font {
103    fn decode_code(&self, code: u32) -> (Option<String>, f64) {
104        let w = self
105            .widths
106            .get(&code)
107            .copied()
108            .unwrap_or(self.default_width);
109        if let Some(s) = self.to_unicode.get(&code) {
110            return (Some(decompose_ligatures(s)), w);
111        }
112        if !self.two_byte {
113            // A GID-style `/Differences` name (no Unicode) overrides the base
114            // encoding, matching docling's verbatim `/g115` fallback.
115            if let Some(name) = self.fallback_names.get(&(code as u8)) {
116                return (Some(format!("/{name}")), w);
117            }
118            if let Some(enc) = &self.simple_encoding {
119                if let Some(&ch) = enc.get(&(code as u8)) {
120                    return (Some(decompose_ligatures(&ch.to_string())), w);
121                }
122            }
123            // Last resort: the embedded Type1 font program's own `/Encoding`
124            // vector (`dup N /glyphname put`). Standard TeX math fonts (CMMI, CMSY,
125            // …) ship no PDF `/Encoding` and no ToUnicode, so a glyph like `λ`
126            // (CMMI code 21 → `/lambda`) or `≤` (CMSY code 20 → `/lessequal`) has
127            // no other mapping and would otherwise be silently dropped. docling
128            // recovers these from the same font program. This only fills codes the
129            // base encoding left unmapped, so it never changes an existing decode.
130            if let Some(&ch) = self.program_encoding.get(&(code as u8)) {
131                return (Some(decompose_ligatures(&ch.to_string())), w);
132            }
133        }
134        (None, w)
135    }
136}
137
138/// Spell out Latin presentation-form ligatures (`fi`→`fi`, `ffi`→`ffi`, …) the way
139/// docling does, so `configuration`/`difficult` don't keep the ligature glyph.
140/// The chars share the ligature's box, so the line sanitizer recomposes them.
141fn decompose_ligatures(s: &str) -> String {
142    if !s.chars().any(|c| ('\u{FB00}'..='\u{FB06}').contains(&c)) {
143        return s.to_string();
144    }
145    s.chars()
146        .map(|c| {
147            match c {
148                '\u{FB00}' => "ff",
149                '\u{FB01}' => "fi",
150                '\u{FB02}' => "fl",
151                '\u{FB03}' => "ffi",
152                '\u{FB04}' => "ffl",
153                '\u{FB05}' => "ft",
154                '\u{FB06}' => "st",
155                _ => return c.to_string(),
156            }
157            .to_string()
158        })
159        .collect()
160}
161
162fn hash_name(name: &[u8]) -> u64 {
163    use std::hash::{Hash, Hasher};
164    let mut h = std::collections::hash_map::DefaultHasher::new();
165    name.hash(&mut h);
166    h.finish()
167}
168
169/// Resolve a possibly-indirect object to a dictionary.
170fn as_dict<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Dictionary> {
171    match obj {
172        Object::Dictionary(d) => Some(d),
173        Object::Reference(id) => doc.get_object(*id).ok().and_then(|o| o.as_dict().ok()),
174        _ => None,
175    }
176}
177
178fn deref<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Object> {
179    match obj {
180        Object::Reference(id) => doc.get_object(*id).ok(),
181        other => Some(other),
182    }
183}
184
185/// Parse one font dictionary into a [`Font`].
186fn parse_font(doc: &Document, name: &[u8], fdict: &Dictionary) -> Font {
187    let subtype: &[u8] = fdict
188        .get(b"Subtype")
189        .ok()
190        .and_then(|o| o.as_name().ok())
191        .unwrap_or(&[]);
192    let two_byte = subtype == b"Type0".as_slice();
193
194    let to_unicode = fdict
195        .get(b"ToUnicode")
196        .ok()
197        .and_then(|o| deref(doc, o))
198        .and_then(|o| o.as_stream().ok())
199        .and_then(|s| s.decompressed_content().ok())
200        .map(|data| parse_tounicode(&data))
201        .unwrap_or_default();
202
203    let (mut widths, mut default_width) = if two_byte {
204        cid_widths(doc, fdict)
205    } else {
206        simple_widths(doc, fdict)
207    };
208
209    let simple_encoding = if two_byte {
210        None
211    } else {
212        Some(simple_encoding_table(doc, fdict))
213    };
214
215    // A standard-14 font referenced without an embedded program usually ships
216    // no `/Widths` and no `/FontDescriptor` either (ReportLab's default, #187).
217    // Every advance then resolved to 0, the cells collapsed to zero width, and
218    // the page's whole text layer was silently dropped — while pdfium, with
219    // its built-in metrics, reads the same file fine. Fill the widths from the
220    // built-in Adobe Core 14 AFM tables via the font's own code→char decode
221    // (base encoding + `/Differences`), so an explicit `/Widths` always wins.
222    if !two_byte && widths.is_empty() && default_width == 0.0 {
223        if let Some(std14) = base_font_name(fdict).and_then(|n| crate::std14::widths_for(&n)) {
224            if let Some(enc) = &simple_encoding {
225                for (&code, &ch) in enc {
226                    if let Some(w) = std14.width(ch) {
227                        widths.insert(u32::from(code), w);
228                    }
229                }
230            }
231            // Codes the table misses still advance a typical width instead of
232            // stacking at x=0 (the failure mode this whole branch fixes).
233            default_width = 500.0;
234        }
235    }
236    let fallback_names = if two_byte {
237        HashMap::new()
238    } else {
239        differences_gid_names(doc, fdict)
240    };
241    let program_encoding = if two_byte {
242        HashMap::new()
243    } else {
244        type1_program_encoding(doc, fdict)
245    };
246
247    let (ascent, descent) = font_ascent_descent(doc, fdict, two_byte);
248
249    Font {
250        two_byte,
251        to_unicode,
252        widths,
253        default_width,
254        simple_encoding,
255        fallback_names,
256        program_encoding,
257        ascent,
258        descent,
259        hash: hash_name(name),
260    }
261}
262
263/// Collect `/Differences` entries whose glyph name is a GID placeholder
264/// (`g115`, `cid42`, `glyph7`, `index9`) with no Unicode mapping. docling-parse
265/// emits such glyphs as the literal name `/g115`; mapping them here keeps the
266/// text from being silently dropped (subsetted fonts with no ToUnicode). The
267/// GID-name restriction keeps real Adobe glyph names on the normal path so this
268/// never invents garbage on the clean files.
269fn differences_gid_names(doc: &Document, fdict: &Dictionary) -> HashMap<u8, String> {
270    let mut map = HashMap::new();
271    let Some(Object::Dictionary(enc)) = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o))
272    else {
273        return map;
274    };
275    let Some(Object::Array(diffs)) = enc.get(b"Differences").ok().and_then(|o| deref(doc, o))
276    else {
277        return map;
278    };
279    let mut code = 0u8;
280    for el in diffs {
281        match el {
282            Object::Integer(i) => code = *i as u8,
283            Object::Name(name) => {
284                if glyph_name_to_char(name).is_none() && is_gid_name(name) {
285                    map.insert(code, String::from_utf8_lossy(name).into_owned());
286                }
287                code = code.wrapping_add(1);
288            }
289            _ => {}
290        }
291    }
292    map
293}
294
295/// Parse the embedded Type1 font program's built-in `/Encoding` vector
296/// (`dup <code> /<glyphname> put` entries in the clear-text header before
297/// `eexec`) into `code → char`. This is how docling recovers glyphs from
298/// standard TeX math fonts (CMMI/CMSY/…) that carry no PDF `/Encoding` and no
299/// ToUnicode — e.g. CMMI's `dup 21 /lambda` or CMSY's `dup 20 /lessequal`.
300/// Only `FontFile` (Type1) is parsed; CFF (`FontFile3`) and TrueType
301/// (`FontFile2`) store their encoding in a binary table and are left alone.
302fn type1_program_encoding(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
303    let mut map = HashMap::new();
304    let Some(desc) = fdict
305        .get(b"FontDescriptor")
306        .ok()
307        .and_then(|o| deref(doc, o))
308        .and_then(|o| o.as_dict().ok())
309    else {
310        return map;
311    };
312    let Some(data) = desc
313        .get(b"FontFile")
314        .ok()
315        .and_then(|o| deref(doc, o))
316        .and_then(|o| o.as_stream().ok())
317        .and_then(|s| s.decompressed_content().ok())
318    else {
319        return map;
320    };
321    // The clear-text header (PostScript) ends at `eexec`; the rest is encrypted.
322    let head_end = data
323        .windows(5)
324        .position(|w| w == b"eexec")
325        .unwrap_or(data.len());
326    let head = String::from_utf8_lossy(&data[..head_end]);
327    // Scan for `dup <code> /<name> put` tokens.
328    let toks: Vec<&str> = head.split_whitespace().collect();
329    for w in toks.windows(4) {
330        if w[0] == "dup" && w[3] == "put" {
331            if let (Ok(code), Some(name)) = (w[1].parse::<u32>(), w[2].strip_prefix('/')) {
332                if code <= 255 {
333                    if let Some(ch) = glyph_name_to_char(name.as_bytes()) {
334                        map.insert(code as u8, ch);
335                    }
336                }
337            }
338        }
339    }
340    map
341}
342
343/// A glyph name that is a synthetic placeholder, not a real Adobe name:
344/// `g115`, `cid42`, `glyph7`, `index9`, `G12`, or a short-prefix code name like
345/// `SM590000` (IBM BookMaster). These carry no Unicode meaning, and docling-parse
346/// emits them verbatim (`/SM590000`). `afii####` / `uni####` are real Adobe names
347/// and excluded. The restriction keeps genuine glyph names on the Unicode path.
348fn is_gid_name(name: &[u8]) -> bool {
349    let Ok(s) = std::str::from_utf8(name) else {
350        return false;
351    };
352    if s.starts_with("afii") || s.starts_with("uni") {
353        return false;
354    }
355    for prefix in ["g", "G", "cid", "CID", "glyph", "index"] {
356        if let Some(rest) = s.strip_prefix(prefix) {
357            if !rest.is_empty() && rest.bytes().all(|b| b.is_ascii_digit()) {
358                return true;
359            }
360        }
361    }
362    // Short alpha prefix (≤3 letters) followed by a run of ≥3 digits — synthetic
363    // code names like `SM590000`, distinct from real Adobe names (whole words or
364    // letter+`.suffix` variants).
365    let alpha = s.bytes().take_while(|b| b.is_ascii_alphabetic()).count();
366    let digits = s.len() - alpha;
367    (1..=3).contains(&alpha)
368        && digits >= 3
369        && s.as_bytes()[alpha..].iter().all(|b| b.is_ascii_digit())
370}
371
372fn font_ascent_descent(doc: &Document, fdict: &Dictionary, two_byte: bool) -> (f64, f64) {
373    // For Type0, the descriptor lives on the descendant CIDFont.
374    let descr_owner = if two_byte {
375        fdict
376            .get(b"DescendantFonts")
377            .ok()
378            .and_then(|o| deref(doc, o))
379            .and_then(|o| match o {
380                Object::Array(a) => a.first(),
381                _ => None,
382            })
383            .and_then(|o| as_dict(doc, o))
384    } else {
385        Some(fdict)
386    };
387    let fd = descr_owner
388        .and_then(|d| d.get(b"FontDescriptor").ok())
389        .and_then(|o| as_dict(doc, o));
390    let asc = fd
391        .and_then(|d| d.get(b"Ascent").ok())
392        .and_then(|o| {
393            o.as_float()
394                .ok()
395                .or_else(|| o.as_i64().ok().map(|i| i as f32))
396        })
397        .unwrap_or(750.0) as f64;
398    let desc = fd
399        .and_then(|d| d.get(b"Descent").ok())
400        .and_then(|o| {
401            o.as_float()
402                .ok()
403                .or_else(|| o.as_i64().ok().map(|i| i as f32))
404        })
405        .unwrap_or(-250.0) as f64;
406    // Some subsetted fonts carry a degenerate FontDescriptor (`/Ascent 0
407    // /Descent 0`) — the real metrics live in the font program. That collapses
408    // the loose box to zero height, so the line cells get zero area and the
409    // layout's region/text assignment drops them (2305's References list lost
410    // every prose line, keeping only the URLs). Fall back to typical text metrics
411    // so the box has height.
412    if asc - desc <= 1.0 {
413        return (750.0, -250.0);
414    }
415    (asc, desc)
416}
417
418/// The `/BaseFont` name with any `ABCDEF+` subset prefix stripped (#187).
419fn base_font_name(fdict: &Dictionary) -> Option<Vec<u8>> {
420    let name = fdict.get(b"BaseFont").ok()?.as_name().ok()?;
421    let stripped = match name.iter().position(|&b| b == b'+') {
422        Some(i) if i == 6 => &name[i + 1..],
423        _ => name,
424    };
425    Some(stripped.to_vec())
426}
427
428/// Simple-font widths: `/FirstChar` + `/Widths` array, `/MissingWidth` default.
429fn simple_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
430    let mut map = HashMap::new();
431    let first = fdict
432        .get(b"FirstChar")
433        .ok()
434        .and_then(|o| o.as_i64().ok())
435        .unwrap_or(0) as u32;
436    if let Some(Object::Array(arr)) = fdict.get(b"Widths").ok().and_then(|o| deref(doc, o)) {
437        for (i, w) in arr.iter().enumerate() {
438            if let Some(w) = num(w) {
439                map.insert(first + i as u32, w);
440            }
441        }
442    }
443    let dw = fdict
444        .get(b"FontDescriptor")
445        .ok()
446        .and_then(|o| as_dict(doc, o))
447        .and_then(|d| d.get(b"MissingWidth").ok())
448        .and_then(num)
449        .unwrap_or(0.0);
450    (map, dw)
451}
452
453/// CIDFont widths: the `/W` array on the descendant font (`/DW` default = 1000).
454fn cid_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
455    let mut map = HashMap::new();
456    let Some(desc) = fdict
457        .get(b"DescendantFonts")
458        .ok()
459        .and_then(|o| deref(doc, o))
460        .and_then(|o| match o {
461            Object::Array(a) => a.first(),
462            _ => None,
463        })
464        .and_then(|o| as_dict(doc, o))
465    else {
466        return (map, 1000.0);
467    };
468    let dw = desc.get(b"DW").ok().and_then(num).unwrap_or(1000.0);
469    if let Some(Object::Array(w)) = desc.get(b"W").ok().and_then(|o| deref(doc, o)) {
470        let mut i = 0;
471        while i < w.len() {
472            let c = w.get(i).and_then(num);
473            match (c, w.get(i + 1)) {
474                // `c [w1 w2 ...]`: consecutive CIDs starting at c.
475                (Some(c), Some(Object::Array(list))) => {
476                    for (k, wv) in list.iter().enumerate() {
477                        if let Some(wv) = num(wv) {
478                            map.insert(c as u32 + k as u32, wv);
479                        }
480                    }
481                    i += 2;
482                }
483                // `c_first c_last w`: a run all of width w.
484                (Some(c1), Some(o2)) => {
485                    if let (Some(c2), Some(wv)) = (num(o2), w.get(i + 2).and_then(num)) {
486                        for cid in c1 as u32..=c2 as u32 {
487                            map.insert(cid, wv);
488                        }
489                    }
490                    i += 3;
491                }
492                _ => break,
493            }
494        }
495    }
496    (map, dw)
497}
498
499fn num(o: &Object) -> Option<f64> {
500    match o {
501        Object::Integer(i) => Some(*i as f64),
502        Object::Real(r) => Some(*r as f64),
503        _ => None,
504    }
505}
506
507/// Parse a ToUnicode CMap's `bfchar` / `bfrange` sections into code→string.
508fn parse_tounicode(data: &[u8]) -> HashMap<u32, String> {
509    let text = String::from_utf8_lossy(data);
510    let mut map = HashMap::new();
511    let hex = |s: &str| -> Option<Vec<u16>> {
512        let s = s.trim();
513        if !s.starts_with('<') || !s.ends_with('>') {
514            return None;
515        }
516        let h = &s[1..s.len() - 1];
517        let bytes: Vec<u8> = (0..h.len())
518            .step_by(2)
519            .filter_map(|i| u8::from_str_radix(h.get(i..i + 2)?, 16).ok())
520            .collect();
521        Some(
522            bytes
523                .chunks(2)
524                .map(|c| {
525                    if c.len() == 2 {
526                        u16::from_be_bytes([c[0], c[1]])
527                    } else {
528                        c[0] as u16
529                    }
530                })
531                .collect(),
532        )
533    };
534    let u16s_to_string = |u: &[u16]| String::from_utf16_lossy(u);
535    let code_of = |u: &[u16]| u.iter().fold(0u32, |acc, &x| (acc << 16) | x as u32);
536
537    // Tokenize by structure, not whitespace: CMap hex groups are often written
538    // back-to-back with no separators (`<21><21><0054>`), so scan for `<…>`
539    // groups, `[`/`]` brackets, and bareword keywords.
540    let tokens: Vec<String> = {
541        let bytes = text.as_bytes();
542        let mut toks = Vec::new();
543        let mut i = 0;
544        while i < bytes.len() {
545            let c = bytes[i];
546            if c.is_ascii_whitespace() {
547                i += 1;
548            } else if c == b'<' {
549                let start = i;
550                while i < bytes.len() && bytes[i] != b'>' {
551                    i += 1;
552                }
553                i += 1; // include '>'
554                toks.push(String::from_utf8_lossy(&bytes[start..i.min(bytes.len())]).into_owned());
555            } else if c == b'[' || c == b']' {
556                toks.push((c as char).to_string());
557                i += 1;
558            } else {
559                let start = i;
560                while i < bytes.len()
561                    && !bytes[i].is_ascii_whitespace()
562                    && bytes[i] != b'<'
563                    && bytes[i] != b'['
564                    && bytes[i] != b']'
565                {
566                    i += 1;
567                }
568                toks.push(String::from_utf8_lossy(&bytes[start..i]).into_owned());
569            }
570        }
571        toks
572    };
573    let tokens: Vec<&str> = tokens.iter().map(|s| s.as_str()).collect();
574    let mut i = 0;
575    while i < tokens.len() {
576        match tokens[i] {
577            "beginbfchar" => {
578                i += 1;
579                while i + 1 < tokens.len() && tokens[i] != "endbfchar" {
580                    if let (Some(src), Some(dst)) = (hex(tokens[i]), hex(tokens[i + 1])) {
581                        map.insert(code_of(&src), u16s_to_string(&dst));
582                    }
583                    i += 2;
584                }
585            }
586            "beginbfrange" => {
587                i += 1;
588                while i + 2 < tokens.len() && tokens[i] != "endbfrange" {
589                    let (Some(lo), Some(hi)) = (hex(tokens[i]), hex(tokens[i + 1])) else {
590                        i += 1;
591                        continue;
592                    };
593                    let lo = code_of(&lo);
594                    let hi = code_of(&hi);
595                    if tokens[i + 2] == "[" {
596                        // `<lo> <hi> [ <d0> <d1> ... ]`: one dst per code in the range.
597                        let mut j = i + 3;
598                        let mut code = lo;
599                        while j < tokens.len() && tokens[j] != "]" {
600                            if let Some(dst) = hex(tokens[j]) {
601                                map.insert(code, u16s_to_string(&dst));
602                            }
603                            code += 1;
604                            j += 1;
605                        }
606                        i = j + 1;
607                    } else if let Some(dst) = hex(tokens[i + 2]) {
608                        // `<lo> <hi> <dst>`: consecutive Unicode from a base.
609                        let base = code_of(&dst);
610                        for (k, code) in (lo..=hi).enumerate() {
611                            if let Some(ch) = char::from_u32(base + k as u32) {
612                                map.insert(code, ch.to_string());
613                            }
614                        }
615                        i += 3;
616                    } else {
617                        i += 1;
618                    }
619                }
620            }
621            _ => i += 1,
622        }
623    }
624    map
625}
626
627/// Decode a PDF string literal in a Tj/TJ operand into raw code units.
628fn codes(font: &Font, bytes: &[u8]) -> Vec<u32> {
629    if font.two_byte {
630        bytes
631            .chunks(2)
632            .map(|c| {
633                if c.len() == 2 {
634                    ((c[0] as u32) << 8) | c[1] as u32
635                } else {
636                    c[0] as u32
637                }
638            })
639            .collect()
640    } else {
641        bytes.iter().map(|&b| b as u32).collect()
642    }
643}
644
645/// Page size (width, height) in PDF points from the MediaBox.
646fn page_size(doc: &Document, page_id: lopdf::ObjectId) -> (f32, f32) {
647    let mb = doc
648        .get_object(page_id)
649        .ok()
650        .and_then(|o| o.as_dict().ok())
651        .and_then(|d| {
652            // MediaBox may be inherited; lopdf resolves via get_page... fall back to a guess.
653            d.get(b"MediaBox").ok().cloned()
654        })
655        .or_else(|| {
656            doc.get_dictionary(page_id)
657                .ok()
658                .and_then(|d| d.get(b"MediaBox").ok().cloned())
659        });
660    if let Some(Object::Array(a)) = mb {
661        let v: Vec<f32> = a.iter().filter_map(|o| num(o).map(|x| x as f32)).collect();
662        if v.len() == 4 {
663            return ((v[2] - v[0]).abs(), (v[3] - v[1]).abs());
664        }
665    }
666    (612.0, 792.0)
667}
668
669/// Localize where a page's text is lost, for the `text_layer` diagnostic.
670/// Extraction can come up empty at three different points — no content stream
671/// reached the parser, the stream did not decode into operators, or it ran but
672/// produced no glyphs (fonts/encodings) — and from the outside all three look
673/// the same. Report them per page.
674pub fn content_diagnosis(bytes: &[u8]) -> String {
675    let Some(doc) = load_document(bytes) else {
676        return "document does not load".into();
677    };
678    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
679    pages.sort_by_key(|(n, _)| *n);
680    let mut out = String::new();
681    let mut caches = DocCaches::default();
682    for (n, pid) in pages.into_iter().take(4) {
683        let content_bytes = doc.get_page_content(pid);
684        let ops = lopdf::content::Content::decode(&content_bytes)
685            .map(|c| c.operations.len())
686            .ok();
687        let res = page_res(&doc, pid);
688        let fonts = res.map(|r| fonts_from_res(&doc, r, &mut caches).len());
689        let glyphs = page_glyphs_cached(&doc, pid, &mut caches).len();
690        out.push_str(&format!(
691            "\n   page {n}: content {} B, ops {}, resources {}, fonts {}, glyphs {}",
692            content_bytes.len(),
693            ops.map_or("UNDECODABLE".to_string(), |n| n.to_string()),
694            if res.is_some() { "ok" } else { "MISSING" },
695            fonts.map_or("-".to_string(), |n| n.to_string()),
696            glyphs,
697        ));
698    }
699    out
700}
701
702/// Is this "text layer" a vestige rather than the document's text?
703///
704/// Scanned forms often carry a handful of typed-in strings — a date filled
705/// into three form fields, say — on top of pages that are otherwise images.
706/// Treating that as a real text layer is the worst of both worlds: the text
707/// path proudly extracts thirteen characters, and no OCR ever runs on the
708/// letter the pages actually show. The reported form did exactly this (3
709/// lines, 13 chars, 3 pages).
710///
711/// The rule is deliberately tight so genuinely sparse *digital* documents are
712/// not misrouted into OCR: only a document averaging at most one line per page
713/// **and** totalling fewer than 32 characters is called vestigial.
714pub fn text_layer_is_vestigial(pages: &[crate::pdfium_backend::PdfPage]) -> bool {
715    let lines: usize = pages.iter().map(|p| p.cells.len()).sum();
716    if lines == 0 {
717        return true;
718    }
719    let chars: usize = pages
720        .iter()
721        .flat_map(|p| &p.cells)
722        .map(|c| c.text.chars().count())
723        .sum();
724    lines <= pages.len() && chars < 32
725}
726
727/// Why the cross-reference repair did or did not fire, for the `text_layer`
728/// diagnostic. A PDF that will not load is indistinguishable from a scan in
729/// production (both convert to nothing), so the reason has to be askable.
730pub fn xref_repair_status(bytes: &[u8]) -> String {
731    if Document::load_mem(bytes).is_ok() {
732        return "loads unaided; no repair needed".into();
733    }
734    match pad_short_xref_entries(bytes) {
735        Ok(fixed) => match Document::load_mem(&fixed) {
736            Ok(_) => "repaired: cross-reference entries padded to 20 bytes".into(),
737            Err(e) => format!("padded the entries, but it still will not load: {e}"),
738        },
739        Err(why) => format!("repair declined — {why}"),
740    }
741}
742
743/// Load a PDF, repairing the one malformation that otherwise costs us the whole
744/// document: **19-byte cross-reference entries**.
745///
746/// The spec fixes an xref entry at 20 bytes — `nnnnnnnnnn ggggg n` plus a
747/// *two*-byte EOL. Some generators (an Austrian telecom's invoices, for one)
748/// emit a bare LF instead, making each entry 19 bytes. lopdf rejects the file
749/// outright (`invalid file trailer`) where pdfium reads it happily, so a
750/// perfectly good text layer looked to the browser exactly like a scan and cost
751/// ten seconds of OCR.
752///
753/// Padding is only attempted when it cannot move anything the xref points at:
754/// a single `xref` section that begins after the last object. The repair then
755/// has to prove itself — the padded bytes are used only if they load — so a
756/// mis-repair degrades to today's behaviour rather than to silent garbage.
757fn load_document(bytes: &[u8]) -> Option<Document> {
758    // Try progressively more repair, and accept a candidate only once the pages
759    // actually carry content — a document whose streams were dropped still
760    // "loads", so loading alone is not evidence the repair helped. A
761    // well-formed file returns on the first attempt and pays for nothing.
762    let mut fallback = None;
763    if let Some(doc) = best_effort_load(bytes, &mut fallback) {
764        return Some(doc);
765    }
766    let xref_fixed = pad_short_xref_entries(bytes).ok();
767    if let Some(fixed) = &xref_fixed {
768        if let Some(doc) = best_effort_load(fixed, &mut fallback) {
769            return Some(doc);
770        }
771    }
772    // Both defects can coexist, and the second only becomes visible once the
773    // first is repaired, so build on whatever the previous step produced.
774    let lengths_fixed = fix_stream_lengths(xref_fixed.as_deref().unwrap_or(bytes));
775    if let Some(doc) = best_effort_load(&lengths_fixed, &mut fallback) {
776        return Some(doc);
777    }
778    fallback
779}
780
781/// Load `data`, returning it only when its pages carry content; a document that
782/// merely parses is remembered as the fallback for when nothing does better.
783fn best_effort_load(data: &[u8], fallback: &mut Option<Document>) -> Option<Document> {
784    match Document::load_mem(data) {
785        Ok(doc) if has_page_content(&doc) => Some(doc),
786        Ok(doc) => {
787            fallback.get_or_insert(doc);
788            None
789        }
790        Err(_) => None,
791    }
792}
793
794/// Does any page actually hand us a content stream? A document whose streams
795/// were dropped still parses — it simply has nothing to read — so this is what
796/// tells a successful repair from a pointless one.
797fn has_page_content(doc: &Document) -> bool {
798    doc.get_pages()
799        .into_values()
800        .take(4)
801        .any(|pid| !doc.get_page_content(pid).is_empty())
802}
803
804/// Correct `/Length` values that disagree with where `endstream` actually is.
805///
806/// The same generator that writes short xref entries also overstates its
807/// content-stream lengths by a byte or two. lopdf trusts `/Length`, reads past
808/// the data, fails to find `endstream` there and drops the stream — the object
809/// comes back as a bare dictionary, so the page has no content at all and the
810/// document looks like a scan. pdfium instead trusts `endstream`, which is what
811/// this does.
812///
813/// The rewrite is length-preserving: the corrected number is written over the
814/// old digits and padded with spaces, so every byte offset in the file — and
815/// therefore the whole cross-reference table — stays valid.
816fn fix_stream_lengths(bytes: &[u8]) -> Vec<u8> {
817    let mut out = bytes.to_vec();
818    let mut i = 0;
819    while let Some(rel) = find(&out[i..], b"stream") {
820        let kw = i + rel;
821        i = kw + 6;
822        // Skip `endstream` (the keyword we are measuring *to*).
823        if kw >= 3 && &out[kw - 3..kw] == b"end" {
824            continue;
825        }
826        // The stream data starts after the EOL that follows the keyword.
827        let mut data = kw + 6;
828        if out.get(data..data + 2) == Some(b"\r\n".as_slice()) {
829            data += 2;
830        } else if matches!(out.get(data), Some(b'\n' | b'\r')) {
831            data += 1;
832        }
833        let Some(end) = find(&out[data..], b"endstream").map(|r| data + r) else {
834            continue;
835        };
836        // `/Length <digits>` in the dictionary just before the keyword.
837        let dict_start = out[..kw].iter().rposition(|&c| c == b'<').unwrap_or(0);
838        let Some(lrel) = find(&out[dict_start..kw], b"/Length") else {
839            continue;
840        };
841        let mut d = dict_start + lrel + 7;
842        while matches!(out.get(d), Some(b' ')) {
843            d += 1;
844        }
845        let digits = out[d..].iter().take_while(|c| c.is_ascii_digit()).count();
846        if digits == 0 {
847            continue;
848        }
849        let declared: usize = match std::str::from_utf8(&out[d..d + digits])
850            .ok()
851            .and_then(|s| s.parse().ok())
852        {
853            Some(v) => v,
854            None => continue,
855        };
856        let actual = end - data;
857        // Only shrink, and only when the new value fits the space the old one
858        // occupied — growing the number would move every following byte.
859        let replacement = actual.to_string();
860        if actual == declared || replacement.len() > digits {
861            continue;
862        }
863        out[d..d + digits].fill(b' ');
864        out[d..d + replacement.len()].copy_from_slice(replacement.as_bytes());
865    }
866    out
867}
868
869fn find(haystack: &[u8], needle: &[u8]) -> Option<usize> {
870    haystack.windows(needle.len()).position(|w| w == needle)
871}
872
873/// Rewrite a classic cross-reference table's entries to the spec's 20 bytes,
874/// or `None` when the file's shape makes that unsafe (see [`load_document`]).
875fn pad_short_xref_entries(bytes: &[u8]) -> Result<Vec<u8>, &'static str> {
876    // Exactly one xref section, and it must start after every object, so that
877    // growing it shifts nothing the table's offsets refer to.
878    let is_boundary = |i: usize| i == 0 || matches!(bytes[i - 1], b'\n' | b'\r');
879    let mut starts = (0..bytes.len().saturating_sub(4))
880        .filter(|&i| &bytes[i..i + 4] == b"xref" && is_boundary(i));
881    let xref_at = starts
882        .next()
883        .ok_or("no classic `xref` section (an xref stream?)")?;
884    if starts.next().is_some() {
885        return Err("more than one xref section (incremental update)");
886    }
887    let last_obj = bytes
888        .windows(3)
889        .rposition(|w| w == b"obj")
890        .ok_or("no objects found")?;
891    if last_obj > xref_at {
892        return Err("an object follows the xref — padding would move it");
893    }
894
895    let mut out = bytes[..xref_at].to_vec();
896    out.extend_from_slice(b"xref\n");
897    let mut i = xref_at + 4;
898    let skip_ws = |i: &mut usize| {
899        while matches!(bytes.get(*i), Some(b'\r' | b'\n' | b' ')) {
900            *i += 1;
901        }
902    };
903    loop {
904        skip_ws(&mut i);
905        // Either the next subsection header ("first count") or the trailer.
906        if bytes[i..].starts_with(b"trailer") {
907            out.extend_from_slice(&bytes[i..]);
908            return Ok(out);
909        }
910        let header_end = i + bytes[i..]
911            .iter()
912            .position(|c| matches!(c, b'\n' | b'\r'))
913            .ok_or("subsection header runs off the end")?;
914        let header = std::str::from_utf8(&bytes[i..header_end])
915            .map_err(|_| "subsection header is not text")?
916            .trim();
917        let mut parts = header.split_whitespace();
918        let count: usize = parts
919            .nth(1)
920            .and_then(|c| c.parse().ok())
921            .ok_or("unparseable subsection header")?;
922        if parts.next().is_some() || count == 0 {
923            return Err("unexpected subsection header shape");
924        }
925        out.extend_from_slice(header.as_bytes());
926        out.push(b'\n');
927        i = header_end;
928        for _ in 0..count {
929            skip_ws(&mut i);
930            // `nnnnnnnnnn ggggg n` — the 18 bytes before whatever EOL follows.
931            let entry = bytes.get(i..i + 18).ok_or("xref entry runs off the end")?;
932            let well_formed = entry[..10].iter().all(u8::is_ascii_digit)
933                && entry[10] == b' '
934                && entry[11..16].iter().all(u8::is_ascii_digit)
935                && entry[16] == b' '
936                && matches!(entry[17], b'n' | b'f');
937            if !well_formed {
938                return Err("xref entry is not `nnnnnnnnnn ggggg n`");
939            }
940            out.extend_from_slice(entry);
941            out.extend_from_slice(b" \n"); // the spec's 2-byte EOL -> 20 bytes
942            i += 18;
943        }
944    }
945}
946
947/// Debug: raw glyph stream `(ch, ll, lr, lb, lt)` (native coords) for page
948/// `index`, before the sanitizer. For comparing char cells to docling-parse.
949pub fn debug_glyphs(bytes: &[u8], index: usize) -> Vec<(char, f32, f32, f32, f32)> {
950    let Some(doc) = load_document(bytes) else {
951        return Vec::new();
952    };
953    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
954    pages.sort_by_key(|(n, _)| *n);
955    let Some((_, pid)) = pages.get(index) else {
956        return Vec::new();
957    };
958    page_glyphs(&doc, *pid)
959        .into_iter()
960        .map(|g| (g.ch, g.ll, g.lr, g.lb, g.lt))
961        .collect()
962}
963
964/// Public entry: per-page (width, height, line cells) for a PDF, via the Rust
965/// text parser + the docling-parse line sanitizer. Used by the pipeline and the
966/// `textparse_dump` example.
967pub fn pdf_textlines(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
968    let Some(doc) = load_document(bytes) else {
969        return Vec::new();
970    };
971    let mut caches = DocCaches::default();
972    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
973    pages.sort_by_key(|(n, _)| *n);
974    pages
975        .into_iter()
976        .map(|(_, pid)| {
977            let (w, h) = page_size(&doc, pid);
978            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
979            let cells = crate::dp_lines::line_cells(&glyphs, h, true);
980            (w, h, cells)
981        })
982        .collect()
983}
984
985/// Debug/diagnostic entry: per-page (width, height, word cells) for a PDF, via
986/// the Rust parser glyphs run through the docling-parse word grouping. Used to
987/// compare parser word cells against docling-parse's `word_cells` oracle (roadmap
988/// item 6).
989pub fn pdf_words(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
990    let Some(doc) = load_document(bytes) else {
991        return Vec::new();
992    };
993    let mut caches = DocCaches::default();
994    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
995    pages.sort_by_key(|(n, _)| *n);
996    pages
997        .into_iter()
998        .map(|(_, pid)| {
999            let (w, h) = page_size(&doc, pid);
1000            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1001            let cells = crate::dp_lines::word_cells(&glyphs, h, true);
1002            (w, h, cells)
1003        })
1004        .collect()
1005}
1006
1007/// One page's text cells from the pure-Rust parser: prose line cells, per-word
1008/// cells, and code line cells — all from a single glyph parse. Replaces the
1009/// pdfium text path (roadmap item 6) when the parser drop is enabled.
1010#[derive(Default)]
1011pub struct PageParserCells {
1012    pub prose: Vec<crate::pdfium_backend::TextCell>,
1013    pub words: Vec<crate::pdfium_backend::TextCell>,
1014    pub code: Vec<crate::pdfium_backend::TextCell>,
1015}
1016
1017/// Full parser text layer: prose + word + code cells per page, glyphs parsed once.
1018/// `prose`/`words` come from the docling-parse contraction ([`crate::dp_lines`]);
1019/// `code` splits only at the parser's own space glyphs (monospace keeps its
1020/// source spacing). Used by the pipeline to retire pdfium's text path.
1021pub fn pdf_all_cells(bytes: &[u8]) -> Vec<PageParserCells> {
1022    let Some(doc) = load_document(bytes) else {
1023        return Vec::new();
1024    };
1025    let mut caches = DocCaches::default();
1026    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1027    pages.sort_by_key(|(n, _)| *n);
1028    pages
1029        .into_iter()
1030        .map(|(_, pid)| {
1031            let (_w, h) = page_size(&doc, pid);
1032            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1033            let (prose, words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1034            PageParserCells {
1035                prose,
1036                words,
1037                code: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1038            }
1039        })
1040        .collect()
1041}
1042
1043/// Whole pages for the text-layer-only conversion ([`crate::convert_text_layer`]):
1044/// the parser's prose/word/code cells plus page geometry, assembled into
1045/// [`PdfPage`]s with no rendered image and no link annotations. Everything here
1046/// is pure Rust (lopdf), so it compiles without the `ml` feature — including on
1047/// wasm32. A page the parser can't read (no text layer) comes back with empty
1048/// cells; there is no pdfium fallback on this path.
1049pub fn pdf_text_pages(bytes: &[u8]) -> Vec<crate::pdfium_backend::PdfPage> {
1050    let Some(doc) = load_document(bytes) else {
1051        return Vec::new();
1052    };
1053    let mut caches = DocCaches::default();
1054    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1055    pages.sort_by_key(|(n, _)| *n);
1056    pages
1057        .into_iter()
1058        .map(|(_, pid)| {
1059            let (w, h) = page_size(&doc, pid);
1060            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1061            let (mut prose, mut words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1062            drop_overpainted_cells(&mut prose);
1063            drop_overpainted_cells(&mut words);
1064            crate::pdfium_backend::PdfPage {
1065                #[cfg(feature = "ocr-prep")]
1066                image_layout: None,
1067                width: w,
1068                height: h,
1069                // Cells are native PDF points; there is no rendered bitmap.
1070                scale: 1.0,
1071                cells: prose,
1072                code_cells: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1073                word_cells: words,
1074                #[cfg(feature = "ocr-prep")]
1075                image: image::RgbImage::new(1, 1),
1076                links: Vec::new(),
1077            }
1078        })
1079        .collect()
1080}
1081
1082/// Drop line cells that are *painted over each other* — glyphs used as artwork.
1083///
1084/// Some generators draw their logo with a symbol font: on the reporting
1085/// invoice, a `TeleLogo` Type1 paints the T-Mobile mark by stacking the glyphs
1086/// encoded as `"` and `==` on top of one another, and the flat text-layer
1087/// output opened with that garbage. Nothing in the font metadata gives it away
1088/// (the *text* fonts in the same file are also flagged symbolic, and the logo
1089/// font names its glyphs `quotedbl` &c.), but the geometry does: two cells with
1090/// different text where one lies inside the other on the same line is
1091/// physically impossible for prose — ink from two words never occupies the
1092/// same box. Both cells of such a pair are paint, not text.
1093///
1094/// Containment (not mere overlap) keeps this narrow: adjacent words touch but
1095/// never contain each other, and a same-text near-duplicate (double-draw faux
1096/// bold) is left alone for the sanitizer's usual handling. Applied on the
1097/// flat/browser path only — the ML pipeline's text layer is byte-pinned by the
1098/// PDF corpus, and there the layout model already sinks logo marks into
1099/// `picture` regions.
1100fn drop_overpainted_cells(cells: &mut Vec<crate::pdfium_backend::TextCell>) {
1101    let mut paint = vec![false; cells.len()];
1102    for i in 0..cells.len() {
1103        for j in 0..cells.len() {
1104            if i == j || cells[i].text == cells[j].text {
1105                continue;
1106            }
1107            let (a, b) = (&cells[i], &cells[j]);
1108            // Same line band: the vertical overlap covers most of the shorter.
1109            let vo = (a.b.min(b.b) - a.t.max(b.t)).max(0.0);
1110            if vo < 0.6 * (a.b - a.t).min(b.b - b.t) {
1111                continue;
1112            }
1113            // `a` horizontally inside `b` (with a small tolerance).
1114            let ho = (a.r.min(b.r) - a.l.max(b.l)).max(0.0);
1115            if ho >= 0.8 * (a.r - a.l) && (a.r - a.l) <= (b.r - b.l) {
1116                paint[i] = true;
1117                paint[j] = true;
1118            }
1119        }
1120    }
1121    let mut keep = paint.iter().map(|p| !p);
1122    cells.retain(|_| keep.next().unwrap());
1123}
1124
1125/// The text-state scalars inherited by a Form XObject when it is invoked via
1126/// `Do` (the PDF graphics state includes the text parameters, but not the text
1127/// matrices, which a form re-establishes inside its own `BT`/`ET`).
1128#[derive(Clone, Copy)]
1129struct TextState {
1130    tc: f64,
1131    tw: f64,
1132    th: f64,
1133    tl: f64,
1134    trise: f64,
1135    fsize: f64,
1136}
1137
1138impl TextState {
1139    const INIT: TextState = TextState {
1140        tc: 0.0,
1141        tw: 0.0,
1142        th: 1.0,
1143        tl: 0.0,
1144        trise: 0.0,
1145        fsize: 0.0,
1146    };
1147}
1148
1149/// The effective `/Resources` dictionary for a page (inline or via reference,
1150/// falling back to an inherited one from a `/Parent`).
1151fn page_res(doc: &Document, page_id: lopdf::ObjectId) -> Option<&Dictionary> {
1152    let (inline, ids) = doc.get_page_resources(page_id).ok()?;
1153    if let Some(d) = inline {
1154        return Some(d);
1155    }
1156    ids.into_iter().find_map(|id| doc.get_dictionary(id).ok())
1157}
1158
1159/// Build the code→[`Font`] map for a resources dictionary's `/Font` sub-dict,
1160/// reusing the per-document cache for fonts referenced indirectly (the common
1161/// case — the same font objects recur on every page).
1162fn fonts_from_res(
1163    doc: &Document,
1164    res: &Dictionary,
1165    caches: &mut DocCaches,
1166) -> HashMap<Vec<u8>, Rc<Font>> {
1167    let mut map = HashMap::new();
1168    let font_dict = res
1169        .get(b"Font")
1170        .ok()
1171        .and_then(|o| deref(doc, o))
1172        .and_then(|o| o.as_dict().ok());
1173    if let Some(fd) = font_dict {
1174        for (name, value) in fd.iter() {
1175            let font = match value {
1176                Object::Reference(id) => {
1177                    let key = (*id, name.clone());
1178                    if let Some(f) = caches.fonts.get(&key) {
1179                        Rc::clone(f)
1180                    } else if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1181                        let f = Rc::new(parse_font(doc, name, fdict));
1182                        caches.fonts.insert(key, Rc::clone(&f));
1183                        f
1184                    } else {
1185                        continue;
1186                    }
1187                }
1188                _ => {
1189                    if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1190                        Rc::new(parse_font(doc, name, fdict))
1191                    } else {
1192                        continue;
1193                    }
1194                }
1195            };
1196            map.insert(name.clone(), font);
1197        }
1198    }
1199    map
1200}
1201
1202/// Extract every glyph on a page as a native-coordinate [`Glyph`].
1203pub(crate) fn page_glyphs(doc: &Document, page_id: lopdf::ObjectId) -> Vec<Glyph> {
1204    page_glyphs_cached(doc, page_id, &mut DocCaches::default())
1205}
1206
1207/// [`page_glyphs`] with an explicit per-document cache, so a multi-page walk
1208/// parses each font / decodes each form once instead of once per page.
1209fn page_glyphs_cached(
1210    doc: &Document,
1211    page_id: lopdf::ObjectId,
1212    caches: &mut DocCaches,
1213) -> Vec<Glyph> {
1214    let mut out = Vec::new();
1215    // lopdf 0.44: get_page_content returns the assembled content-stream bytes
1216    // directly (an empty Vec when the page has none).
1217    let content_bytes = doc.get_page_content(page_id);
1218    let Ok(content) = lopdf::content::Content::decode(&content_bytes) else {
1219        return out;
1220    };
1221    if let Some(res) = page_res(doc, page_id) {
1222        run_content(
1223            doc,
1224            res,
1225            &content,
1226            Mat::ID,
1227            TextState::INIT,
1228            0,
1229            caches,
1230            &mut out,
1231        );
1232    }
1233    out
1234}
1235
1236/// Run a content stream's operators, emitting glyphs into `out`. Recurses into
1237/// Form XObjects on `Do` (bulk body text in heavy PDFs lives inside a form, not
1238/// the page content stream). `res` is the resources dict in scope (the page's,
1239/// or the form's own); `base_ctm` is the CTM at the point of invocation.
1240#[allow(clippy::too_many_arguments)]
1241fn run_content(
1242    doc: &Document,
1243    res: &Dictionary,
1244    content: &lopdf::content::Content,
1245    base_ctm: Mat,
1246    init: TextState,
1247    depth: u32,
1248    caches: &mut DocCaches,
1249    out: &mut Vec<Glyph>,
1250) {
1251    let fonts = fonts_from_res(doc, res, caches);
1252    let xobjects = res
1253        .get(b"XObject")
1254        .ok()
1255        .and_then(|o| deref(doc, o))
1256        .and_then(|o| o.as_dict().ok());
1257
1258    // Graphics + text state. `q`/`Q` save and restore the whole graphics state,
1259    // which includes the text parameters (Tc, Tw, Tz, TL, Tfs, Trise, font) —
1260    // *not* the text matrix (that is reset by BT). Saving only the CTM let a Tc
1261    // set inside a `q…Q` block leak out and drift every later glyph.
1262    #[allow(clippy::type_complexity)]
1263    let mut gstate_stack: Vec<(Mat, f64, f64, f64, f64, f64, f64, Option<&Rc<Font>>)> = Vec::new();
1264    let mut ctm = base_ctm;
1265    let mut tm = Mat::ID;
1266    let mut tlm = Mat::ID;
1267    let mut font: Option<&Rc<Font>> = None;
1268    let mut fsize = init.fsize;
1269    let mut tc = init.tc; // char spacing
1270    let mut tw = init.tw; // word spacing
1271    let mut th = init.th; // horizontal scale (Tz/100)
1272    let mut tl = init.tl; // leading
1273    let mut trise = init.trise;
1274
1275    let op_f = |operands: &[Object], i: usize| operands.get(i).and_then(num).unwrap_or(0.0);
1276
1277    for op in &content.operations {
1278        let operands = &op.operands;
1279        match op.operator.as_str() {
1280            "q" => gstate_stack.push((ctm, tc, tw, th, tl, trise, fsize, font)),
1281            "Q" => {
1282                if let Some((c, a, b, h, l, r, fs, f)) = gstate_stack.pop() {
1283                    ctm = c;
1284                    tc = a;
1285                    tw = b;
1286                    th = h;
1287                    tl = l;
1288                    trise = r;
1289                    fsize = fs;
1290                    font = f;
1291                }
1292            }
1293            "cm" => {
1294                let m = Mat {
1295                    a: op_f(operands, 0),
1296                    b: op_f(operands, 1),
1297                    c: op_f(operands, 2),
1298                    d: op_f(operands, 3),
1299                    e: op_f(operands, 4),
1300                    f: op_f(operands, 5),
1301                };
1302                ctm = m.then(ctm);
1303            }
1304            "BT" => {
1305                tm = Mat::ID;
1306                tlm = Mat::ID;
1307            }
1308            "ET" => {}
1309            "Tf" => {
1310                if let Some(Object::Name(n)) = operands.first() {
1311                    font = fonts.get(n.as_slice());
1312                }
1313                fsize = op_f(operands, 1);
1314            }
1315            "Td" => {
1316                tlm = Mat {
1317                    a: 1.0,
1318                    b: 0.0,
1319                    c: 0.0,
1320                    d: 1.0,
1321                    e: op_f(operands, 0),
1322                    f: op_f(operands, 1),
1323                }
1324                .then(tlm);
1325                tm = tlm;
1326            }
1327            "TD" => {
1328                tl = -op_f(operands, 1);
1329                tlm = Mat {
1330                    a: 1.0,
1331                    b: 0.0,
1332                    c: 0.0,
1333                    d: 1.0,
1334                    e: op_f(operands, 0),
1335                    f: op_f(operands, 1),
1336                }
1337                .then(tlm);
1338                tm = tlm;
1339            }
1340            "Tm" => {
1341                tlm = Mat {
1342                    a: op_f(operands, 0),
1343                    b: op_f(operands, 1),
1344                    c: op_f(operands, 2),
1345                    d: op_f(operands, 3),
1346                    e: op_f(operands, 4),
1347                    f: op_f(operands, 5),
1348                };
1349                tm = tlm;
1350            }
1351            "T*" => {
1352                tlm = Mat {
1353                    a: 1.0,
1354                    b: 0.0,
1355                    c: 0.0,
1356                    d: 1.0,
1357                    e: 0.0,
1358                    f: -tl,
1359                }
1360                .then(tlm);
1361                tm = tlm;
1362            }
1363            "Tc" => tc = op_f(operands, 0),
1364            "Tw" => tw = op_f(operands, 0),
1365            "Tz" => th = op_f(operands, 0) / 100.0,
1366            "TL" => tl = op_f(operands, 0),
1367            "Ts" => trise = op_f(operands, 0),
1368            "Tj" | "'" | "\"" => {
1369                if op.operator == "'" || op.operator == "\"" {
1370                    // move to next line first
1371                    tlm = Mat {
1372                        a: 1.0,
1373                        b: 0.0,
1374                        c: 0.0,
1375                        d: 1.0,
1376                        e: 0.0,
1377                        f: -tl,
1378                    }
1379                    .then(tlm);
1380                    tm = tlm;
1381                }
1382                if op.operator == "\"" {
1383                    // `aw ac string "` sets word- and char-spacing before
1384                    // showing the string (PDF 32000-1 §9.4.3), persisting after.
1385                    tw = op_f(operands, 0);
1386                    tc = op_f(operands, 1);
1387                }
1388                if let (Some(f), Some(Object::String(s, _))) = (font, operands.last()) {
1389                    show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out);
1390                }
1391            }
1392            "TJ" => {
1393                if let (Some(f), Some(Object::Array(arr))) = (font, operands.first()) {
1394                    for el in arr {
1395                        match el {
1396                            Object::String(s, _) => {
1397                                show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out)
1398                            }
1399                            other => {
1400                                if let Some(adj) = num(other) {
1401                                    // negative number moves text right (PDF: subtract)
1402                                    let tx = -adj / 1000.0 * fsize * th;
1403                                    tm = Mat {
1404                                        a: 1.0,
1405                                        b: 0.0,
1406                                        c: 0.0,
1407                                        d: 1.0,
1408                                        e: tx,
1409                                        f: 0.0,
1410                                    }
1411                                    .then(tm);
1412                                }
1413                            }
1414                        }
1415                    }
1416                }
1417            }
1418            "Do" => {
1419                // Invoke a Form XObject: bulk body text in many PDFs lives inside
1420                // a form, reached only here. Image XObjects are skipped (no text).
1421                if depth >= 8 {
1422                    continue;
1423                }
1424                let Some(Object::Name(n)) = operands.first() else {
1425                    continue;
1426                };
1427                let obj = xobjects.and_then(|d| d.get(n.as_slice()).ok());
1428                let form_id = match obj {
1429                    Some(Object::Reference(id)) => Some(*id),
1430                    _ => None,
1431                };
1432                let stream = obj
1433                    .and_then(|o| deref(doc, o))
1434                    .and_then(|o| o.as_stream().ok());
1435                let Some(stream) = stream else { continue };
1436                let is_form = stream
1437                    .dict
1438                    .get(b"Subtype")
1439                    .ok()
1440                    .and_then(|o| o.as_name().ok())
1441                    == Some(b"Form".as_slice());
1442                if !is_form {
1443                    continue;
1444                }
1445                // Decode the form's content once per document (headers/footers
1446                // and bulk body text invoke the same form on every page).
1447                let cached = form_id.and_then(|id| caches.forms.get(&id).cloned());
1448                let form_content = match cached {
1449                    Some(c) => c,
1450                    None => {
1451                        let Ok(data) = stream.decompressed_content() else {
1452                            continue;
1453                        };
1454                        let Ok(c) = lopdf::content::Content::decode(&data) else {
1455                            continue;
1456                        };
1457                        let c = Rc::new(c);
1458                        if let Some(id) = form_id {
1459                            caches.forms.insert(id, Rc::clone(&c));
1460                        }
1461                        c
1462                    }
1463                };
1464                // The form's /Matrix maps form space into the CTM at invocation.
1465                let form_mat = match stream.dict.get(b"Matrix").ok() {
1466                    Some(Object::Array(a)) if a.len() == 6 => {
1467                        let v: Vec<f64> = a.iter().filter_map(num).collect();
1468                        if v.len() == 6 {
1469                            Mat {
1470                                a: v[0],
1471                                b: v[1],
1472                                c: v[2],
1473                                d: v[3],
1474                                e: v[4],
1475                                f: v[5],
1476                            }
1477                        } else {
1478                            Mat::ID
1479                        }
1480                    }
1481                    _ => Mat::ID,
1482                };
1483                // The form's own /Resources, falling back to the inherited ones.
1484                let form_res = stream
1485                    .dict
1486                    .get(b"Resources")
1487                    .ok()
1488                    .and_then(|o| deref(doc, o))
1489                    .and_then(|o| o.as_dict().ok())
1490                    .unwrap_or(res);
1491                let state = TextState {
1492                    tc,
1493                    tw,
1494                    th,
1495                    tl,
1496                    trise,
1497                    fsize,
1498                };
1499                run_content(
1500                    doc,
1501                    form_res,
1502                    &form_content,
1503                    form_mat.then(ctm),
1504                    state,
1505                    depth + 1,
1506                    caches,
1507                    out,
1508                );
1509            }
1510            _ => {}
1511        }
1512    }
1513}
1514
1515#[allow(clippy::too_many_arguments)]
1516fn show_text(
1517    font: &Font,
1518    bytes: &[u8],
1519    fsize: f64,
1520    tc: f64,
1521    tw: f64,
1522    th: f64,
1523    trise: f64,
1524    tm: &mut Mat,
1525    ctm: Mat,
1526    out: &mut Vec<Glyph>,
1527) {
1528    for code in codes(font, bytes) {
1529        let (text, w) = font.decode_code(code);
1530        let w0 = w / 1000.0; // advance in text-space (em) units
1531                             // The glyph→user transform: scale glyph space by font size, then Tm, CTM.
1532        let scale = Mat {
1533            a: fsize * th,
1534            b: 0.0,
1535            c: 0.0,
1536            d: fsize,
1537            e: 0.0,
1538            f: trise,
1539        };
1540        let trm = scale.then(*tm).then(ctm);
1541        // Box in glyph space (1000-unit em): x 0..w, y descent..ascent.
1542        let (x0, y0) = trm.apply(0.0, font.descent / 1000.0);
1543        let (x1, _y1) = trm.apply(w0, font.descent / 1000.0);
1544        let (_x2, y2) = trm.apply(0.0, font.ascent / 1000.0);
1545        let (left, right) = (x0.min(x1), x0.max(x1));
1546        let (bot, top) = (y0.min(y2), y0.max(y2));
1547        if let Some(s) = text {
1548            // A run may map one code to multiple chars (ligature/fraction); share box.
1549            for ch in s.chars() {
1550                if ch != '\u{0}' {
1551                    out.push(Glyph {
1552                        ch,
1553                        l: left as f32,
1554                        b: bot as f32,
1555                        r: right as f32,
1556                        t: top as f32,
1557                        ll: left as f32,
1558                        lb: bot as f32,
1559                        lr: right as f32,
1560                        lt: top as f32,
1561                        font: font.hash,
1562                    });
1563                }
1564            }
1565        }
1566        // Advance the text matrix. Word spacing applies to single-byte code 32.
1567        let is_space = !font.two_byte && code == 32;
1568        let tx = (w0 * fsize + tc + if is_space { tw } else { 0.0 }) * th;
1569        *tm = Mat {
1570            a: 1.0,
1571            b: 0.0,
1572            c: 0.0,
1573            d: 1.0,
1574            e: tx,
1575            f: 0.0,
1576        }
1577        .then(*tm);
1578    }
1579}
1580
1581/// Build a simple font's code→char table from its `/Encoding`: the base
1582/// encoding (WinAnsi / MacRoman) plus any `/Differences` overrides (glyph names
1583/// resolved through a small Adobe-glyph-name subset).
1584fn simple_encoding_table(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
1585    let enc = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o));
1586    let base_name = match enc {
1587        Some(Object::Name(n)) => n.clone(),
1588        Some(Object::Dictionary(d)) => d
1589            .get(b"BaseEncoding")
1590            .ok()
1591            .and_then(|o| o.as_name().ok())
1592            .map(|n| n.to_vec())
1593            .unwrap_or_default(),
1594        _ => Vec::new(),
1595    };
1596    let mut m = if base_name == b"MacRomanEncoding" {
1597        macroman_table()
1598    } else if base_name.is_empty() {
1599        // No PDF /Encoding at all: the font's *built-in* encoding applies. For
1600        // the standard TeX math fonts that is their fixed TeX layout — falling
1601        // back to StandardEncoding read CMSY's braces as `f`/`g`, `→` as `!`,
1602        // `∈` as `2` (2203's `{ahn,…}` author line). The font program (often
1603        // CFF, which this parser does not read) carries the same mapping;
1604        // docling-parse decodes it from there.
1605        tex_math_builtin(fdict).unwrap_or_else(winansi_table)
1606    } else {
1607        winansi_table()
1608    };
1609    // Apply /Differences: `code /glyphname /glyphname ... code ...`.
1610    if let Some(Object::Dictionary(d)) = enc {
1611        if let Some(Object::Array(diffs)) = d.get(b"Differences").ok().and_then(|o| deref(doc, o)) {
1612            let mut code = 0u8;
1613            for el in diffs {
1614                match el {
1615                    Object::Integer(i) => code = *i as u8,
1616                    Object::Name(name) => {
1617                        if let Some(ch) = glyph_name_to_char(name) {
1618                            m.insert(code, ch);
1619                        }
1620                        code = code.wrapping_add(1);
1621                    }
1622                    _ => {}
1623                }
1624            }
1625        }
1626    }
1627    m
1628}
1629
1630/// The fixed built-in encodings of the standard TeX math fonts (TeXbook
1631/// Appendix F), keyed off the base font name: `CMSY*` (symbols; `CMBSY` is its
1632/// bold) and `CMMI*` (math italic). These fonts ship no PDF `/Encoding` and no
1633/// ToUnicode, and their program is usually CFF — without this table the codes
1634/// fell through to StandardEncoding and rendered as the wrong ASCII.
1635fn tex_math_builtin(fdict: &Dictionary) -> Option<HashMap<u8, char>> {
1636    const CMSY: [char; 128] = [
1637        '−', '·', '×', '∗', '÷', '⋄', '±', '∓', '⊕', '⊖', '⊗', '⊘', '⊙', '◯', '∘', '•', '≍', '≡',
1638        '⊆', '⊇', '≤', '≥', '≼', '≽', '∼', '≈', '⊂', '⊃', '≪', '≫', '≺', '≻', '←', '→', '↑', '↓',
1639        '↔', '↗', '↘', '≃', '⇐', '⇒', '⇑', '⇓', '⇔', '↖', '↙', '∝', '′', '∞', '∈', '∋', '△', '▽',
1640        '\u{338}', '↦', '∀', '∃', '¬', '∅', 'ℜ', 'ℑ', '⊤', '⊥', 'ℵ', 'A', 'B', 'C', 'D', 'E', 'F',
1641        'G', 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X',
1642        'Y', 'Z', '∪', '∩', '⊎', '∧', '∨', '⊢', '⊣', '⌊', '⌋', '⌈', '⌉', '{', '}', '⟨', '⟩', '|',
1643        '∥', '↕', '⇕', '\\', '≀', '√', '∐', '∇', '∫', '⊔', '⊓', '⊑', '⊒', '§', '†', '‡', '¶', '♣',
1644        '♢', '♡', '♠',
1645    ];
1646    const CMMI: [char; 128] = [
1647        'Γ', 'Δ', 'Θ', 'Λ', 'Ξ', 'Π', 'Σ', 'Υ', 'Φ', 'Ψ', 'Ω', 'α', 'β', 'γ', 'δ', 'ε', 'ζ', 'η',
1648        'θ', 'ι', 'κ', 'λ', 'μ', 'ν', 'ξ', 'π', 'ρ', 'σ', 'τ', 'υ', 'φ', 'χ', 'ψ', 'ω', 'ϵ', 'ϑ',
1649        'ϖ', 'ϱ', 'ς', 'ϕ', '↼', '↽', '⇀', '⇁', '↩', '↪', '▷', '◁', '0', '1', '2', '3', '4', '5',
1650        '6', '7', '8', '9', '.', ',', '<', '/', '>', '⋆', '∂', 'A', 'B', 'C', 'D', 'E', 'F', 'G',
1651        'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y',
1652        'Z', '♭', '♮', '♯', '⌣', '⌢', 'ℓ', 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k',
1653        'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', 'ı', 'ȷ', '℘',
1654        '\u{20d7}', '⁀',
1655    ];
1656    let name = base_font_name(fdict)?;
1657    let up = name.to_ascii_uppercase();
1658    let table: &[char; 128] = if up.starts_with(b"CMSY") || up.starts_with(b"CMBSY") {
1659        &CMSY
1660    } else if up.starts_with(b"CMMI") {
1661        &CMMI
1662    } else {
1663        return None;
1664    };
1665    Some(
1666        table
1667            .iter()
1668            .enumerate()
1669            .map(|(i, &c)| (i as u8, c))
1670            .collect(),
1671    )
1672}
1673
1674/// Resolve an Adobe glyph name to Unicode: `uniXXXX`, single ASCII letters, the
1675/// digit/punctuation names from the Adobe Glyph List, and common typographic
1676/// names. A `.suffix` (`one.taboldstyle`, `a.sc`) is stripped and the base name
1677/// retried — docling renders these as the base character.
1678fn glyph_name_to_char(name: &[u8]) -> Option<char> {
1679    let s = std::str::from_utf8(name).ok()?;
1680    if let Some(hex) = s.strip_prefix("uni") {
1681        if let Ok(cp) = u32::from_str_radix(hex.get(0..4)?, 16) {
1682            return char::from_u32(cp);
1683        }
1684    }
1685    // Single ASCII letter names (`A`, `m`) map to themselves.
1686    if s.len() == 1 {
1687        let b = s.as_bytes()[0];
1688        if b.is_ascii_alphabetic() {
1689            return Some(b as char);
1690        }
1691    }
1692    let resolved = match s {
1693        "space" => ' ',
1694        "exclam" => '!',
1695        "quotedbl" => '"',
1696        "numbersign" => '#',
1697        "dollar" => '$',
1698        "percent" => '%',
1699        "ampersand" => '&',
1700        "quotesingle" => '\'',
1701        "parenleft" => '(',
1702        "parenright" => ')',
1703        "asterisk" => '*',
1704        "plus" => '+',
1705        "comma" => ',',
1706        "hyphen" => '-',
1707        "period" => '.',
1708        "slash" => '/',
1709        "zero" => '0',
1710        "one" => '1',
1711        "two" => '2',
1712        "three" => '3',
1713        "four" => '4',
1714        "five" => '5',
1715        "six" => '6',
1716        "seven" => '7',
1717        "eight" => '8',
1718        "nine" => '9',
1719        "colon" => ':',
1720        "semicolon" => ';',
1721        "less" => '<',
1722        "equal" => '=',
1723        "greater" => '>',
1724        "question" => '?',
1725        "at" => '@',
1726        "bracketleft" => '[',
1727        "backslash" => '\\',
1728        "bracketright" => ']',
1729        "asciicircum" => '^',
1730        "underscore" => '_',
1731        "grave" => '`',
1732        "braceleft" => '{',
1733        "bar" => '|',
1734        "braceright" => '}',
1735        "asciitilde" => '~',
1736        "bullet" => '\u{2022}',
1737        "periodcentered" => '\u{00B7}',
1738        "endash" => '\u{2013}',
1739        "emdash" => '\u{2014}',
1740        "quoteright" => '\u{2019}',
1741        "quoteleft" => '\u{2018}',
1742        "quotedblleft" => '\u{201C}',
1743        "quotedblright" => '\u{201D}',
1744        "quotedblbase" => '\u{201E}',
1745        "quotesinglbase" => '\u{201A}',
1746        // Latin f-ligatures named in `/Differences` (e.g. 2305's body font). These
1747        // map to the presentation-form code points, which `decompose_ligatures`
1748        // then spells back out (`ff`→"ff") — without them the glyph decodes to
1749        // nothing and the sanitizer fills the gap with a space (`di erences`).
1750        "ff" => '\u{FB00}',
1751        "fi" => '\u{FB01}',
1752        "fl" => '\u{FB02}',
1753        "ffi" => '\u{FB03}',
1754        "ffl" => '\u{FB04}',
1755        "ft" => '\u{FB05}',
1756        "st" => '\u{FB06}',
1757        "degree" => '\u{00B0}',
1758        "trademark" => '\u{2122}',
1759        "registered" => '\u{00AE}',
1760        "copyright" => '\u{00A9}',
1761        "ellipsis" => '\u{2026}',
1762        "minus" => '\u{2212}',
1763        "fraction" => '\u{2044}',
1764        "nbspace" => '\u{00A0}',
1765        // Greek + math glyph names (standard Adobe Glyph List). Standard TeX math
1766        // fonts (CMMI/CMSY/…) name their glyphs this way in the embedded font
1767        // program's `/Encoding`; without these a `λ`/`≤` decodes to nothing and is
1768        // dropped from body text (`and λ set to 0.5` → `and set to 0.5`).
1769        "alpha" => '\u{03B1}',
1770        "beta" => '\u{03B2}',
1771        "gamma" => '\u{03B3}',
1772        "delta" => '\u{03B4}',
1773        "epsilon" | "epsilon1" => '\u{03B5}',
1774        "zeta" => '\u{03B6}',
1775        "eta" => '\u{03B7}',
1776        "theta" | "theta1" => '\u{03B8}',
1777        "iota" => '\u{03B9}',
1778        "kappa" => '\u{03BA}',
1779        "lambda" => '\u{03BB}',
1780        "mu" => '\u{03BC}',
1781        "nu" => '\u{03BD}',
1782        "xi" => '\u{03BE}',
1783        "omicron" => '\u{03BF}',
1784        "pi" | "pi1" => '\u{03C0}',
1785        "rho" | "rho1" => '\u{03C1}',
1786        "sigma" => '\u{03C3}',
1787        "sigma1" => '\u{03C2}',
1788        "tau" => '\u{03C4}',
1789        "upsilon" => '\u{03C5}',
1790        "phi" | "phi1" => '\u{03C6}',
1791        "chi" => '\u{03C7}',
1792        "psi" => '\u{03C8}',
1793        "omega" | "omega1" => '\u{03C9}',
1794        "Gamma" => '\u{0393}',
1795        "Delta" => '\u{0394}',
1796        "Theta" => '\u{0398}',
1797        "Lambda" => '\u{039B}',
1798        "Xi" => '\u{039E}',
1799        "Pi" => '\u{03A0}',
1800        "Sigma" => '\u{03A3}',
1801        "Upsilon" => '\u{03A5}',
1802        "Phi" => '\u{03A6}',
1803        "Psi" => '\u{03A8}',
1804        "Omega" => '\u{03A9}',
1805        "lessequal" => '\u{2264}',
1806        "greaterequal" => '\u{2265}',
1807        "notequal" => '\u{2260}',
1808        "approxequal" => '\u{2248}',
1809        "equivalence" => '\u{2261}',
1810        "element" => '\u{2208}',
1811        "plusminus" => '\u{00B1}',
1812        "multiply" => '\u{00D7}',
1813        "divide" => '\u{00F7}',
1814        "infinity" => '\u{221E}',
1815        "partialdiff" => '\u{2202}',
1816        "gradient" => '\u{2207}',
1817        "summation" => '\u{2211}',
1818        "product" => '\u{220F}',
1819        "integral" => '\u{222B}',
1820        "radical" => '\u{221A}',
1821        "proportional" => '\u{221D}',
1822        "arrowright" => '\u{2192}',
1823        "arrowleft" => '\u{2190}',
1824        "arrowup" => '\u{2191}',
1825        "arrowdown" => '\u{2193}',
1826        "arrowboth" => '\u{2194}',
1827        "arrowdblright" => '\u{21D2}',
1828        "logicaland" => '\u{2227}',
1829        "logicalor" => '\u{2228}',
1830        "intersection" => '\u{2229}',
1831        "union" => '\u{222A}',
1832        "similar" => '\u{223C}',
1833        "congruent" => '\u{2245}',
1834        "dotmath" => '\u{22C5}',
1835        "asteriskmath" => '\u{2217}',
1836        _ => {
1837            // Strip an AGL `.suffix` (oldstyle/small-cap variant) and retry.
1838            if let Some((base, _)) = s.split_once('.') {
1839                if !base.is_empty() {
1840                    return glyph_name_to_char(base.as_bytes());
1841                }
1842            }
1843            return None;
1844        }
1845    };
1846    Some(resolved)
1847}
1848
1849/// Minimal WinAnsiEncoding (Latin-1-ish) for simple fonts lacking ToUnicode.
1850fn winansi_table() -> HashMap<u8, char> {
1851    let mut m = HashMap::new();
1852    for b in 0x20u8..=0x7e {
1853        m.insert(b, b as char);
1854    }
1855    // High range: Windows-1252 printable points that differ from Latin-1.
1856    let extra: &[(u8, char)] = &[
1857        (0x91, '\u{2018}'),
1858        (0x92, '\u{2019}'),
1859        (0x93, '\u{201C}'),
1860        (0x94, '\u{201D}'),
1861        (0x95, '\u{2022}'),
1862        (0x96, '\u{2013}'),
1863        (0x97, '\u{2014}'),
1864        (0x85, '\u{2026}'),
1865        (0xA0, '\u{00A0}'),
1866    ];
1867    for &(b, c) in extra {
1868        m.insert(b, c);
1869    }
1870    for b in 0xA1u8..=0xFF {
1871        m.entry(b).or_insert(b as char);
1872    }
1873    m
1874}
1875
1876/// Minimal MacRomanEncoding: ASCII plus the high-range points our corpus hits
1877/// (notably 0xA5 = bullet, used as a list marker).
1878fn macroman_table() -> HashMap<u8, char> {
1879    let mut m = HashMap::new();
1880    for b in 0x20u8..=0x7e {
1881        m.insert(b, b as char);
1882    }
1883    let high: &[(u8, char)] = &[
1884        (0xA5, '\u{2022}'), // bullet
1885        (0xD0, '\u{2013}'), // endash
1886        (0xD1, '\u{2014}'), // emdash
1887        (0xD2, '\u{201C}'),
1888        (0xD3, '\u{201D}'),
1889        (0xD4, '\u{2018}'),
1890        (0xD5, '\u{2019}'),
1891        (0xCA, '\u{00A0}'),
1892        (0xC9, '\u{2026}'),
1893        (0xDE, '\u{FB01}'),
1894        (0xDF, '\u{FB02}'),
1895    ];
1896    for &(b, c) in high {
1897        m.insert(b, c);
1898    }
1899    m
1900}
1901
1902#[cfg(test)]
1903mod xref_repair {
1904    /// Build a tiny one-page PDF whose cross-reference entries are either the
1905    /// spec's 20 bytes (`two_byte_eol`) or the 19-byte form some generators
1906    /// emit — everything else about the two files is identical.
1907    fn pdf_with_xref(two_byte_eol: bool) -> Vec<u8> {
1908        let content = b"BT /F1 12 Tf 72 700 Td (Invoice 922769430725) Tj ET\n";
1909        let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
1910        let objs: Vec<Vec<u8>> = vec![
1911            b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
1912            b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
1913            b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
1914               /Resources<</Font<</F1 5 0 R>>>>>>"
1915                .to_vec(),
1916            [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
1917            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".to_vec(),
1918        ];
1919
1920        let mut out = b"%PDF-1.4\n".to_vec();
1921        let mut offsets = Vec::new();
1922        for (i, body) in objs.iter().enumerate() {
1923            offsets.push(out.len());
1924            out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
1925            out.extend_from_slice(body);
1926            out.extend_from_slice(b"endobj\n");
1927        }
1928        let xref_at = out.len();
1929        let eol: &[u8] = if two_byte_eol { b" \n" } else { b"\n" };
1930        out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
1931        out.extend_from_slice(b"0000000000 65535 f");
1932        out.extend_from_slice(eol);
1933        for off in &offsets {
1934            out.extend_from_slice(format!("{off:010} 00000 n").as_bytes());
1935            out.extend_from_slice(eol);
1936        }
1937        out.extend_from_slice(
1938            format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
1939        );
1940        out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
1941        out
1942    }
1943
1944    /// A 19-byte cross-reference entry (a bare LF where the spec wants a
1945    /// two-byte EOL) makes lopdf reject the whole file, so a readable text
1946    /// layer used to look exactly like a scan — in the browser that meant ten
1947    /// seconds of OCR for nothing. The repair must recover *the same* parse the
1948    /// well-formed file gives.
1949    #[test]
1950    fn short_xref_entries_still_parse() {
1951        let good = pdf_with_xref(true);
1952        let broken = pdf_with_xref(false);
1953        assert!(
1954            broken.len() < good.len(),
1955            "the broken file is the shorter one"
1956        );
1957        assert!(
1958            lopdf::Document::load_mem(&good).is_ok(),
1959            "the control file must load unaided"
1960        );
1961        assert!(
1962            lopdf::Document::load_mem(&broken).is_err(),
1963            "lopdf rejects 19-byte entries — if this ever passes, drop the repair"
1964        );
1965
1966        let cells = |b: &[u8]| -> Vec<String> {
1967            super::pdf_textlines(b)
1968                .into_iter()
1969                .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
1970                .collect()
1971        };
1972        let from_good = cells(&good);
1973        assert!(
1974            from_good.iter().any(|t| t.contains("922769430725")),
1975            "control text: {from_good:?}"
1976        );
1977        assert_eq!(
1978            cells(&broken),
1979            from_good,
1980            "repair must match the good parse"
1981        );
1982    }
1983
1984    /// The same generator overstates `/Length`, so lopdf reads past the data,
1985    /// misses `endstream` and drops the stream — the object comes back as a
1986    /// bare dictionary and the page has no content at all. Trust `endstream`
1987    /// instead, and do it without moving a single byte.
1988    #[test]
1989    fn overstated_stream_length_still_yields_content() {
1990        let good = pdf_with_xref(true);
1991        // Inflate the content stream's /Length by one, exactly as the invoice
1992        // that prompted this does.
1993        let broken = {
1994            let at = good
1995                .windows(8)
1996                .position(|w| w == b"/Length ")
1997                .expect("a /Length")
1998                + 8;
1999            let digits = good[at..].iter().take_while(|c| c.is_ascii_digit()).count();
2000            let n: usize = std::str::from_utf8(&good[at..at + digits])
2001                .unwrap()
2002                .parse()
2003                .unwrap();
2004            let inflated = (n + 1).to_string();
2005            assert_eq!(inflated.len(), digits, "keep the digit count");
2006            let mut b = good.clone();
2007            b[at..at + digits].copy_from_slice(inflated.as_bytes());
2008            b
2009        };
2010        assert_eq!(broken.len(), good.len(), "the defect must not move bytes");
2011        // lopdf alone loses the stream: the page parses but carries no content.
2012        let raw = lopdf::Document::load_mem(&broken).expect("still loads");
2013        assert!(
2014            raw.get_pages()
2015                .into_values()
2016                .all(|p| raw.get_page_content(p).is_empty()),
2017            "lopdf should drop the stream — if it stops, drop this repair"
2018        );
2019        // Ours recovers the same text the well-formed file gives.
2020        let text = |b: &[u8]| -> Vec<String> {
2021            super::pdf_textlines(b)
2022                .into_iter()
2023                .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2024                .collect()
2025        };
2026        let expected = text(&good);
2027        assert!(!expected.is_empty(), "control must produce text");
2028        assert_eq!(text(&broken), expected);
2029    }
2030
2031    /// The repair only fires where padding cannot move an object: it declines a
2032    /// file whose xref precedes an object (an incremental update), rather than
2033    /// shifting every offset the table records.
2034    #[test]
2035    fn repair_declines_when_padding_would_move_objects() {
2036        let mut incremental = pdf_with_xref(false);
2037        incremental.extend_from_slice(b"6 0 obj<</Type/Whatever>>endobj\n");
2038        let declined = super::pad_short_xref_entries(&incremental).unwrap_err();
2039        assert!(
2040            declined.contains("object follows the xref"),
2041            "reason: {declined}"
2042        );
2043    }
2044}
2045
2046/// #187: standard-14 fonts referenced without an embedded program (and thus
2047/// usually without `/Widths` or a `/FontDescriptor`) must decode with the
2048/// built-in Adobe Core 14 metrics instead of collapsing every cell to zero
2049/// width — the failure mode where a valid text layer was silently dropped
2050/// while pdfium read the same file fine.
2051#[cfg(test)]
2052mod base14_fonts {
2053    /// A one-page PDF whose single `Tj` uses `fontdict` (no embedded program).
2054    fn pdf_with_font(fontdict: &[u8], text: &[u8]) -> Vec<u8> {
2055        let content = [b"BT /F1 12 Tf 72 700 Td (".as_slice(), text, b") Tj ET\n"].concat();
2056        let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2057        let objs: Vec<Vec<u8>> = vec![
2058            b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2059            b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2060            b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2061               /Resources<</Font<</F1 5 0 R>>>>>>"
2062                .to_vec(),
2063            [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
2064            fontdict.to_vec(),
2065        ];
2066        let mut out = b"%PDF-1.4\n".to_vec();
2067        let mut offsets = Vec::new();
2068        for (i, body) in objs.iter().enumerate() {
2069            offsets.push(out.len());
2070            out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2071            out.extend_from_slice(body);
2072            out.extend_from_slice(b"endobj\n");
2073        }
2074        let xref_at = out.len();
2075        out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2076        out.extend_from_slice(b"0000000000 65535 f \n");
2077        for off in &offsets {
2078            out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2079        }
2080        out.extend_from_slice(
2081            format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2082        );
2083        out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2084        out
2085    }
2086
2087    /// The parsed cells of the only page.
2088    fn cells(pdf: &[u8]) -> Vec<crate::pdfium_backend::TextCell> {
2089        super::pdf_textlines(pdf)
2090            .into_iter()
2091            .flat_map(|(_, _, c)| c)
2092            .collect()
2093    }
2094
2095    /// Every standard-14 alias/style decodes with real (positive-width) boxes.
2096    #[test]
2097    fn standard14_faces_get_builtin_widths() {
2098        for fontdict in [
2099            // ReportLab's default: base-14 Helvetica, WinAnsi, nothing else.
2100            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>".as_slice(),
2101            // No /Encoding at all (StandardEncoding-ish default).
2102            b"<</Type/Font/Subtype/Type1/BaseFont/Times-BoldItalic>>",
2103            // Substitution aliases + a subset prefix.
2104            b"<</Type/Font/Subtype/TrueType/BaseFont/Arial,Bold>>",
2105            b"<</Type/Font/Subtype/Type1/BaseFont/ABCDEF+Courier-Oblique>>",
2106        ] {
2107            let pdf = pdf_with_font(fontdict, b"Words have width now");
2108            let cs = cells(&pdf);
2109            let text: String = cs
2110                .iter()
2111                .map(|c| c.text.as_str())
2112                .collect::<Vec<_>>()
2113                .join(" ");
2114            assert!(
2115                text.contains("Words have width now"),
2116                "{}: text lost: {text:?}",
2117                String::from_utf8_lossy(fontdict)
2118            );
2119            assert!(
2120                cs.iter().all(|c| c.r > c.l),
2121                "{}: zero-width cells: {cs:?}",
2122                String::from_utf8_lossy(fontdict)
2123            );
2124        }
2125    }
2126
2127    /// An explicit `/Widths` array always wins over the built-in metrics, and a
2128    /// non-standard face without `/Widths` stays as before (no invented boxes).
2129    #[test]
2130    fn explicit_widths_win_and_unknown_faces_are_untouched() {
2131        // Helvetica with explicit 100/1000-em widths: the word's box must be
2132        // ~4×100 units at 12pt = 4.8pt wide — far narrower than the ~2.7×
2133        // wider built-in Helvetica advances would make it.
2134        let explicit = pdf_with_font(
2135            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/FirstChar 65\
2136               /Widths[100 100 100 100]/Encoding/WinAnsiEncoding>>",
2137            b"ABBA",
2138        );
2139        let builtin = pdf_with_font(
2140            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>",
2141            b"ABBA",
2142        );
2143        let w = |pdf: &[u8]| {
2144            let cs = cells(pdf);
2145            assert_eq!(cs.len(), 1, "one word cell: {cs:?}");
2146            cs[0].r - cs[0].l
2147        };
2148        let (we, wb) = (w(&explicit), w(&builtin));
2149        assert!(
2150            (we - 4.8).abs() < 0.1,
2151            "explicit widths must win: got {we}, want 4×100×12/1000"
2152        );
2153        assert!(
2154            wb > 2.0 * we,
2155            "built-in Helvetica is much wider: {wb} vs {we}"
2156        );
2157
2158        // An unknown face with no /Widths: still parses (text kept), but no
2159        // built-in table applies — the old zero-width behavior is preserved
2160        // rather than inventing Helvetica metrics for an arbitrary font.
2161        let unknown = pdf_with_font(
2162            b"<</Type/Font/Subtype/Type1/BaseFont/FancyCorp-Display>>",
2163            b"Mystery",
2164        );
2165        let cs = cells(&unknown);
2166        let text: String = cs.iter().map(|c| c.text.as_str()).collect();
2167        assert!(text.contains("Mystery"), "text still decodes: {cs:?}");
2168    }
2169}
2170
2171#[cfg(test)]
2172mod overpainted {
2173    use crate::pdfium_backend::TextCell;
2174
2175    fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
2176        TextCell {
2177            text: text.into(),
2178            l,
2179            t,
2180            r,
2181            b,
2182        }
2183    }
2184
2185    /// The reporting invoice's logo: a `"` painted inside a `==` on one band —
2186    /// artwork drawn with glyphs. Both cells go; the real text on the next
2187    /// band stays.
2188    #[test]
2189    fn stacked_logo_glyphs_are_dropped() {
2190        let mut cells = vec![
2191            cell("\"", 72.7, 21.5, 86.4, 31.5),
2192            cell("==", 59.4, 21.5, 99.6, 31.5),
2193            cell("Herr", 65.2, 151.3, 81.7, 161.3),
2194        ];
2195        super::drop_overpainted_cells(&mut cells);
2196        assert_eq!(cells.len(), 1, "cells: {cells:?}");
2197        assert_eq!(cells[0].text, "Herr");
2198    }
2199
2200    /// Adjacent words on a line touch but never contain each other — prose is
2201    /// untouched, and so is a same-text near-duplicate (double-drawn faux
2202    /// bold), which is not evidence of artwork.
2203    #[test]
2204    fn prose_and_double_draw_are_kept() {
2205        let mut cells = vec![
2206            cell("Telefon", 354.3, 133.2, 381.5, 143.2),
2207            cell("0676/2000", 387.3, 133.2, 428.7, 143.2),
2208            cell("Bold", 100.0, 50.0, 130.0, 60.0),
2209            cell("Bold", 100.3, 50.0, 130.3, 60.0),
2210        ];
2211        super::drop_overpainted_cells(&mut cells);
2212        assert_eq!(cells.len(), 4);
2213    }
2214}
2215
2216#[cfg(test)]
2217mod vestigial_layer {
2218    use crate::pdfium_backend::{PdfPage, TextCell};
2219
2220    fn page_with(texts: &[&str]) -> PdfPage {
2221        let cells = texts
2222            .iter()
2223            .enumerate()
2224            .map(|(i, t)| TextCell {
2225                text: t.to_string(),
2226                l: 10.0,
2227                t: 10.0 + 12.0 * i as f32,
2228                r: 90.0,
2229                b: 20.0 + 12.0 * i as f32,
2230            })
2231            .collect();
2232        PdfPage::from_cells(595.0, 842.0, 1.0, cells)
2233    }
2234
2235    /// The reported scanned form: three typed-in field values ("03", "05",
2236    /// "2025") over three image pages. That must read as *no usable layer*,
2237    /// so the browser routes the document to OCR instead of extracting
2238    /// thirteen characters and skipping the letter entirely.
2239    #[test]
2240    fn typed_in_form_fields_are_not_a_text_layer() {
2241        let pages = vec![
2242            page_with(&["03", "05", "2025"]),
2243            page_with(&[]),
2244            page_with(&[]),
2245        ];
2246        assert!(super::text_layer_is_vestigial(&pages));
2247        assert!(super::text_layer_is_vestigial(&[page_with(&[])]));
2248    }
2249
2250    /// A short but genuine digital document — one page, a few real lines —
2251    /// keeps the fast text path.
2252    #[test]
2253    fn sparse_but_real_documents_pass() {
2254        let one_pager = vec![page_with(&[
2255            "Confidential briefing",
2256            "Prepared for the board meeting",
2257            "Do not distribute",
2258        ])];
2259        assert!(!super::text_layer_is_vestigial(&one_pager));
2260    }
2261}