Skip to main content

docling_pdf/
textparse.rs

1//! Pure-Rust PDF text extraction (replacing pdfium's glyph layer).
2//!
3//! pdfium reports *rendered* glyph boxes, which diverge from docling's
4//! `docling-parse` C++ parser at exactly the points that drive conformance:
5//! generated spaces get a zero-width box, combining diacritics get a real-width
6//! box, and ligature/fraction glyphs land at different x. This module instead
7//! reconstructs each glyph's box from the **font's own advance widths** and the
8//! PDF text/graphics matrices — the same information docling-parse uses — so a
9//! space is as wide as the font says and a combining mark has zero advance.
10//!
11//! The output is the same [`Glyph`] stream pdfium produces (native PDF
12//! coordinates, y-up), fed straight into the existing docling-parse line
13//! sanitizer ([`crate::dp_lines`]). Only the digital text layer is handled here;
14//! pages without one still fall back to OCR upstream.
15
16use std::collections::HashMap;
17use std::sync::Arc;
18
19use lopdf::{Dictionary, Document, Object};
20
21use crate::pdfium_backend::Glyph;
22
23/// Per-document caches for the content-stream interpreter. Fonts are indirect
24/// objects shared by many pages, but were fully re-parsed — ToUnicode CMap
25/// decompression + tokenization, embedded Type1 program scan, width tables —
26/// for **every page and every Form XObject invocation**; decoded form content
27/// streams were likewise re-inflated on every `Do`. Cached per document,
28/// keyed by the referenced object id (fonts also by resource name, which
29/// feeds the docling-parse font hash). Inline (non-reference) dicts are rare
30/// and stay uncached.
31#[derive(Default)]
32struct DocCaches {
33    fonts: HashMap<(lopdf::ObjectId, Vec<u8>), Arc<Font>>,
34    forms: HashMap<lopdf::ObjectId, Arc<lopdf::content::Content>>,
35}
36
37/// A 2×3 affine matrix `[a b c d e f]`: maps `(x,y)` → `(a·x+c·y+e, b·x+d·y+f)`.
38#[derive(Clone, Copy)]
39struct Mat {
40    a: f64,
41    b: f64,
42    c: f64,
43    d: f64,
44    e: f64,
45    f: f64,
46}
47
48impl Mat {
49    const ID: Mat = Mat {
50        a: 1.0,
51        b: 0.0,
52        c: 0.0,
53        d: 1.0,
54        e: 0.0,
55        f: 0.0,
56    };
57
58    /// `self ∘ m`: the matrix that applies `self` first, then `m`.
59    fn then(self, m: Mat) -> Mat {
60        Mat {
61            a: self.a * m.a + self.b * m.c,
62            b: self.a * m.b + self.b * m.d,
63            c: self.c * m.a + self.d * m.c,
64            d: self.c * m.b + self.d * m.d,
65            e: self.e * m.a + self.f * m.c + m.e,
66            f: self.e * m.b + self.f * m.d + m.f,
67        }
68    }
69
70    fn apply(self, x: f64, y: f64) -> (f64, f64) {
71        (
72            self.a * x + self.c * y + self.e,
73            self.b * x + self.d * y + self.f,
74        )
75    }
76}
77
78/// A parsed font: how to turn raw string bytes into (unicode, advance) pairs.
79struct Font {
80    /// 2-byte codes (Type0 / Identity-H) vs 1-byte (simple fonts).
81    two_byte: bool,
82    /// code → Unicode string (from ToUnicode; may be multi-char, e.g. ligatures).
83    to_unicode: HashMap<u32, String>,
84    /// code → glyph advance, in 1000-unit glyph space.
85    widths: HashMap<u32, f64>,
86    default_width: f64,
87    /// 1-byte fallback decoding when ToUnicode lacks a code (WinAnsi-ish).
88    simple_encoding: Option<HashMap<u8, char>>,
89    /// code → raw `/Differences` glyph name, for GID-style names (`g115`) that
90    /// have no Unicode mapping. docling-parse emits these verbatim as `/g115`
91    /// (see the redp5110 bulleted list); matching it keeps no text skipped.
92    fallback_names: HashMap<u8, String>,
93    /// code → char from the embedded Type1 font program's own `/Encoding` vector,
94    /// used only as a last resort for glyphs the base encoding leaves unmapped
95    /// (standard TeX math fonts: `λ`, `≤`, …).
96    program_encoding: HashMap<u8, char>,
97    ascent: f64,
98    descent: f64,
99    hash: u64,
100}
101
102impl Font {
103    fn decode_code(&self, code: u32) -> (Option<String>, f64) {
104        let w = self
105            .widths
106            .get(&code)
107            .copied()
108            .unwrap_or(self.default_width);
109        if let Some(s) = self.to_unicode.get(&code) {
110            return (Some(decompose_ligatures(s)), w);
111        }
112        if !self.two_byte {
113            // A GID-style `/Differences` name (no Unicode) overrides the base
114            // encoding, matching docling's verbatim `/g115` fallback.
115            if let Some(name) = self.fallback_names.get(&(code as u8)) {
116                return (Some(format!("/{name}")), w);
117            }
118            if let Some(enc) = &self.simple_encoding {
119                if let Some(&ch) = enc.get(&(code as u8)) {
120                    return (Some(decompose_ligatures(&ch.to_string())), w);
121                }
122            }
123            // Last resort: the embedded Type1 font program's own `/Encoding`
124            // vector (`dup N /glyphname put`). Standard TeX math fonts (CMMI, CMSY,
125            // …) ship no PDF `/Encoding` and no ToUnicode, so a glyph like `λ`
126            // (CMMI code 21 → `/lambda`) or `≤` (CMSY code 20 → `/lessequal`) has
127            // no other mapping and would otherwise be silently dropped. docling
128            // recovers these from the same font program. This only fills codes the
129            // base encoding left unmapped, so it never changes an existing decode.
130            if let Some(&ch) = self.program_encoding.get(&(code as u8)) {
131                return (Some(decompose_ligatures(&ch.to_string())), w);
132            }
133        }
134        (None, w)
135    }
136}
137
138/// Spell out Latin presentation-form ligatures (`fi`→`fi`, `ffi`→`ffi`, …) the way
139/// docling does, so `configuration`/`difficult` don't keep the ligature glyph.
140/// The chars share the ligature's box, so the line sanitizer recomposes them.
141fn decompose_ligatures(s: &str) -> String {
142    if !s.chars().any(|c| ('\u{FB00}'..='\u{FB06}').contains(&c)) {
143        return s.to_string();
144    }
145    s.chars()
146        .map(|c| {
147            match c {
148                '\u{FB00}' => "ff",
149                '\u{FB01}' => "fi",
150                '\u{FB02}' => "fl",
151                '\u{FB03}' => "ffi",
152                '\u{FB04}' => "ffl",
153                '\u{FB05}' => "ft",
154                '\u{FB06}' => "st",
155                _ => return c.to_string(),
156            }
157            .to_string()
158        })
159        .collect()
160}
161
162fn hash_name(name: &[u8]) -> u64 {
163    use std::hash::{Hash, Hasher};
164    let mut h = std::collections::hash_map::DefaultHasher::new();
165    name.hash(&mut h);
166    h.finish()
167}
168
169/// Resolve a possibly-indirect object to a dictionary.
170fn as_dict<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Dictionary> {
171    match obj {
172        Object::Dictionary(d) => Some(d),
173        Object::Reference(id) => doc.get_object(*id).ok().and_then(|o| o.as_dict().ok()),
174        _ => None,
175    }
176}
177
178fn deref<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Object> {
179    match obj {
180        Object::Reference(id) => doc.get_object(*id).ok(),
181        other => Some(other),
182    }
183}
184
185/// Parse one font dictionary into a [`Font`].
186fn parse_font(doc: &Document, name: &[u8], fdict: &Dictionary) -> Font {
187    let subtype: &[u8] = fdict
188        .get(b"Subtype")
189        .ok()
190        .and_then(|o| o.as_name().ok())
191        .unwrap_or(&[]);
192    let two_byte = subtype == b"Type0".as_slice();
193
194    let to_unicode = fdict
195        .get(b"ToUnicode")
196        .ok()
197        .and_then(|o| deref(doc, o))
198        .and_then(|o| o.as_stream().ok())
199        .and_then(|s| s.decompressed_content().ok())
200        .map(|data| parse_tounicode(&data))
201        .unwrap_or_default();
202
203    let (mut widths, mut default_width) = if two_byte {
204        cid_widths(doc, fdict)
205    } else {
206        simple_widths(doc, fdict)
207    };
208
209    let simple_encoding = if two_byte {
210        None
211    } else {
212        Some(simple_encoding_table(doc, fdict))
213    };
214
215    // A standard-14 font referenced without an embedded program usually ships
216    // no `/Widths` and no `/FontDescriptor` either (ReportLab's default, #187).
217    // Every advance then resolved to 0, the cells collapsed to zero width, and
218    // the page's whole text layer was silently dropped — while pdfium, with
219    // its built-in metrics, reads the same file fine. Fill the widths from the
220    // built-in Adobe Core 14 AFM tables via the font's own code→char decode
221    // (base encoding + `/Differences`), so an explicit `/Widths` always wins.
222    if !two_byte && widths.is_empty() && default_width == 0.0 {
223        if let Some(std14) = base_font_name(fdict).and_then(|n| crate::std14::widths_for(&n)) {
224            if let Some(enc) = &simple_encoding {
225                for (&code, &ch) in enc {
226                    if let Some(w) = std14.width(ch) {
227                        widths.insert(u32::from(code), w);
228                    }
229                }
230            }
231            // Codes the table misses still advance a typical width instead of
232            // stacking at x=0 (the failure mode this whole branch fixes).
233            default_width = 500.0;
234        }
235    }
236    let fallback_names = if two_byte {
237        HashMap::new()
238    } else {
239        differences_gid_names(doc, fdict)
240    };
241    let program_encoding = if two_byte {
242        HashMap::new()
243    } else {
244        type1_program_encoding(doc, fdict)
245    };
246
247    let (ascent, descent) = font_ascent_descent(doc, fdict, two_byte);
248
249    Font {
250        two_byte,
251        to_unicode,
252        widths,
253        default_width,
254        simple_encoding,
255        fallback_names,
256        program_encoding,
257        ascent,
258        descent,
259        hash: hash_name(name),
260    }
261}
262
263/// Collect `/Differences` entries whose glyph name is a GID placeholder
264/// (`g115`, `cid42`, `glyph7`, `index9`) with no Unicode mapping. docling-parse
265/// emits such glyphs as the literal name `/g115`; mapping them here keeps the
266/// text from being silently dropped (subsetted fonts with no ToUnicode). The
267/// GID-name restriction keeps real Adobe glyph names on the normal path so this
268/// never invents garbage on the clean files.
269fn differences_gid_names(doc: &Document, fdict: &Dictionary) -> HashMap<u8, String> {
270    let mut map = HashMap::new();
271    let Some(Object::Dictionary(enc)) = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o))
272    else {
273        return map;
274    };
275    let Some(Object::Array(diffs)) = enc.get(b"Differences").ok().and_then(|o| deref(doc, o))
276    else {
277        return map;
278    };
279    let mut code = 0u8;
280    for el in diffs {
281        match el {
282            Object::Integer(i) => code = *i as u8,
283            Object::Name(name) => {
284                if glyph_name_to_char(name).is_none() && is_gid_name(name) {
285                    map.insert(code, String::from_utf8_lossy(name).into_owned());
286                }
287                code = code.wrapping_add(1);
288            }
289            _ => {}
290        }
291    }
292    map
293}
294
295/// Parse the embedded Type1 font program's built-in `/Encoding` vector
296/// (`dup <code> /<glyphname> put` entries in the clear-text header before
297/// `eexec`) into `code → char`. This is how docling recovers glyphs from
298/// standard TeX math fonts (CMMI/CMSY/…) that carry no PDF `/Encoding` and no
299/// ToUnicode — e.g. CMMI's `dup 21 /lambda` or CMSY's `dup 20 /lessequal`.
300/// Only `FontFile` (Type1) is parsed; CFF (`FontFile3`) and TrueType
301/// (`FontFile2`) store their encoding in a binary table and are left alone.
302fn type1_program_encoding(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
303    let mut map = HashMap::new();
304    let Some(desc) = fdict
305        .get(b"FontDescriptor")
306        .ok()
307        .and_then(|o| deref(doc, o))
308        .and_then(|o| o.as_dict().ok())
309    else {
310        return map;
311    };
312    let Some(data) = desc
313        .get(b"FontFile")
314        .ok()
315        .and_then(|o| deref(doc, o))
316        .and_then(|o| o.as_stream().ok())
317        .and_then(|s| s.decompressed_content().ok())
318    else {
319        return map;
320    };
321    // The clear-text header (PostScript) ends at `eexec`; the rest is encrypted.
322    let head_end = data
323        .windows(5)
324        .position(|w| w == b"eexec")
325        .unwrap_or(data.len());
326    let head = String::from_utf8_lossy(&data[..head_end]);
327    // Scan for `dup <code> /<name> put` tokens.
328    let toks: Vec<&str> = head.split_whitespace().collect();
329    for w in toks.windows(4) {
330        if w[0] == "dup" && w[3] == "put" {
331            if let (Ok(code), Some(name)) = (w[1].parse::<u32>(), w[2].strip_prefix('/')) {
332                if code <= 255 {
333                    if let Some(ch) = glyph_name_to_char(name.as_bytes()) {
334                        map.insert(code as u8, ch);
335                    }
336                }
337            }
338        }
339    }
340    map
341}
342
343/// A glyph name that is a synthetic placeholder, not a real Adobe name:
344/// `g115`, `cid42`, `glyph7`, `index9`, `G12`, or a short-prefix code name like
345/// `SM590000` (IBM BookMaster). These carry no Unicode meaning, and docling-parse
346/// emits them verbatim (`/SM590000`). `afii####` / `uni####` are real Adobe names
347/// and excluded. The restriction keeps genuine glyph names on the Unicode path.
348fn is_gid_name(name: &[u8]) -> bool {
349    let Ok(s) = std::str::from_utf8(name) else {
350        return false;
351    };
352    if s.starts_with("afii") || s.starts_with("uni") {
353        return false;
354    }
355    for prefix in ["g", "G", "cid", "CID", "glyph", "index"] {
356        if let Some(rest) = s.strip_prefix(prefix) {
357            if !rest.is_empty() && rest.bytes().all(|b| b.is_ascii_digit()) {
358                return true;
359            }
360        }
361    }
362    // Short alpha prefix (≤3 letters) followed by a run of ≥3 digits — synthetic
363    // code names like `SM590000`, distinct from real Adobe names (whole words or
364    // letter+`.suffix` variants).
365    let alpha = s.bytes().take_while(|b| b.is_ascii_alphabetic()).count();
366    let digits = s.len() - alpha;
367    (1..=3).contains(&alpha)
368        && digits >= 3
369        && s.as_bytes()[alpha..].iter().all(|b| b.is_ascii_digit())
370}
371
372fn font_ascent_descent(doc: &Document, fdict: &Dictionary, two_byte: bool) -> (f64, f64) {
373    // For Type0, the descriptor lives on the descendant CIDFont.
374    let descr_owner = if two_byte {
375        fdict
376            .get(b"DescendantFonts")
377            .ok()
378            .and_then(|o| deref(doc, o))
379            .and_then(|o| match o {
380                Object::Array(a) => a.first(),
381                _ => None,
382            })
383            .and_then(|o| as_dict(doc, o))
384    } else {
385        Some(fdict)
386    };
387    let fd = descr_owner
388        .and_then(|d| d.get(b"FontDescriptor").ok())
389        .and_then(|o| as_dict(doc, o));
390    let asc = fd
391        .and_then(|d| d.get(b"Ascent").ok())
392        .and_then(|o| {
393            o.as_float()
394                .ok()
395                .or_else(|| o.as_i64().ok().map(|i| i as f32))
396        })
397        .unwrap_or(750.0) as f64;
398    let desc = fd
399        .and_then(|d| d.get(b"Descent").ok())
400        .and_then(|o| {
401            o.as_float()
402                .ok()
403                .or_else(|| o.as_i64().ok().map(|i| i as f32))
404        })
405        .unwrap_or(-250.0) as f64;
406    // Some subsetted fonts carry a degenerate FontDescriptor (`/Ascent 0
407    // /Descent 0`) — the real metrics live in the font program. That collapses
408    // the loose box to zero height, so the line cells get zero area and the
409    // layout's region/text assignment drops them (2305's References list lost
410    // every prose line, keeping only the URLs). Fall back to typical text metrics
411    // so the box has height.
412    if asc - desc <= 1.0 {
413        return (750.0, -250.0);
414    }
415    (asc, desc)
416}
417
418/// The `/BaseFont` name with any `ABCDEF+` subset prefix stripped (#187).
419fn base_font_name(fdict: &Dictionary) -> Option<Vec<u8>> {
420    let name = fdict.get(b"BaseFont").ok()?.as_name().ok()?;
421    let stripped = match name.iter().position(|&b| b == b'+') {
422        Some(i) if i == 6 => &name[i + 1..],
423        _ => name,
424    };
425    Some(stripped.to_vec())
426}
427
428/// Simple-font widths: `/FirstChar` + `/Widths` array, `/MissingWidth` default.
429fn simple_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
430    let mut map = HashMap::new();
431    let first = fdict
432        .get(b"FirstChar")
433        .ok()
434        .and_then(|o| o.as_i64().ok())
435        .unwrap_or(0) as u32;
436    if let Some(Object::Array(arr)) = fdict.get(b"Widths").ok().and_then(|o| deref(doc, o)) {
437        for (i, w) in arr.iter().enumerate() {
438            if let Some(w) = num(w) {
439                map.insert(first + i as u32, w);
440            }
441        }
442    }
443    let dw = fdict
444        .get(b"FontDescriptor")
445        .ok()
446        .and_then(|o| as_dict(doc, o))
447        .and_then(|d| d.get(b"MissingWidth").ok())
448        .and_then(num)
449        .unwrap_or(0.0);
450    (map, dw)
451}
452
453/// CIDFont widths: the `/W` array on the descendant font (`/DW` default = 1000).
454fn cid_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
455    let mut map = HashMap::new();
456    let Some(desc) = fdict
457        .get(b"DescendantFonts")
458        .ok()
459        .and_then(|o| deref(doc, o))
460        .and_then(|o| match o {
461            Object::Array(a) => a.first(),
462            _ => None,
463        })
464        .and_then(|o| as_dict(doc, o))
465    else {
466        return (map, 1000.0);
467    };
468    let dw = desc.get(b"DW").ok().and_then(num).unwrap_or(1000.0);
469    if let Some(Object::Array(w)) = desc.get(b"W").ok().and_then(|o| deref(doc, o)) {
470        let mut i = 0;
471        while i < w.len() {
472            let c = w.get(i).and_then(num);
473            match (c, w.get(i + 1)) {
474                // `c [w1 w2 ...]`: consecutive CIDs starting at c.
475                (Some(c), Some(Object::Array(list))) => {
476                    for (k, wv) in list.iter().enumerate() {
477                        if let Some(wv) = num(wv) {
478                            map.insert(c as u32 + k as u32, wv);
479                        }
480                    }
481                    i += 2;
482                }
483                // `c_first c_last w`: a run all of width w.
484                (Some(c1), Some(o2)) => {
485                    if let (Some(c2), Some(wv)) = (num(o2), w.get(i + 2).and_then(num)) {
486                        for cid in c1 as u32..=c2 as u32 {
487                            map.insert(cid, wv);
488                        }
489                    }
490                    i += 3;
491                }
492                _ => break,
493            }
494        }
495    }
496    (map, dw)
497}
498
499fn num(o: &Object) -> Option<f64> {
500    match o {
501        Object::Integer(i) => Some(*i as f64),
502        Object::Real(r) => Some(*r as f64),
503        _ => None,
504    }
505}
506
507/// Parse a ToUnicode CMap's `bfchar` / `bfrange` sections into code→string.
508fn parse_tounicode(data: &[u8]) -> HashMap<u32, String> {
509    let text = String::from_utf8_lossy(data);
510    let mut map = HashMap::new();
511    let hex = |s: &str| -> Option<Vec<u16>> {
512        let s = s.trim();
513        if !s.starts_with('<') || !s.ends_with('>') {
514            return None;
515        }
516        let h = &s[1..s.len() - 1];
517        let bytes: Vec<u8> = (0..h.len())
518            .step_by(2)
519            .filter_map(|i| u8::from_str_radix(h.get(i..i + 2)?, 16).ok())
520            .collect();
521        Some(
522            bytes
523                .chunks(2)
524                .map(|c| {
525                    if c.len() == 2 {
526                        u16::from_be_bytes([c[0], c[1]])
527                    } else {
528                        c[0] as u16
529                    }
530                })
531                .collect(),
532        )
533    };
534    let u16s_to_string = |u: &[u16]| String::from_utf16_lossy(u);
535    let code_of = |u: &[u16]| u.iter().fold(0u32, |acc, &x| (acc << 16) | x as u32);
536
537    // Tokenize by structure, not whitespace: CMap hex groups are often written
538    // back-to-back with no separators (`<21><21><0054>`), so scan for `<…>`
539    // groups, `[`/`]` brackets, and bareword keywords.
540    let tokens: Vec<String> = {
541        let bytes = text.as_bytes();
542        let mut toks = Vec::new();
543        let mut i = 0;
544        while i < bytes.len() {
545            let c = bytes[i];
546            if c.is_ascii_whitespace() {
547                i += 1;
548            } else if c == b'<' {
549                let start = i;
550                while i < bytes.len() && bytes[i] != b'>' {
551                    i += 1;
552                }
553                i += 1; // include '>'
554                toks.push(String::from_utf8_lossy(&bytes[start..i.min(bytes.len())]).into_owned());
555            } else if c == b'[' || c == b']' {
556                toks.push((c as char).to_string());
557                i += 1;
558            } else {
559                let start = i;
560                while i < bytes.len()
561                    && !bytes[i].is_ascii_whitespace()
562                    && bytes[i] != b'<'
563                    && bytes[i] != b'['
564                    && bytes[i] != b']'
565                {
566                    i += 1;
567                }
568                toks.push(String::from_utf8_lossy(&bytes[start..i]).into_owned());
569            }
570        }
571        toks
572    };
573    let tokens: Vec<&str> = tokens.iter().map(|s| s.as_str()).collect();
574    let mut i = 0;
575    while i < tokens.len() {
576        match tokens[i] {
577            "beginbfchar" => {
578                i += 1;
579                while i + 1 < tokens.len() && tokens[i] != "endbfchar" {
580                    if let (Some(src), Some(dst)) = (hex(tokens[i]), hex(tokens[i + 1])) {
581                        map.insert(code_of(&src), u16s_to_string(&dst));
582                    }
583                    i += 2;
584                }
585            }
586            "beginbfrange" => {
587                i += 1;
588                while i + 2 < tokens.len() && tokens[i] != "endbfrange" {
589                    let (Some(lo), Some(hi)) = (hex(tokens[i]), hex(tokens[i + 1])) else {
590                        i += 1;
591                        continue;
592                    };
593                    let lo = code_of(&lo);
594                    let hi = code_of(&hi);
595                    if tokens[i + 2] == "[" {
596                        // `<lo> <hi> [ <d0> <d1> ... ]`: one dst per code in the range.
597                        let mut j = i + 3;
598                        let mut code = lo;
599                        while j < tokens.len() && tokens[j] != "]" {
600                            if let Some(dst) = hex(tokens[j]) {
601                                map.insert(code, u16s_to_string(&dst));
602                            }
603                            code += 1;
604                            j += 1;
605                        }
606                        i = j + 1;
607                    } else if let Some(dst) = hex(tokens[i + 2]) {
608                        // `<lo> <hi> <dst>`: consecutive Unicode from a base.
609                        let base = code_of(&dst);
610                        for (k, code) in (lo..=hi).enumerate() {
611                            if let Some(ch) = char::from_u32(base + k as u32) {
612                                map.insert(code, ch.to_string());
613                            }
614                        }
615                        i += 3;
616                    } else {
617                        i += 1;
618                    }
619                }
620            }
621            _ => i += 1,
622        }
623    }
624    map
625}
626
627/// Decode a PDF string literal in a Tj/TJ operand into raw code units.
628fn codes(font: &Font, bytes: &[u8]) -> Vec<u32> {
629    if font.two_byte {
630        bytes
631            .chunks(2)
632            .map(|c| {
633                if c.len() == 2 {
634                    ((c[0] as u32) << 8) | c[1] as u32
635                } else {
636                    c[0] as u32
637                }
638            })
639            .collect()
640    } else {
641        bytes.iter().map(|&b| b as u32).collect()
642    }
643}
644
645/// Page size (width, height) in PDF points from the MediaBox.
646fn page_size(doc: &Document, page_id: lopdf::ObjectId) -> (f32, f32) {
647    let mb = doc
648        .get_object(page_id)
649        .ok()
650        .and_then(|o| o.as_dict().ok())
651        .and_then(|d| {
652            // MediaBox may be inherited; lopdf resolves via get_page... fall back to a guess.
653            d.get(b"MediaBox").ok().cloned()
654        })
655        .or_else(|| {
656            doc.get_dictionary(page_id)
657                .ok()
658                .and_then(|d| d.get(b"MediaBox").ok().cloned())
659        });
660    if let Some(Object::Array(a)) = mb {
661        let v: Vec<f32> = a.iter().filter_map(|o| num(o).map(|x| x as f32)).collect();
662        if v.len() == 4 {
663            return ((v[2] - v[0]).abs(), (v[3] - v[1]).abs());
664        }
665    }
666    (612.0, 792.0)
667}
668
669/// Localize where a page's text is lost, for the `text_layer` diagnostic.
670/// Extraction can come up empty at three different points — no content stream
671/// reached the parser, the stream did not decode into operators, or it ran but
672/// produced no glyphs (fonts/encodings) — and from the outside all three look
673/// the same. Report them per page.
674pub fn content_diagnosis(bytes: &[u8]) -> String {
675    let Some(doc) = load_document(bytes) else {
676        return "document does not load".into();
677    };
678    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
679    pages.sort_by_key(|(n, _)| *n);
680    let mut out = String::new();
681    let mut caches = DocCaches::default();
682    for (n, pid) in pages.into_iter().take(4) {
683        let content_bytes = doc.get_page_content(pid);
684        let ops = lopdf::content::Content::decode(&content_bytes)
685            .map(|c| c.operations.len())
686            .ok();
687        let res = page_res(&doc, pid);
688        let fonts = res.map(|r| fonts_from_res(&doc, r, &mut caches).len());
689        let glyphs = page_glyphs_cached(&doc, pid, &mut caches).len();
690        out.push_str(&format!(
691            "\n   page {n}: content {} B, ops {}, resources {}, fonts {}, glyphs {}",
692            content_bytes.len(),
693            ops.map_or("UNDECODABLE".to_string(), |n| n.to_string()),
694            if res.is_some() { "ok" } else { "MISSING" },
695            fonts.map_or("-".to_string(), |n| n.to_string()),
696            glyphs,
697        ));
698    }
699    out
700}
701
702/// Is this "text layer" a vestige rather than the document's text?
703///
704/// Scanned forms often carry a handful of typed-in strings — a date filled
705/// into three form fields, say — on top of pages that are otherwise images.
706/// Treating that as a real text layer is the worst of both worlds: the text
707/// path proudly extracts thirteen characters, and no OCR ever runs on the
708/// letter the pages actually show. The reported form did exactly this (3
709/// lines, 13 chars, 3 pages).
710///
711/// The rule is deliberately tight so genuinely sparse *digital* documents are
712/// not misrouted into OCR: only a document averaging at most one line per page
713/// **and** totalling fewer than 32 characters is called vestigial.
714pub fn text_layer_is_vestigial(pages: &[crate::pdfium_backend::PdfPage]) -> bool {
715    let lines: usize = pages.iter().map(|p| p.cells.len()).sum();
716    if lines == 0 {
717        return true;
718    }
719    let chars: usize = pages
720        .iter()
721        .flat_map(|p| &p.cells)
722        .map(|c| c.text.chars().count())
723        .sum();
724    lines <= pages.len() && chars < 32
725}
726
727/// Why the cross-reference repair did or did not fire, for the `text_layer`
728/// diagnostic. A PDF that will not load is indistinguishable from a scan in
729/// production (both convert to nothing), so the reason has to be askable.
730pub fn xref_repair_status(bytes: &[u8]) -> String {
731    if Document::load_mem(bytes).is_ok() {
732        return "loads unaided; no repair needed".into();
733    }
734    match pad_short_xref_entries(bytes) {
735        Ok(fixed) => match Document::load_mem(&fixed) {
736            Ok(_) => "repaired: cross-reference entries padded to 20 bytes".into(),
737            Err(e) => format!("padded the entries, but it still will not load: {e}"),
738        },
739        Err(why) => format!("repair declined — {why}"),
740    }
741}
742
743/// Load a PDF, repairing the one malformation that otherwise costs us the whole
744/// document: **19-byte cross-reference entries**.
745///
746/// The spec fixes an xref entry at 20 bytes — `nnnnnnnnnn ggggg n` plus a
747/// *two*-byte EOL. Some generators (an Austrian telecom's invoices, for one)
748/// emit a bare LF instead, making each entry 19 bytes. lopdf rejects the file
749/// outright (`invalid file trailer`) where pdfium reads it happily, so a
750/// perfectly good text layer looked to the browser exactly like a scan and cost
751/// ten seconds of OCR.
752///
753/// Padding is only attempted when it cannot move anything the xref points at:
754/// a single `xref` section that begins after the last object. The repair then
755/// has to prove itself — the padded bytes are used only if they load — so a
756/// mis-repair degrades to today's behaviour rather than to silent garbage.
757fn load_document(bytes: &[u8]) -> Option<Document> {
758    // Try progressively more repair, and accept a candidate only once the pages
759    // actually carry content — a document whose streams were dropped still
760    // "loads", so loading alone is not evidence the repair helped. A
761    // well-formed file returns on the first attempt and pays for nothing.
762    let mut fallback = None;
763    if let Some(doc) = best_effort_load(bytes, &mut fallback) {
764        return Some(doc);
765    }
766    let xref_fixed = pad_short_xref_entries(bytes).ok();
767    if let Some(fixed) = &xref_fixed {
768        if let Some(doc) = best_effort_load(fixed, &mut fallback) {
769            return Some(doc);
770        }
771    }
772    // Both defects can coexist, and the second only becomes visible once the
773    // first is repaired, so build on whatever the previous step produced.
774    let lengths_fixed = fix_stream_lengths(xref_fixed.as_deref().unwrap_or(bytes));
775    if let Some(doc) = best_effort_load(&lengths_fixed, &mut fallback) {
776        return Some(doc);
777    }
778    fallback
779}
780
781/// Load `data`, returning it only when its pages carry content; a document that
782/// merely parses is remembered as the fallback for when nothing does better.
783fn best_effort_load(data: &[u8], fallback: &mut Option<Document>) -> Option<Document> {
784    match Document::load_mem(data) {
785        Ok(doc) if has_page_content(&doc) => Some(doc),
786        Ok(doc) => {
787            fallback.get_or_insert(doc);
788            None
789        }
790        Err(_) => None,
791    }
792}
793
794/// Does any page actually hand us a content stream? A document whose streams
795/// were dropped still parses — it simply has nothing to read — so this is what
796/// tells a successful repair from a pointless one.
797fn has_page_content(doc: &Document) -> bool {
798    doc.get_pages()
799        .into_values()
800        .take(4)
801        .any(|pid| !doc.get_page_content(pid).is_empty())
802}
803
804/// Correct `/Length` values that disagree with where `endstream` actually is.
805///
806/// The same generator that writes short xref entries also overstates its
807/// content-stream lengths by a byte or two. lopdf trusts `/Length`, reads past
808/// the data, fails to find `endstream` there and drops the stream — the object
809/// comes back as a bare dictionary, so the page has no content at all and the
810/// document looks like a scan. pdfium instead trusts `endstream`, which is what
811/// this does.
812///
813/// The rewrite is length-preserving: the corrected number is written over the
814/// old digits and padded with spaces, so every byte offset in the file — and
815/// therefore the whole cross-reference table — stays valid.
816fn fix_stream_lengths(bytes: &[u8]) -> Vec<u8> {
817    let mut out = bytes.to_vec();
818    let mut i = 0;
819    while let Some(rel) = find(&out[i..], b"stream") {
820        let kw = i + rel;
821        i = kw + 6;
822        // Skip `endstream` (the keyword we are measuring *to*).
823        if kw >= 3 && &out[kw - 3..kw] == b"end" {
824            continue;
825        }
826        // The stream data starts after the EOL that follows the keyword.
827        let mut data = kw + 6;
828        if out.get(data..data + 2) == Some(b"\r\n".as_slice()) {
829            data += 2;
830        } else if matches!(out.get(data), Some(b'\n' | b'\r')) {
831            data += 1;
832        }
833        let Some(end) = find(&out[data..], b"endstream").map(|r| data + r) else {
834            continue;
835        };
836        // `/Length <digits>` in the dictionary just before the keyword.
837        let dict_start = out[..kw].iter().rposition(|&c| c == b'<').unwrap_or(0);
838        let Some(lrel) = find(&out[dict_start..kw], b"/Length") else {
839            continue;
840        };
841        let mut d = dict_start + lrel + 7;
842        while matches!(out.get(d), Some(b' ')) {
843            d += 1;
844        }
845        let digits = out[d..].iter().take_while(|c| c.is_ascii_digit()).count();
846        if digits == 0 {
847            continue;
848        }
849        let declared: usize = match std::str::from_utf8(&out[d..d + digits])
850            .ok()
851            .and_then(|s| s.parse().ok())
852        {
853            Some(v) => v,
854            None => continue,
855        };
856        let actual = end - data;
857        // Only shrink, and only when the new value fits the space the old one
858        // occupied — growing the number would move every following byte.
859        let replacement = actual.to_string();
860        if actual == declared || replacement.len() > digits {
861            continue;
862        }
863        out[d..d + digits].fill(b' ');
864        out[d..d + replacement.len()].copy_from_slice(replacement.as_bytes());
865    }
866    out
867}
868
869fn find(haystack: &[u8], needle: &[u8]) -> Option<usize> {
870    haystack.windows(needle.len()).position(|w| w == needle)
871}
872
873/// Rewrite a classic cross-reference table's entries to the spec's 20 bytes,
874/// or `None` when the file's shape makes that unsafe (see [`load_document`]).
875fn pad_short_xref_entries(bytes: &[u8]) -> Result<Vec<u8>, &'static str> {
876    // Exactly one xref section, and it must start after every object, so that
877    // growing it shifts nothing the table's offsets refer to.
878    let is_boundary = |i: usize| i == 0 || matches!(bytes[i - 1], b'\n' | b'\r');
879    let mut starts = (0..bytes.len().saturating_sub(4))
880        .filter(|&i| &bytes[i..i + 4] == b"xref" && is_boundary(i));
881    let xref_at = starts
882        .next()
883        .ok_or("no classic `xref` section (an xref stream?)")?;
884    if starts.next().is_some() {
885        return Err("more than one xref section (incremental update)");
886    }
887    let last_obj = bytes
888        .windows(3)
889        .rposition(|w| w == b"obj")
890        .ok_or("no objects found")?;
891    if last_obj > xref_at {
892        return Err("an object follows the xref — padding would move it");
893    }
894
895    let mut out = bytes[..xref_at].to_vec();
896    out.extend_from_slice(b"xref\n");
897    let mut i = xref_at + 4;
898    let skip_ws = |i: &mut usize| {
899        while matches!(bytes.get(*i), Some(b'\r' | b'\n' | b' ')) {
900            *i += 1;
901        }
902    };
903    loop {
904        skip_ws(&mut i);
905        // Either the next subsection header ("first count") or the trailer.
906        if bytes[i..].starts_with(b"trailer") {
907            out.extend_from_slice(&bytes[i..]);
908            return Ok(out);
909        }
910        let header_end = i + bytes[i..]
911            .iter()
912            .position(|c| matches!(c, b'\n' | b'\r'))
913            .ok_or("subsection header runs off the end")?;
914        let header = std::str::from_utf8(&bytes[i..header_end])
915            .map_err(|_| "subsection header is not text")?
916            .trim();
917        let mut parts = header.split_whitespace();
918        let count: usize = parts
919            .nth(1)
920            .and_then(|c| c.parse().ok())
921            .ok_or("unparseable subsection header")?;
922        if parts.next().is_some() || count == 0 {
923            return Err("unexpected subsection header shape");
924        }
925        out.extend_from_slice(header.as_bytes());
926        out.push(b'\n');
927        i = header_end;
928        for _ in 0..count {
929            skip_ws(&mut i);
930            // `nnnnnnnnnn ggggg n` — the 18 bytes before whatever EOL follows.
931            let entry = bytes.get(i..i + 18).ok_or("xref entry runs off the end")?;
932            let well_formed = entry[..10].iter().all(u8::is_ascii_digit)
933                && entry[10] == b' '
934                && entry[11..16].iter().all(u8::is_ascii_digit)
935                && entry[16] == b' '
936                && matches!(entry[17], b'n' | b'f');
937            if !well_formed {
938                return Err("xref entry is not `nnnnnnnnnn ggggg n`");
939            }
940            out.extend_from_slice(entry);
941            out.extend_from_slice(b" \n"); // the spec's 2-byte EOL -> 20 bytes
942            i += 18;
943        }
944    }
945}
946
947/// Debug: raw glyph stream `(ch, ll, lr, lb, lt)` (native coords) for page
948/// `index`, before the sanitizer. For comparing char cells to docling-parse.
949pub fn debug_glyphs(bytes: &[u8], index: usize) -> Vec<(char, f32, f32, f32, f32)> {
950    let Some(doc) = load_document(bytes) else {
951        return Vec::new();
952    };
953    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
954    pages.sort_by_key(|(n, _)| *n);
955    let Some((_, pid)) = pages.get(index) else {
956        return Vec::new();
957    };
958    page_glyphs(&doc, *pid)
959        .into_iter()
960        .map(|g| (g.ch, g.ll, g.lr, g.lb, g.lt))
961        .collect()
962}
963
964/// Public entry: per-page (width, height, line cells) for a PDF, via the Rust
965/// text parser + the docling-parse line sanitizer. Used by the pipeline and the
966/// `textparse_dump` example.
967pub fn pdf_textlines(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
968    let Some(doc) = load_document(bytes) else {
969        return Vec::new();
970    };
971    let mut caches = DocCaches::default();
972    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
973    pages.sort_by_key(|(n, _)| *n);
974    pages
975        .into_iter()
976        .map(|(_, pid)| {
977            let (w, h) = page_size(&doc, pid);
978            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
979            let cells = crate::dp_lines::line_cells(&glyphs, h, true);
980            (w, h, cells)
981        })
982        .collect()
983}
984
985/// Debug/diagnostic entry: per-page (width, height, word cells) for a PDF, via
986/// the Rust parser glyphs run through the docling-parse word grouping. Used to
987/// compare parser word cells against docling-parse's `word_cells` oracle (roadmap
988/// item 6).
989pub fn pdf_words(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
990    let Some(doc) = load_document(bytes) else {
991        return Vec::new();
992    };
993    let mut caches = DocCaches::default();
994    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
995    pages.sort_by_key(|(n, _)| *n);
996    pages
997        .into_iter()
998        .map(|(_, pid)| {
999            let (w, h) = page_size(&doc, pid);
1000            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1001            let cells = crate::dp_lines::word_cells(&glyphs, h, true);
1002            (w, h, cells)
1003        })
1004        .collect()
1005}
1006
1007/// One page's text cells from the pure-Rust parser: prose line cells, per-word
1008/// cells, and code line cells — all from a single glyph parse. Replaces the
1009/// pdfium text path (roadmap item 6) when the parser drop is enabled.
1010#[derive(Default)]
1011pub struct PageParserCells {
1012    pub prose: Vec<crate::pdfium_backend::TextCell>,
1013    pub words: Vec<crate::pdfium_backend::TextCell>,
1014    pub code: Vec<crate::pdfium_backend::TextCell>,
1015}
1016
1017/// The parser text layer, driven one page at a time: the document is loaded
1018/// (and repaired, see [`load_document`]) once, the font/form caches persist
1019/// across pages, and each page's glyphs are parsed only when asked for.
1020///
1021/// The eager whole-document walk this replaces ran *before* the first page
1022/// was rendered, so on a long PDF it was a serial prefix the page-worker pool
1023/// sat idle through — 6.2 s on the 1913-page .NET reference, in front of a
1024/// pipeline that otherwise overlaps parsing with inference — and a `--pages`
1025/// window still paid for every page in the file. Pulling pages on demand
1026/// keeps the parse on the producer thread but interleaved with rendering,
1027/// and skips unselected pages entirely. Output per page is unchanged: same
1028/// glyph walk, same shared caches, same contraction.
1029pub struct PageTextParser {
1030    doc: Document,
1031    caches: DocCaches,
1032    /// Page object ids in document order (page 1 first).
1033    pages: Vec<lopdf::ObjectId>,
1034}
1035
1036impl PageTextParser {
1037    /// Load the document; `None` when it has no parseable text layer at all
1038    /// (the caller then keeps pdfium's cells, as before).
1039    pub fn open(bytes: &[u8]) -> Option<Self> {
1040        let doc = load_document(bytes)?;
1041        let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1042        pages.sort_by_key(|(n, _)| *n);
1043        Some(Self {
1044            doc,
1045            caches: DocCaches::default(),
1046            pages: pages.into_iter().map(|(_, pid)| pid).collect(),
1047        })
1048    }
1049
1050    /// Prose, word and code cells of the 0-based page `index` — empty for an
1051    /// index the parser's page tree doesn't have (pdfium and lopdf can
1052    /// disagree on a damaged file; the caller falls back to pdfium's text).
1053    pub fn cells(&mut self, index: usize) -> PageParserCells {
1054        let Some(&pid) = self.pages.get(index) else {
1055            return PageParserCells::default();
1056        };
1057        let (_w, h) = page_size(&self.doc, pid);
1058        let glyphs = page_glyphs_cached(&self.doc, pid, &mut self.caches);
1059        let (prose, words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1060        PageParserCells {
1061            prose,
1062            words,
1063            code: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1064        }
1065    }
1066}
1067
1068/// Full parser text layer: prose + word + code cells per page, glyphs parsed once.
1069/// `prose`/`words` come from the docling-parse contraction ([`crate::dp_lines`]);
1070/// `code` splits only at the parser's own space glyphs (monospace keeps its
1071/// source spacing). The eager form of [`PageTextParser`].
1072pub fn pdf_all_cells(bytes: &[u8]) -> Vec<PageParserCells> {
1073    let Some(mut parser) = PageTextParser::open(bytes) else {
1074        return Vec::new();
1075    };
1076    (0..parser.pages.len()).map(|i| parser.cells(i)).collect()
1077}
1078
1079/// Whole pages for the text-layer-only conversion ([`crate::convert_text_layer`]):
1080/// the parser's prose/word/code cells plus page geometry, assembled into
1081/// [`PdfPage`]s with no rendered image and no link annotations. Everything here
1082/// is pure Rust (lopdf), so it compiles without the `ml` feature — including on
1083/// wasm32. A page the parser can't read (no text layer) comes back with empty
1084/// cells; there is no pdfium fallback on this path.
1085pub fn pdf_text_pages(bytes: &[u8]) -> Vec<crate::pdfium_backend::PdfPage> {
1086    let Some(doc) = load_document(bytes) else {
1087        return Vec::new();
1088    };
1089    let mut caches = DocCaches::default();
1090    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1091    pages.sort_by_key(|(n, _)| *n);
1092    pages
1093        .into_iter()
1094        .map(|(_, pid)| {
1095            let (w, h) = page_size(&doc, pid);
1096            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1097            let (mut prose, mut words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1098            drop_overpainted_cells(&mut prose);
1099            drop_overpainted_cells(&mut words);
1100            crate::pdfium_backend::PdfPage {
1101                #[cfg(feature = "ocr-prep")]
1102                image_layout: None,
1103                width: w,
1104                height: h,
1105                // Cells are native PDF points; there is no rendered bitmap.
1106                scale: 1.0,
1107                cells: prose,
1108                code_cells: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1109                word_cells: words,
1110                #[cfg(feature = "ocr-prep")]
1111                image: image::RgbImage::new(1, 1),
1112                links: Vec::new(),
1113                rotation: 0,
1114            }
1115        })
1116        .collect()
1117}
1118
1119/// Drop line cells that are *painted over each other* — glyphs used as artwork.
1120///
1121/// Some generators draw their logo with a symbol font: on the reporting
1122/// invoice, a `TeleLogo` Type1 paints the T-Mobile mark by stacking the glyphs
1123/// encoded as `"` and `==` on top of one another, and the flat text-layer
1124/// output opened with that garbage. Nothing in the font metadata gives it away
1125/// (the *text* fonts in the same file are also flagged symbolic, and the logo
1126/// font names its glyphs `quotedbl` &c.), but the geometry does: two cells with
1127/// different text where one lies inside the other on the same line is
1128/// physically impossible for prose — ink from two words never occupies the
1129/// same box. Both cells of such a pair are paint, not text.
1130///
1131/// Containment (not mere overlap) keeps this narrow: adjacent words touch but
1132/// never contain each other, and a same-text near-duplicate (double-draw faux
1133/// bold) is left alone for the sanitizer's usual handling. Applied on the
1134/// flat/browser path only — the ML pipeline's text layer is byte-pinned by the
1135/// PDF corpus, and there the layout model already sinks logo marks into
1136/// `picture` regions.
1137fn drop_overpainted_cells(cells: &mut Vec<crate::pdfium_backend::TextCell>) {
1138    let mut paint = vec![false; cells.len()];
1139    for i in 0..cells.len() {
1140        for j in 0..cells.len() {
1141            if i == j || cells[i].text == cells[j].text {
1142                continue;
1143            }
1144            let (a, b) = (&cells[i], &cells[j]);
1145            // Same line band: the vertical overlap covers most of the shorter.
1146            let vo = (a.b.min(b.b) - a.t.max(b.t)).max(0.0);
1147            if vo < 0.6 * (a.b - a.t).min(b.b - b.t) {
1148                continue;
1149            }
1150            // `a` horizontally inside `b` (with a small tolerance).
1151            let ho = (a.r.min(b.r) - a.l.max(b.l)).max(0.0);
1152            if ho >= 0.8 * (a.r - a.l) && (a.r - a.l) <= (b.r - b.l) {
1153                paint[i] = true;
1154                paint[j] = true;
1155            }
1156        }
1157    }
1158    let mut keep = paint.iter().map(|p| !p);
1159    cells.retain(|_| keep.next().unwrap());
1160}
1161
1162/// The text-state scalars inherited by a Form XObject when it is invoked via
1163/// `Do` (the PDF graphics state includes the text parameters, but not the text
1164/// matrices, which a form re-establishes inside its own `BT`/`ET`).
1165#[derive(Clone, Copy)]
1166struct TextState {
1167    tc: f64,
1168    tw: f64,
1169    th: f64,
1170    tl: f64,
1171    trise: f64,
1172    fsize: f64,
1173}
1174
1175impl TextState {
1176    const INIT: TextState = TextState {
1177        tc: 0.0,
1178        tw: 0.0,
1179        th: 1.0,
1180        tl: 0.0,
1181        trise: 0.0,
1182        fsize: 0.0,
1183    };
1184}
1185
1186/// The effective `/Resources` dictionary for a page (inline or via reference,
1187/// falling back to an inherited one from a `/Parent`).
1188fn page_res(doc: &Document, page_id: lopdf::ObjectId) -> Option<&Dictionary> {
1189    let (inline, ids) = doc.get_page_resources(page_id).ok()?;
1190    if let Some(d) = inline {
1191        return Some(d);
1192    }
1193    ids.into_iter().find_map(|id| doc.get_dictionary(id).ok())
1194}
1195
1196/// Build the code→[`Font`] map for a resources dictionary's `/Font` sub-dict,
1197/// reusing the per-document cache for fonts referenced indirectly (the common
1198/// case — the same font objects recur on every page).
1199fn fonts_from_res(
1200    doc: &Document,
1201    res: &Dictionary,
1202    caches: &mut DocCaches,
1203) -> HashMap<Vec<u8>, Arc<Font>> {
1204    let mut map = HashMap::new();
1205    let font_dict = res
1206        .get(b"Font")
1207        .ok()
1208        .and_then(|o| deref(doc, o))
1209        .and_then(|o| o.as_dict().ok());
1210    if let Some(fd) = font_dict {
1211        for (name, value) in fd.iter() {
1212            let font = match value {
1213                Object::Reference(id) => {
1214                    let key = (*id, name.clone());
1215                    if let Some(f) = caches.fonts.get(&key) {
1216                        Arc::clone(f)
1217                    } else if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1218                        let f = Arc::new(parse_font(doc, name, fdict));
1219                        caches.fonts.insert(key, Arc::clone(&f));
1220                        f
1221                    } else {
1222                        continue;
1223                    }
1224                }
1225                _ => {
1226                    if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1227                        Arc::new(parse_font(doc, name, fdict))
1228                    } else {
1229                        continue;
1230                    }
1231                }
1232            };
1233            map.insert(name.clone(), font);
1234        }
1235    }
1236    map
1237}
1238
1239/// Extract every glyph on a page as a native-coordinate [`Glyph`].
1240pub(crate) fn page_glyphs(doc: &Document, page_id: lopdf::ObjectId) -> Vec<Glyph> {
1241    page_glyphs_cached(doc, page_id, &mut DocCaches::default())
1242}
1243
1244/// [`page_glyphs`] with an explicit per-document cache, so a multi-page walk
1245/// parses each font / decodes each form once instead of once per page.
1246fn page_glyphs_cached(
1247    doc: &Document,
1248    page_id: lopdf::ObjectId,
1249    caches: &mut DocCaches,
1250) -> Vec<Glyph> {
1251    let mut out = Vec::new();
1252    // lopdf 0.44: get_page_content returns the assembled content-stream bytes
1253    // directly (an empty Vec when the page has none).
1254    let content_bytes = doc.get_page_content(page_id);
1255    let Ok(content) = lopdf::content::Content::decode(&content_bytes) else {
1256        return out;
1257    };
1258    if let Some(res) = page_res(doc, page_id) {
1259        run_content(
1260            doc,
1261            res,
1262            &content,
1263            Mat::ID,
1264            TextState::INIT,
1265            0,
1266            caches,
1267            &mut out,
1268        );
1269    }
1270    out
1271}
1272
1273/// Run a content stream's operators, emitting glyphs into `out`. Recurses into
1274/// Form XObjects on `Do` (bulk body text in heavy PDFs lives inside a form, not
1275/// the page content stream). `res` is the resources dict in scope (the page's,
1276/// or the form's own); `base_ctm` is the CTM at the point of invocation.
1277#[allow(clippy::too_many_arguments)]
1278fn run_content(
1279    doc: &Document,
1280    res: &Dictionary,
1281    content: &lopdf::content::Content,
1282    base_ctm: Mat,
1283    init: TextState,
1284    depth: u32,
1285    caches: &mut DocCaches,
1286    out: &mut Vec<Glyph>,
1287) {
1288    let fonts = fonts_from_res(doc, res, caches);
1289    let xobjects = res
1290        .get(b"XObject")
1291        .ok()
1292        .and_then(|o| deref(doc, o))
1293        .and_then(|o| o.as_dict().ok());
1294
1295    // Graphics + text state. `q`/`Q` save and restore the whole graphics state,
1296    // which includes the text parameters (Tc, Tw, Tz, TL, Tfs, Trise, font) —
1297    // *not* the text matrix (that is reset by BT). Saving only the CTM let a Tc
1298    // set inside a `q…Q` block leak out and drift every later glyph.
1299    #[allow(clippy::type_complexity)]
1300    let mut gstate_stack: Vec<(Mat, f64, f64, f64, f64, f64, f64, Option<&Arc<Font>>)> = Vec::new();
1301    let mut ctm = base_ctm;
1302    let mut tm = Mat::ID;
1303    let mut tlm = Mat::ID;
1304    let mut font: Option<&Arc<Font>> = None;
1305    let mut fsize = init.fsize;
1306    let mut tc = init.tc; // char spacing
1307    let mut tw = init.tw; // word spacing
1308    let mut th = init.th; // horizontal scale (Tz/100)
1309    let mut tl = init.tl; // leading
1310    let mut trise = init.trise;
1311
1312    let op_f = |operands: &[Object], i: usize| operands.get(i).and_then(num).unwrap_or(0.0);
1313
1314    for op in &content.operations {
1315        let operands = &op.operands;
1316        match op.operator.as_str() {
1317            "q" => gstate_stack.push((ctm, tc, tw, th, tl, trise, fsize, font)),
1318            "Q" => {
1319                if let Some((c, a, b, h, l, r, fs, f)) = gstate_stack.pop() {
1320                    ctm = c;
1321                    tc = a;
1322                    tw = b;
1323                    th = h;
1324                    tl = l;
1325                    trise = r;
1326                    fsize = fs;
1327                    font = f;
1328                }
1329            }
1330            "cm" => {
1331                let m = Mat {
1332                    a: op_f(operands, 0),
1333                    b: op_f(operands, 1),
1334                    c: op_f(operands, 2),
1335                    d: op_f(operands, 3),
1336                    e: op_f(operands, 4),
1337                    f: op_f(operands, 5),
1338                };
1339                ctm = m.then(ctm);
1340            }
1341            "BT" => {
1342                tm = Mat::ID;
1343                tlm = Mat::ID;
1344            }
1345            "ET" => {}
1346            "Tf" => {
1347                if let Some(Object::Name(n)) = operands.first() {
1348                    font = fonts.get(n.as_slice());
1349                }
1350                fsize = op_f(operands, 1);
1351            }
1352            "Td" => {
1353                tlm = Mat {
1354                    a: 1.0,
1355                    b: 0.0,
1356                    c: 0.0,
1357                    d: 1.0,
1358                    e: op_f(operands, 0),
1359                    f: op_f(operands, 1),
1360                }
1361                .then(tlm);
1362                tm = tlm;
1363            }
1364            "TD" => {
1365                tl = -op_f(operands, 1);
1366                tlm = Mat {
1367                    a: 1.0,
1368                    b: 0.0,
1369                    c: 0.0,
1370                    d: 1.0,
1371                    e: op_f(operands, 0),
1372                    f: op_f(operands, 1),
1373                }
1374                .then(tlm);
1375                tm = tlm;
1376            }
1377            "Tm" => {
1378                tlm = Mat {
1379                    a: op_f(operands, 0),
1380                    b: op_f(operands, 1),
1381                    c: op_f(operands, 2),
1382                    d: op_f(operands, 3),
1383                    e: op_f(operands, 4),
1384                    f: op_f(operands, 5),
1385                };
1386                tm = tlm;
1387            }
1388            "T*" => {
1389                tlm = Mat {
1390                    a: 1.0,
1391                    b: 0.0,
1392                    c: 0.0,
1393                    d: 1.0,
1394                    e: 0.0,
1395                    f: -tl,
1396                }
1397                .then(tlm);
1398                tm = tlm;
1399            }
1400            "Tc" => tc = op_f(operands, 0),
1401            "Tw" => tw = op_f(operands, 0),
1402            "Tz" => th = op_f(operands, 0) / 100.0,
1403            "TL" => tl = op_f(operands, 0),
1404            "Ts" => trise = op_f(operands, 0),
1405            "Tj" | "'" | "\"" => {
1406                if op.operator == "'" || op.operator == "\"" {
1407                    // move to next line first
1408                    tlm = Mat {
1409                        a: 1.0,
1410                        b: 0.0,
1411                        c: 0.0,
1412                        d: 1.0,
1413                        e: 0.0,
1414                        f: -tl,
1415                    }
1416                    .then(tlm);
1417                    tm = tlm;
1418                }
1419                if op.operator == "\"" {
1420                    // `aw ac string "` sets word- and char-spacing before
1421                    // showing the string (PDF 32000-1 §9.4.3), persisting after.
1422                    tw = op_f(operands, 0);
1423                    tc = op_f(operands, 1);
1424                }
1425                if let (Some(f), Some(Object::String(s, _))) = (font, operands.last()) {
1426                    show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out);
1427                }
1428            }
1429            "TJ" => {
1430                if let (Some(f), Some(Object::Array(arr))) = (font, operands.first()) {
1431                    for el in arr {
1432                        match el {
1433                            Object::String(s, _) => {
1434                                show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out)
1435                            }
1436                            other => {
1437                                if let Some(adj) = num(other) {
1438                                    // negative number moves text right (PDF: subtract)
1439                                    let tx = -adj / 1000.0 * fsize * th;
1440                                    tm = Mat {
1441                                        a: 1.0,
1442                                        b: 0.0,
1443                                        c: 0.0,
1444                                        d: 1.0,
1445                                        e: tx,
1446                                        f: 0.0,
1447                                    }
1448                                    .then(tm);
1449                                }
1450                            }
1451                        }
1452                    }
1453                }
1454            }
1455            "Do" => {
1456                // Invoke a Form XObject: bulk body text in many PDFs lives inside
1457                // a form, reached only here. Image XObjects are skipped (no text).
1458                if depth >= 8 {
1459                    continue;
1460                }
1461                let Some(Object::Name(n)) = operands.first() else {
1462                    continue;
1463                };
1464                let obj = xobjects.and_then(|d| d.get(n.as_slice()).ok());
1465                let form_id = match obj {
1466                    Some(Object::Reference(id)) => Some(*id),
1467                    _ => None,
1468                };
1469                let stream = obj
1470                    .and_then(|o| deref(doc, o))
1471                    .and_then(|o| o.as_stream().ok());
1472                let Some(stream) = stream else { continue };
1473                let is_form = stream
1474                    .dict
1475                    .get(b"Subtype")
1476                    .ok()
1477                    .and_then(|o| o.as_name().ok())
1478                    == Some(b"Form".as_slice());
1479                if !is_form {
1480                    continue;
1481                }
1482                // Decode the form's content once per document (headers/footers
1483                // and bulk body text invoke the same form on every page).
1484                let cached = form_id.and_then(|id| caches.forms.get(&id).cloned());
1485                let form_content = match cached {
1486                    Some(c) => c,
1487                    None => {
1488                        let Ok(data) = stream.decompressed_content() else {
1489                            continue;
1490                        };
1491                        let Ok(c) = lopdf::content::Content::decode(&data) else {
1492                            continue;
1493                        };
1494                        let c = Arc::new(c);
1495                        if let Some(id) = form_id {
1496                            caches.forms.insert(id, Arc::clone(&c));
1497                        }
1498                        c
1499                    }
1500                };
1501                // The form's /Matrix maps form space into the CTM at invocation.
1502                let form_mat = match stream.dict.get(b"Matrix").ok() {
1503                    Some(Object::Array(a)) if a.len() == 6 => {
1504                        let v: Vec<f64> = a.iter().filter_map(num).collect();
1505                        if v.len() == 6 {
1506                            Mat {
1507                                a: v[0],
1508                                b: v[1],
1509                                c: v[2],
1510                                d: v[3],
1511                                e: v[4],
1512                                f: v[5],
1513                            }
1514                        } else {
1515                            Mat::ID
1516                        }
1517                    }
1518                    _ => Mat::ID,
1519                };
1520                // The form's own /Resources, falling back to the inherited ones.
1521                let form_res = stream
1522                    .dict
1523                    .get(b"Resources")
1524                    .ok()
1525                    .and_then(|o| deref(doc, o))
1526                    .and_then(|o| o.as_dict().ok())
1527                    .unwrap_or(res);
1528                let state = TextState {
1529                    tc,
1530                    tw,
1531                    th,
1532                    tl,
1533                    trise,
1534                    fsize,
1535                };
1536                run_content(
1537                    doc,
1538                    form_res,
1539                    &form_content,
1540                    form_mat.then(ctm),
1541                    state,
1542                    depth + 1,
1543                    caches,
1544                    out,
1545                );
1546            }
1547            _ => {}
1548        }
1549    }
1550}
1551
1552#[allow(clippy::too_many_arguments)]
1553fn show_text(
1554    font: &Font,
1555    bytes: &[u8],
1556    fsize: f64,
1557    tc: f64,
1558    tw: f64,
1559    th: f64,
1560    trise: f64,
1561    tm: &mut Mat,
1562    ctm: Mat,
1563    out: &mut Vec<Glyph>,
1564) {
1565    for code in codes(font, bytes) {
1566        let (text, w) = font.decode_code(code);
1567        let w0 = w / 1000.0; // advance in text-space (em) units
1568                             // The glyph→user transform: scale glyph space by font size, then Tm, CTM.
1569        let scale = Mat {
1570            a: fsize * th,
1571            b: 0.0,
1572            c: 0.0,
1573            d: fsize,
1574            e: 0.0,
1575            f: trise,
1576        };
1577        let trm = scale.then(*tm).then(ctm);
1578        // Box in glyph space (1000-unit em): x 0..w, y descent..ascent.
1579        let (x0, y0) = trm.apply(0.0, font.descent / 1000.0);
1580        let (x1, _y1) = trm.apply(w0, font.descent / 1000.0);
1581        let (_x2, y2) = trm.apply(0.0, font.ascent / 1000.0);
1582        let (left, right) = (x0.min(x1), x0.max(x1));
1583        let (bot, top) = (y0.min(y2), y0.max(y2));
1584        if let Some(s) = text {
1585            // A run may map one code to multiple chars (ligature/fraction); share box.
1586            for ch in s.chars() {
1587                if ch != '\u{0}' {
1588                    out.push(Glyph {
1589                        ch,
1590                        l: left as f32,
1591                        b: bot as f32,
1592                        r: right as f32,
1593                        t: top as f32,
1594                        ll: left as f32,
1595                        lb: bot as f32,
1596                        lr: right as f32,
1597                        lt: top as f32,
1598                        font: font.hash,
1599                    });
1600                }
1601            }
1602        }
1603        // Advance the text matrix. Word spacing applies to single-byte code 32.
1604        let is_space = !font.two_byte && code == 32;
1605        let tx = (w0 * fsize + tc + if is_space { tw } else { 0.0 }) * th;
1606        *tm = Mat {
1607            a: 1.0,
1608            b: 0.0,
1609            c: 0.0,
1610            d: 1.0,
1611            e: tx,
1612            f: 0.0,
1613        }
1614        .then(*tm);
1615    }
1616}
1617
1618/// Build a simple font's code→char table from its `/Encoding`: the base
1619/// encoding (WinAnsi / MacRoman) plus any `/Differences` overrides (glyph names
1620/// resolved through a small Adobe-glyph-name subset).
1621fn simple_encoding_table(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
1622    let enc = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o));
1623    let base_name = match enc {
1624        Some(Object::Name(n)) => n.clone(),
1625        Some(Object::Dictionary(d)) => d
1626            .get(b"BaseEncoding")
1627            .ok()
1628            .and_then(|o| o.as_name().ok())
1629            .map(|n| n.to_vec())
1630            .unwrap_or_default(),
1631        _ => Vec::new(),
1632    };
1633    let mut m = if base_name == b"MacRomanEncoding" {
1634        macroman_table()
1635    } else if base_name.is_empty() {
1636        // No PDF /Encoding at all: the font's *built-in* encoding applies. For
1637        // the standard TeX math fonts that is their fixed TeX layout — falling
1638        // back to StandardEncoding read CMSY's braces as `f`/`g`, `→` as `!`,
1639        // `∈` as `2` (2203's `{ahn,…}` author line). The font program (often
1640        // CFF, which this parser does not read) carries the same mapping;
1641        // docling-parse decodes it from there.
1642        tex_math_builtin(fdict).unwrap_or_else(winansi_table)
1643    } else {
1644        winansi_table()
1645    };
1646    // Apply /Differences: `code /glyphname /glyphname ... code ...`.
1647    if let Some(Object::Dictionary(d)) = enc {
1648        if let Some(Object::Array(diffs)) = d.get(b"Differences").ok().and_then(|o| deref(doc, o)) {
1649            let mut code = 0u8;
1650            for el in diffs {
1651                match el {
1652                    Object::Integer(i) => code = *i as u8,
1653                    Object::Name(name) => {
1654                        if let Some(ch) = glyph_name_to_char(name) {
1655                            m.insert(code, ch);
1656                        }
1657                        code = code.wrapping_add(1);
1658                    }
1659                    _ => {}
1660                }
1661            }
1662        }
1663    }
1664    m
1665}
1666
1667/// The fixed built-in encodings of the standard TeX math fonts (TeXbook
1668/// Appendix F), keyed off the base font name: `CMSY*` (symbols; `CMBSY` is its
1669/// bold) and `CMMI*` (math italic). These fonts ship no PDF `/Encoding` and no
1670/// ToUnicode, and their program is usually CFF — without this table the codes
1671/// fell through to StandardEncoding and rendered as the wrong ASCII.
1672fn tex_math_builtin(fdict: &Dictionary) -> Option<HashMap<u8, char>> {
1673    const CMSY: [char; 128] = [
1674        '−', '·', '×', '∗', '÷', '⋄', '±', '∓', '⊕', '⊖', '⊗', '⊘', '⊙', '◯', '∘', '•', '≍', '≡',
1675        '⊆', '⊇', '≤', '≥', '≼', '≽', '∼', '≈', '⊂', '⊃', '≪', '≫', '≺', '≻', '←', '→', '↑', '↓',
1676        '↔', '↗', '↘', '≃', '⇐', '⇒', '⇑', '⇓', '⇔', '↖', '↙', '∝', '′', '∞', '∈', '∋', '△', '▽',
1677        '\u{338}', '↦', '∀', '∃', '¬', '∅', 'ℜ', 'ℑ', '⊤', '⊥', 'ℵ', 'A', 'B', 'C', 'D', 'E', 'F',
1678        'G', 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X',
1679        'Y', 'Z', '∪', '∩', '⊎', '∧', '∨', '⊢', '⊣', '⌊', '⌋', '⌈', '⌉', '{', '}', '⟨', '⟩', '|',
1680        '∥', '↕', '⇕', '\\', '≀', '√', '∐', '∇', '∫', '⊔', '⊓', '⊑', '⊒', '§', '†', '‡', '¶', '♣',
1681        '♢', '♡', '♠',
1682    ];
1683    const CMMI: [char; 128] = [
1684        'Γ', 'Δ', 'Θ', 'Λ', 'Ξ', 'Π', 'Σ', 'Υ', 'Φ', 'Ψ', 'Ω', 'α', 'β', 'γ', 'δ', 'ε', 'ζ', 'η',
1685        'θ', 'ι', 'κ', 'λ', 'μ', 'ν', 'ξ', 'π', 'ρ', 'σ', 'τ', 'υ', 'φ', 'χ', 'ψ', 'ω', 'ϵ', 'ϑ',
1686        'ϖ', 'ϱ', 'ς', 'ϕ', '↼', '↽', '⇀', '⇁', '↩', '↪', '▷', '◁', '0', '1', '2', '3', '4', '5',
1687        '6', '7', '8', '9', '.', ',', '<', '/', '>', '⋆', '∂', 'A', 'B', 'C', 'D', 'E', 'F', 'G',
1688        'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y',
1689        'Z', '♭', '♮', '♯', '⌣', '⌢', 'ℓ', 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k',
1690        'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', 'ı', 'ȷ', '℘',
1691        '\u{20d7}', '⁀',
1692    ];
1693    let name = base_font_name(fdict)?;
1694    let up = name.to_ascii_uppercase();
1695    let table: &[char; 128] = if up.starts_with(b"CMSY") || up.starts_with(b"CMBSY") {
1696        &CMSY
1697    } else if up.starts_with(b"CMMI") {
1698        &CMMI
1699    } else {
1700        return None;
1701    };
1702    Some(
1703        table
1704            .iter()
1705            .enumerate()
1706            .map(|(i, &c)| (i as u8, c))
1707            .collect(),
1708    )
1709}
1710
1711/// Resolve an Adobe glyph name to Unicode: `uniXXXX`, single ASCII letters, the
1712/// digit/punctuation names from the Adobe Glyph List, and common typographic
1713/// names. A `.suffix` (`one.taboldstyle`, `a.sc`) is stripped and the base name
1714/// retried — docling renders these as the base character.
1715fn glyph_name_to_char(name: &[u8]) -> Option<char> {
1716    let s = std::str::from_utf8(name).ok()?;
1717    if let Some(hex) = s.strip_prefix("uni") {
1718        if let Ok(cp) = u32::from_str_radix(hex.get(0..4)?, 16) {
1719            return char::from_u32(cp);
1720        }
1721    }
1722    // Single ASCII letter names (`A`, `m`) map to themselves.
1723    if s.len() == 1 {
1724        let b = s.as_bytes()[0];
1725        if b.is_ascii_alphabetic() {
1726            return Some(b as char);
1727        }
1728    }
1729    let resolved = match s {
1730        "space" => ' ',
1731        "exclam" => '!',
1732        "quotedbl" => '"',
1733        "numbersign" => '#',
1734        "dollar" => '$',
1735        "percent" => '%',
1736        "ampersand" => '&',
1737        "quotesingle" => '\'',
1738        "parenleft" => '(',
1739        "parenright" => ')',
1740        "asterisk" => '*',
1741        "plus" => '+',
1742        "comma" => ',',
1743        "hyphen" => '-',
1744        "period" => '.',
1745        "slash" => '/',
1746        "zero" => '0',
1747        "one" => '1',
1748        "two" => '2',
1749        "three" => '3',
1750        "four" => '4',
1751        "five" => '5',
1752        "six" => '6',
1753        "seven" => '7',
1754        "eight" => '8',
1755        "nine" => '9',
1756        "colon" => ':',
1757        "semicolon" => ';',
1758        "less" => '<',
1759        "equal" => '=',
1760        "greater" => '>',
1761        "question" => '?',
1762        "at" => '@',
1763        "bracketleft" => '[',
1764        "backslash" => '\\',
1765        "bracketright" => ']',
1766        "asciicircum" => '^',
1767        "underscore" => '_',
1768        "grave" => '`',
1769        "braceleft" => '{',
1770        "bar" => '|',
1771        "braceright" => '}',
1772        "asciitilde" => '~',
1773        "bullet" => '\u{2022}',
1774        "periodcentered" => '\u{00B7}',
1775        "endash" => '\u{2013}',
1776        "emdash" => '\u{2014}',
1777        "quoteright" => '\u{2019}',
1778        "quoteleft" => '\u{2018}',
1779        "quotedblleft" => '\u{201C}',
1780        "quotedblright" => '\u{201D}',
1781        "quotedblbase" => '\u{201E}',
1782        "quotesinglbase" => '\u{201A}',
1783        // Latin f-ligatures named in `/Differences` (e.g. 2305's body font). These
1784        // map to the presentation-form code points, which `decompose_ligatures`
1785        // then spells back out (`ff`→"ff") — without them the glyph decodes to
1786        // nothing and the sanitizer fills the gap with a space (`di erences`).
1787        "ff" => '\u{FB00}',
1788        "fi" => '\u{FB01}',
1789        "fl" => '\u{FB02}',
1790        "ffi" => '\u{FB03}',
1791        "ffl" => '\u{FB04}',
1792        "ft" => '\u{FB05}',
1793        "st" => '\u{FB06}',
1794        "degree" => '\u{00B0}',
1795        "trademark" => '\u{2122}',
1796        "registered" => '\u{00AE}',
1797        "copyright" => '\u{00A9}',
1798        "ellipsis" => '\u{2026}',
1799        "minus" => '\u{2212}',
1800        "fraction" => '\u{2044}',
1801        "nbspace" => '\u{00A0}',
1802        // Greek + math glyph names (standard Adobe Glyph List). Standard TeX math
1803        // fonts (CMMI/CMSY/…) name their glyphs this way in the embedded font
1804        // program's `/Encoding`; without these a `λ`/`≤` decodes to nothing and is
1805        // dropped from body text (`and λ set to 0.5` → `and set to 0.5`).
1806        "alpha" => '\u{03B1}',
1807        "beta" => '\u{03B2}',
1808        "gamma" => '\u{03B3}',
1809        "delta" => '\u{03B4}',
1810        "epsilon" | "epsilon1" => '\u{03B5}',
1811        "zeta" => '\u{03B6}',
1812        "eta" => '\u{03B7}',
1813        "theta" | "theta1" => '\u{03B8}',
1814        "iota" => '\u{03B9}',
1815        "kappa" => '\u{03BA}',
1816        "lambda" => '\u{03BB}',
1817        "mu" => '\u{03BC}',
1818        "nu" => '\u{03BD}',
1819        "xi" => '\u{03BE}',
1820        "omicron" => '\u{03BF}',
1821        "pi" | "pi1" => '\u{03C0}',
1822        "rho" | "rho1" => '\u{03C1}',
1823        "sigma" => '\u{03C3}',
1824        "sigma1" => '\u{03C2}',
1825        "tau" => '\u{03C4}',
1826        "upsilon" => '\u{03C5}',
1827        "phi" | "phi1" => '\u{03C6}',
1828        "chi" => '\u{03C7}',
1829        "psi" => '\u{03C8}',
1830        "omega" | "omega1" => '\u{03C9}',
1831        "Gamma" => '\u{0393}',
1832        "Delta" => '\u{0394}',
1833        "Theta" => '\u{0398}',
1834        "Lambda" => '\u{039B}',
1835        "Xi" => '\u{039E}',
1836        "Pi" => '\u{03A0}',
1837        "Sigma" => '\u{03A3}',
1838        "Upsilon" => '\u{03A5}',
1839        "Phi" => '\u{03A6}',
1840        "Psi" => '\u{03A8}',
1841        "Omega" => '\u{03A9}',
1842        "lessequal" => '\u{2264}',
1843        "greaterequal" => '\u{2265}',
1844        "notequal" => '\u{2260}',
1845        "approxequal" => '\u{2248}',
1846        "equivalence" => '\u{2261}',
1847        "element" => '\u{2208}',
1848        "plusminus" => '\u{00B1}',
1849        "multiply" => '\u{00D7}',
1850        "divide" => '\u{00F7}',
1851        "infinity" => '\u{221E}',
1852        "partialdiff" => '\u{2202}',
1853        "gradient" => '\u{2207}',
1854        "summation" => '\u{2211}',
1855        "product" => '\u{220F}',
1856        "integral" => '\u{222B}',
1857        "radical" => '\u{221A}',
1858        "proportional" => '\u{221D}',
1859        "arrowright" => '\u{2192}',
1860        "arrowleft" => '\u{2190}',
1861        "arrowup" => '\u{2191}',
1862        "arrowdown" => '\u{2193}',
1863        "arrowboth" => '\u{2194}',
1864        "arrowdblright" => '\u{21D2}',
1865        "logicaland" => '\u{2227}',
1866        "logicalor" => '\u{2228}',
1867        "intersection" => '\u{2229}',
1868        "union" => '\u{222A}',
1869        "similar" => '\u{223C}',
1870        "congruent" => '\u{2245}',
1871        "dotmath" => '\u{22C5}',
1872        "asteriskmath" => '\u{2217}',
1873        _ => {
1874            // Strip an AGL `.suffix` (oldstyle/small-cap variant) and retry.
1875            if let Some((base, _)) = s.split_once('.') {
1876                if !base.is_empty() {
1877                    return glyph_name_to_char(base.as_bytes());
1878                }
1879            }
1880            return None;
1881        }
1882    };
1883    Some(resolved)
1884}
1885
1886/// Minimal WinAnsiEncoding (Latin-1-ish) for simple fonts lacking ToUnicode.
1887fn winansi_table() -> HashMap<u8, char> {
1888    let mut m = HashMap::new();
1889    for b in 0x20u8..=0x7e {
1890        m.insert(b, b as char);
1891    }
1892    // High range: Windows-1252 printable points that differ from Latin-1.
1893    let extra: &[(u8, char)] = &[
1894        (0x91, '\u{2018}'),
1895        (0x92, '\u{2019}'),
1896        (0x93, '\u{201C}'),
1897        (0x94, '\u{201D}'),
1898        (0x95, '\u{2022}'),
1899        (0x96, '\u{2013}'),
1900        (0x97, '\u{2014}'),
1901        (0x85, '\u{2026}'),
1902        (0xA0, '\u{00A0}'),
1903    ];
1904    for &(b, c) in extra {
1905        m.insert(b, c);
1906    }
1907    for b in 0xA1u8..=0xFF {
1908        m.entry(b).or_insert(b as char);
1909    }
1910    m
1911}
1912
1913/// Minimal MacRomanEncoding: ASCII plus the high-range points our corpus hits
1914/// (notably 0xA5 = bullet, used as a list marker).
1915fn macroman_table() -> HashMap<u8, char> {
1916    let mut m = HashMap::new();
1917    for b in 0x20u8..=0x7e {
1918        m.insert(b, b as char);
1919    }
1920    let high: &[(u8, char)] = &[
1921        (0xA5, '\u{2022}'), // bullet
1922        (0xD0, '\u{2013}'), // endash
1923        (0xD1, '\u{2014}'), // emdash
1924        (0xD2, '\u{201C}'),
1925        (0xD3, '\u{201D}'),
1926        (0xD4, '\u{2018}'),
1927        (0xD5, '\u{2019}'),
1928        (0xCA, '\u{00A0}'),
1929        (0xC9, '\u{2026}'),
1930        (0xDE, '\u{FB01}'),
1931        (0xDF, '\u{FB02}'),
1932    ];
1933    for &(b, c) in high {
1934        m.insert(b, c);
1935    }
1936    m
1937}
1938
1939#[cfg(test)]
1940mod xref_repair {
1941    /// Build a tiny one-page PDF whose cross-reference entries are either the
1942    /// spec's 20 bytes (`two_byte_eol`) or the 19-byte form some generators
1943    /// emit — everything else about the two files is identical.
1944    fn pdf_with_xref(two_byte_eol: bool) -> Vec<u8> {
1945        let content = b"BT /F1 12 Tf 72 700 Td (Invoice 922769430725) Tj ET\n";
1946        let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
1947        let objs: Vec<Vec<u8>> = vec![
1948            b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
1949            b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
1950            b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
1951               /Resources<</Font<</F1 5 0 R>>>>>>"
1952                .to_vec(),
1953            [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
1954            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".to_vec(),
1955        ];
1956
1957        let mut out = b"%PDF-1.4\n".to_vec();
1958        let mut offsets = Vec::new();
1959        for (i, body) in objs.iter().enumerate() {
1960            offsets.push(out.len());
1961            out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
1962            out.extend_from_slice(body);
1963            out.extend_from_slice(b"endobj\n");
1964        }
1965        let xref_at = out.len();
1966        let eol: &[u8] = if two_byte_eol { b" \n" } else { b"\n" };
1967        out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
1968        out.extend_from_slice(b"0000000000 65535 f");
1969        out.extend_from_slice(eol);
1970        for off in &offsets {
1971            out.extend_from_slice(format!("{off:010} 00000 n").as_bytes());
1972            out.extend_from_slice(eol);
1973        }
1974        out.extend_from_slice(
1975            format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
1976        );
1977        out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
1978        out
1979    }
1980
1981    /// A 19-byte cross-reference entry (a bare LF where the spec wants a
1982    /// two-byte EOL) makes lopdf reject the whole file, so a readable text
1983    /// layer used to look exactly like a scan — in the browser that meant ten
1984    /// seconds of OCR for nothing. The repair must recover *the same* parse the
1985    /// well-formed file gives.
1986    #[test]
1987    fn short_xref_entries_still_parse() {
1988        let good = pdf_with_xref(true);
1989        let broken = pdf_with_xref(false);
1990        assert!(
1991            broken.len() < good.len(),
1992            "the broken file is the shorter one"
1993        );
1994        assert!(
1995            lopdf::Document::load_mem(&good).is_ok(),
1996            "the control file must load unaided"
1997        );
1998        assert!(
1999            lopdf::Document::load_mem(&broken).is_err(),
2000            "lopdf rejects 19-byte entries — if this ever passes, drop the repair"
2001        );
2002
2003        let cells = |b: &[u8]| -> Vec<String> {
2004            super::pdf_textlines(b)
2005                .into_iter()
2006                .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2007                .collect()
2008        };
2009        let from_good = cells(&good);
2010        assert!(
2011            from_good.iter().any(|t| t.contains("922769430725")),
2012            "control text: {from_good:?}"
2013        );
2014        assert_eq!(
2015            cells(&broken),
2016            from_good,
2017            "repair must match the good parse"
2018        );
2019    }
2020
2021    /// The same generator overstates `/Length`, so lopdf reads past the data,
2022    /// misses `endstream` and drops the stream — the object comes back as a
2023    /// bare dictionary and the page has no content at all. Trust `endstream`
2024    /// instead, and do it without moving a single byte.
2025    #[test]
2026    fn overstated_stream_length_still_yields_content() {
2027        let good = pdf_with_xref(true);
2028        // Inflate the content stream's /Length by one, exactly as the invoice
2029        // that prompted this does.
2030        let broken = {
2031            let at = good
2032                .windows(8)
2033                .position(|w| w == b"/Length ")
2034                .expect("a /Length")
2035                + 8;
2036            let digits = good[at..].iter().take_while(|c| c.is_ascii_digit()).count();
2037            let n: usize = std::str::from_utf8(&good[at..at + digits])
2038                .unwrap()
2039                .parse()
2040                .unwrap();
2041            let inflated = (n + 1).to_string();
2042            assert_eq!(inflated.len(), digits, "keep the digit count");
2043            let mut b = good.clone();
2044            b[at..at + digits].copy_from_slice(inflated.as_bytes());
2045            b
2046        };
2047        assert_eq!(broken.len(), good.len(), "the defect must not move bytes");
2048        // lopdf alone loses the stream: the page parses but carries no content.
2049        let raw = lopdf::Document::load_mem(&broken).expect("still loads");
2050        assert!(
2051            raw.get_pages()
2052                .into_values()
2053                .all(|p| raw.get_page_content(p).is_empty()),
2054            "lopdf should drop the stream — if it stops, drop this repair"
2055        );
2056        // Ours recovers the same text the well-formed file gives.
2057        let text = |b: &[u8]| -> Vec<String> {
2058            super::pdf_textlines(b)
2059                .into_iter()
2060                .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2061                .collect()
2062        };
2063        let expected = text(&good);
2064        assert!(!expected.is_empty(), "control must produce text");
2065        assert_eq!(text(&broken), expected);
2066    }
2067
2068    /// The repair only fires where padding cannot move an object: it declines a
2069    /// file whose xref precedes an object (an incremental update), rather than
2070    /// shifting every offset the table records.
2071    #[test]
2072    fn repair_declines_when_padding_would_move_objects() {
2073        let mut incremental = pdf_with_xref(false);
2074        incremental.extend_from_slice(b"6 0 obj<</Type/Whatever>>endobj\n");
2075        let declined = super::pad_short_xref_entries(&incremental).unwrap_err();
2076        assert!(
2077            declined.contains("object follows the xref"),
2078            "reason: {declined}"
2079        );
2080    }
2081}
2082
2083/// #187: standard-14 fonts referenced without an embedded program (and thus
2084/// usually without `/Widths` or a `/FontDescriptor`) must decode with the
2085/// built-in Adobe Core 14 metrics instead of collapsing every cell to zero
2086/// width — the failure mode where a valid text layer was silently dropped
2087/// while pdfium read the same file fine.
2088#[cfg(test)]
2089mod base14_fonts {
2090    /// A one-page PDF whose single `Tj` uses `fontdict` (no embedded program).
2091    fn pdf_with_font(fontdict: &[u8], text: &[u8]) -> Vec<u8> {
2092        let content = [b"BT /F1 12 Tf 72 700 Td (".as_slice(), text, b") Tj ET\n"].concat();
2093        let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2094        let objs: Vec<Vec<u8>> = vec![
2095            b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2096            b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2097            b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2098               /Resources<</Font<</F1 5 0 R>>>>>>"
2099                .to_vec(),
2100            [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
2101            fontdict.to_vec(),
2102        ];
2103        let mut out = b"%PDF-1.4\n".to_vec();
2104        let mut offsets = Vec::new();
2105        for (i, body) in objs.iter().enumerate() {
2106            offsets.push(out.len());
2107            out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2108            out.extend_from_slice(body);
2109            out.extend_from_slice(b"endobj\n");
2110        }
2111        let xref_at = out.len();
2112        out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2113        out.extend_from_slice(b"0000000000 65535 f \n");
2114        for off in &offsets {
2115            out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2116        }
2117        out.extend_from_slice(
2118            format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2119        );
2120        out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2121        out
2122    }
2123
2124    /// The parsed cells of the only page.
2125    fn cells(pdf: &[u8]) -> Vec<crate::pdfium_backend::TextCell> {
2126        super::pdf_textlines(pdf)
2127            .into_iter()
2128            .flat_map(|(_, _, c)| c)
2129            .collect()
2130    }
2131
2132    /// Every standard-14 alias/style decodes with real (positive-width) boxes.
2133    #[test]
2134    fn standard14_faces_get_builtin_widths() {
2135        for fontdict in [
2136            // ReportLab's default: base-14 Helvetica, WinAnsi, nothing else.
2137            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>".as_slice(),
2138            // No /Encoding at all (StandardEncoding-ish default).
2139            b"<</Type/Font/Subtype/Type1/BaseFont/Times-BoldItalic>>",
2140            // Substitution aliases + a subset prefix.
2141            b"<</Type/Font/Subtype/TrueType/BaseFont/Arial,Bold>>",
2142            b"<</Type/Font/Subtype/Type1/BaseFont/ABCDEF+Courier-Oblique>>",
2143        ] {
2144            let pdf = pdf_with_font(fontdict, b"Words have width now");
2145            let cs = cells(&pdf);
2146            let text: String = cs
2147                .iter()
2148                .map(|c| c.text.as_str())
2149                .collect::<Vec<_>>()
2150                .join(" ");
2151            assert!(
2152                text.contains("Words have width now"),
2153                "{}: text lost: {text:?}",
2154                String::from_utf8_lossy(fontdict)
2155            );
2156            assert!(
2157                cs.iter().all(|c| c.r > c.l),
2158                "{}: zero-width cells: {cs:?}",
2159                String::from_utf8_lossy(fontdict)
2160            );
2161        }
2162    }
2163
2164    /// An explicit `/Widths` array always wins over the built-in metrics, and a
2165    /// non-standard face without `/Widths` stays as before (no invented boxes).
2166    #[test]
2167    fn explicit_widths_win_and_unknown_faces_are_untouched() {
2168        // Helvetica with explicit 100/1000-em widths: the word's box must be
2169        // ~4×100 units at 12pt = 4.8pt wide — far narrower than the ~2.7×
2170        // wider built-in Helvetica advances would make it.
2171        let explicit = pdf_with_font(
2172            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/FirstChar 65\
2173               /Widths[100 100 100 100]/Encoding/WinAnsiEncoding>>",
2174            b"ABBA",
2175        );
2176        let builtin = pdf_with_font(
2177            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>",
2178            b"ABBA",
2179        );
2180        let w = |pdf: &[u8]| {
2181            let cs = cells(pdf);
2182            assert_eq!(cs.len(), 1, "one word cell: {cs:?}");
2183            cs[0].r - cs[0].l
2184        };
2185        let (we, wb) = (w(&explicit), w(&builtin));
2186        assert!(
2187            (we - 4.8).abs() < 0.1,
2188            "explicit widths must win: got {we}, want 4×100×12/1000"
2189        );
2190        assert!(
2191            wb > 2.0 * we,
2192            "built-in Helvetica is much wider: {wb} vs {we}"
2193        );
2194
2195        // An unknown face with no /Widths: still parses (text kept), but no
2196        // built-in table applies — the old zero-width behavior is preserved
2197        // rather than inventing Helvetica metrics for an arbitrary font.
2198        let unknown = pdf_with_font(
2199            b"<</Type/Font/Subtype/Type1/BaseFont/FancyCorp-Display>>",
2200            b"Mystery",
2201        );
2202        let cs = cells(&unknown);
2203        let text: String = cs.iter().map(|c| c.text.as_str()).collect();
2204        assert!(text.contains("Mystery"), "text still decodes: {cs:?}");
2205    }
2206}
2207
2208#[cfg(test)]
2209mod overpainted {
2210    use crate::pdfium_backend::TextCell;
2211
2212    fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
2213        TextCell {
2214            text: text.into(),
2215            l,
2216            t,
2217            r,
2218            b,
2219        }
2220    }
2221
2222    /// The reporting invoice's logo: a `"` painted inside a `==` on one band —
2223    /// artwork drawn with glyphs. Both cells go; the real text on the next
2224    /// band stays.
2225    #[test]
2226    fn stacked_logo_glyphs_are_dropped() {
2227        let mut cells = vec![
2228            cell("\"", 72.7, 21.5, 86.4, 31.5),
2229            cell("==", 59.4, 21.5, 99.6, 31.5),
2230            cell("Herr", 65.2, 151.3, 81.7, 161.3),
2231        ];
2232        super::drop_overpainted_cells(&mut cells);
2233        assert_eq!(cells.len(), 1, "cells: {cells:?}");
2234        assert_eq!(cells[0].text, "Herr");
2235    }
2236
2237    /// Adjacent words on a line touch but never contain each other — prose is
2238    /// untouched, and so is a same-text near-duplicate (double-drawn faux
2239    /// bold), which is not evidence of artwork.
2240    #[test]
2241    fn prose_and_double_draw_are_kept() {
2242        let mut cells = vec![
2243            cell("Telefon", 354.3, 133.2, 381.5, 143.2),
2244            cell("0676/2000", 387.3, 133.2, 428.7, 143.2),
2245            cell("Bold", 100.0, 50.0, 130.0, 60.0),
2246            cell("Bold", 100.3, 50.0, 130.3, 60.0),
2247        ];
2248        super::drop_overpainted_cells(&mut cells);
2249        assert_eq!(cells.len(), 4);
2250    }
2251}
2252
2253#[cfg(test)]
2254mod vestigial_layer {
2255    use crate::pdfium_backend::{PdfPage, TextCell};
2256
2257    fn page_with(texts: &[&str]) -> PdfPage {
2258        let cells = texts
2259            .iter()
2260            .enumerate()
2261            .map(|(i, t)| TextCell {
2262                text: t.to_string(),
2263                l: 10.0,
2264                t: 10.0 + 12.0 * i as f32,
2265                r: 90.0,
2266                b: 20.0 + 12.0 * i as f32,
2267            })
2268            .collect();
2269        PdfPage::from_cells(595.0, 842.0, 1.0, cells)
2270    }
2271
2272    /// The reported scanned form: three typed-in field values ("03", "05",
2273    /// "2025") over three image pages. That must read as *no usable layer*,
2274    /// so the browser routes the document to OCR instead of extracting
2275    /// thirteen characters and skipping the letter entirely.
2276    #[test]
2277    fn typed_in_form_fields_are_not_a_text_layer() {
2278        let pages = vec![
2279            page_with(&["03", "05", "2025"]),
2280            page_with(&[]),
2281            page_with(&[]),
2282        ];
2283        assert!(super::text_layer_is_vestigial(&pages));
2284        assert!(super::text_layer_is_vestigial(&[page_with(&[])]));
2285    }
2286
2287    /// A short but genuine digital document — one page, a few real lines —
2288    /// keeps the fast text path.
2289    #[test]
2290    fn sparse_but_real_documents_pass() {
2291        let one_pager = vec![page_with(&[
2292            "Confidential briefing",
2293            "Prepared for the board meeting",
2294            "Do not distribute",
2295        ])];
2296        assert!(!super::text_layer_is_vestigial(&one_pager));
2297    }
2298}