Skip to main content

docling_pdf/
textparse.rs

1//! Pure-Rust PDF text extraction (replacing pdfium's glyph layer).
2//!
3//! pdfium reports *rendered* glyph boxes, which diverge from docling's
4//! `docling-parse` C++ parser at exactly the points that drive conformance:
5//! generated spaces get a zero-width box, combining diacritics get a real-width
6//! box, and ligature/fraction glyphs land at different x. This module instead
7//! reconstructs each glyph's box from the **font's own advance widths** and the
8//! PDF text/graphics matrices — the same information docling-parse uses — so a
9//! space is as wide as the font says and a combining mark has zero advance.
10//!
11//! The output is the same [`Glyph`] stream pdfium produces (native PDF
12//! coordinates, y-up), fed straight into the existing docling-parse line
13//! sanitizer ([`crate::dp_lines`]). Only the digital text layer is handled here;
14//! pages without one still fall back to OCR upstream.
15
16use std::collections::HashMap;
17use std::rc::Rc;
18
19use lopdf::{Dictionary, Document, Object};
20
21use crate::pdfium_backend::Glyph;
22
23/// Per-document caches for the content-stream interpreter. Fonts are indirect
24/// objects shared by many pages, but were fully re-parsed — ToUnicode CMap
25/// decompression + tokenization, embedded Type1 program scan, width tables —
26/// for **every page and every Form XObject invocation**; decoded form content
27/// streams were likewise re-inflated on every `Do`. Cached per document,
28/// keyed by the referenced object id (fonts also by resource name, which
29/// feeds the docling-parse font hash). Inline (non-reference) dicts are rare
30/// and stay uncached.
31#[derive(Default)]
32struct DocCaches {
33    fonts: HashMap<(lopdf::ObjectId, Vec<u8>), Rc<Font>>,
34    forms: HashMap<lopdf::ObjectId, Rc<lopdf::content::Content>>,
35}
36
37/// A 2×3 affine matrix `[a b c d e f]`: maps `(x,y)` → `(a·x+c·y+e, b·x+d·y+f)`.
38#[derive(Clone, Copy)]
39struct Mat {
40    a: f64,
41    b: f64,
42    c: f64,
43    d: f64,
44    e: f64,
45    f: f64,
46}
47
48impl Mat {
49    const ID: Mat = Mat {
50        a: 1.0,
51        b: 0.0,
52        c: 0.0,
53        d: 1.0,
54        e: 0.0,
55        f: 0.0,
56    };
57
58    /// `self ∘ m`: the matrix that applies `self` first, then `m`.
59    fn then(self, m: Mat) -> Mat {
60        Mat {
61            a: self.a * m.a + self.b * m.c,
62            b: self.a * m.b + self.b * m.d,
63            c: self.c * m.a + self.d * m.c,
64            d: self.c * m.b + self.d * m.d,
65            e: self.e * m.a + self.f * m.c + m.e,
66            f: self.e * m.b + self.f * m.d + m.f,
67        }
68    }
69
70    fn apply(self, x: f64, y: f64) -> (f64, f64) {
71        (
72            self.a * x + self.c * y + self.e,
73            self.b * x + self.d * y + self.f,
74        )
75    }
76}
77
78/// A parsed font: how to turn raw string bytes into (unicode, advance) pairs.
79struct Font {
80    /// 2-byte codes (Type0 / Identity-H) vs 1-byte (simple fonts).
81    two_byte: bool,
82    /// code → Unicode string (from ToUnicode; may be multi-char, e.g. ligatures).
83    to_unicode: HashMap<u32, String>,
84    /// code → glyph advance, in 1000-unit glyph space.
85    widths: HashMap<u32, f64>,
86    default_width: f64,
87    /// 1-byte fallback decoding when ToUnicode lacks a code (WinAnsi-ish).
88    simple_encoding: Option<HashMap<u8, char>>,
89    /// code → raw `/Differences` glyph name, for GID-style names (`g115`) that
90    /// have no Unicode mapping. docling-parse emits these verbatim as `/g115`
91    /// (see the redp5110 bulleted list); matching it keeps no text skipped.
92    fallback_names: HashMap<u8, String>,
93    /// code → char from the embedded Type1 font program's own `/Encoding` vector,
94    /// used only as a last resort for glyphs the base encoding leaves unmapped
95    /// (standard TeX math fonts: `λ`, `≤`, …).
96    program_encoding: HashMap<u8, char>,
97    ascent: f64,
98    descent: f64,
99    hash: u64,
100}
101
102impl Font {
103    fn decode_code(&self, code: u32) -> (Option<String>, f64) {
104        let w = self
105            .widths
106            .get(&code)
107            .copied()
108            .unwrap_or(self.default_width);
109        if let Some(s) = self.to_unicode.get(&code) {
110            return (Some(decompose_ligatures(s)), w);
111        }
112        if !self.two_byte {
113            // A GID-style `/Differences` name (no Unicode) overrides the base
114            // encoding, matching docling's verbatim `/g115` fallback.
115            if let Some(name) = self.fallback_names.get(&(code as u8)) {
116                return (Some(format!("/{name}")), w);
117            }
118            if let Some(enc) = &self.simple_encoding {
119                if let Some(&ch) = enc.get(&(code as u8)) {
120                    return (Some(decompose_ligatures(&ch.to_string())), w);
121                }
122            }
123            // Last resort: the embedded Type1 font program's own `/Encoding`
124            // vector (`dup N /glyphname put`). Standard TeX math fonts (CMMI, CMSY,
125            // …) ship no PDF `/Encoding` and no ToUnicode, so a glyph like `λ`
126            // (CMMI code 21 → `/lambda`) or `≤` (CMSY code 20 → `/lessequal`) has
127            // no other mapping and would otherwise be silently dropped. docling
128            // recovers these from the same font program. This only fills codes the
129            // base encoding left unmapped, so it never changes an existing decode.
130            if let Some(&ch) = self.program_encoding.get(&(code as u8)) {
131                return (Some(decompose_ligatures(&ch.to_string())), w);
132            }
133        }
134        (None, w)
135    }
136}
137
138/// Spell out Latin presentation-form ligatures (`fi`→`fi`, `ffi`→`ffi`, …) the way
139/// docling does, so `configuration`/`difficult` don't keep the ligature glyph.
140/// The chars share the ligature's box, so the line sanitizer recomposes them.
141fn decompose_ligatures(s: &str) -> String {
142    if !s.chars().any(|c| ('\u{FB00}'..='\u{FB06}').contains(&c)) {
143        return s.to_string();
144    }
145    s.chars()
146        .map(|c| {
147            match c {
148                '\u{FB00}' => "ff",
149                '\u{FB01}' => "fi",
150                '\u{FB02}' => "fl",
151                '\u{FB03}' => "ffi",
152                '\u{FB04}' => "ffl",
153                '\u{FB05}' => "ft",
154                '\u{FB06}' => "st",
155                _ => return c.to_string(),
156            }
157            .to_string()
158        })
159        .collect()
160}
161
162fn hash_name(name: &[u8]) -> u64 {
163    use std::hash::{Hash, Hasher};
164    let mut h = std::collections::hash_map::DefaultHasher::new();
165    name.hash(&mut h);
166    h.finish()
167}
168
169/// Resolve a possibly-indirect object to a dictionary.
170fn as_dict<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Dictionary> {
171    match obj {
172        Object::Dictionary(d) => Some(d),
173        Object::Reference(id) => doc.get_object(*id).ok().and_then(|o| o.as_dict().ok()),
174        _ => None,
175    }
176}
177
178fn deref<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Object> {
179    match obj {
180        Object::Reference(id) => doc.get_object(*id).ok(),
181        other => Some(other),
182    }
183}
184
185/// Parse one font dictionary into a [`Font`].
186fn parse_font(doc: &Document, name: &[u8], fdict: &Dictionary) -> Font {
187    let subtype: &[u8] = fdict
188        .get(b"Subtype")
189        .ok()
190        .and_then(|o| o.as_name().ok())
191        .unwrap_or(&[]);
192    let two_byte = subtype == b"Type0".as_slice();
193
194    let to_unicode = fdict
195        .get(b"ToUnicode")
196        .ok()
197        .and_then(|o| deref(doc, o))
198        .and_then(|o| o.as_stream().ok())
199        .and_then(|s| s.decompressed_content().ok())
200        .map(|data| parse_tounicode(&data))
201        .unwrap_or_default();
202
203    let (mut widths, mut default_width) = if two_byte {
204        cid_widths(doc, fdict)
205    } else {
206        simple_widths(doc, fdict)
207    };
208
209    let simple_encoding = if two_byte {
210        None
211    } else {
212        Some(simple_encoding_table(doc, fdict))
213    };
214
215    // A standard-14 font referenced without an embedded program usually ships
216    // no `/Widths` and no `/FontDescriptor` either (ReportLab's default, #187).
217    // Every advance then resolved to 0, the cells collapsed to zero width, and
218    // the page's whole text layer was silently dropped — while pdfium, with
219    // its built-in metrics, reads the same file fine. Fill the widths from the
220    // built-in Adobe Core 14 AFM tables via the font's own code→char decode
221    // (base encoding + `/Differences`), so an explicit `/Widths` always wins.
222    if !two_byte && widths.is_empty() && default_width == 0.0 {
223        if let Some(std14) = base_font_name(fdict).and_then(|n| crate::std14::widths_for(&n)) {
224            if let Some(enc) = &simple_encoding {
225                for (&code, &ch) in enc {
226                    if let Some(w) = std14.width(ch) {
227                        widths.insert(u32::from(code), w);
228                    }
229                }
230            }
231            // Codes the table misses still advance a typical width instead of
232            // stacking at x=0 (the failure mode this whole branch fixes).
233            default_width = 500.0;
234        }
235    }
236    let fallback_names = if two_byte {
237        HashMap::new()
238    } else {
239        differences_gid_names(doc, fdict)
240    };
241    let program_encoding = if two_byte {
242        HashMap::new()
243    } else {
244        type1_program_encoding(doc, fdict)
245    };
246
247    let (ascent, descent) = font_ascent_descent(doc, fdict, two_byte);
248
249    Font {
250        two_byte,
251        to_unicode,
252        widths,
253        default_width,
254        simple_encoding,
255        fallback_names,
256        program_encoding,
257        ascent,
258        descent,
259        hash: hash_name(name),
260    }
261}
262
263/// Collect `/Differences` entries whose glyph name is a GID placeholder
264/// (`g115`, `cid42`, `glyph7`, `index9`) with no Unicode mapping. docling-parse
265/// emits such glyphs as the literal name `/g115`; mapping them here keeps the
266/// text from being silently dropped (subsetted fonts with no ToUnicode). The
267/// GID-name restriction keeps real Adobe glyph names on the normal path so this
268/// never invents garbage on the clean files.
269fn differences_gid_names(doc: &Document, fdict: &Dictionary) -> HashMap<u8, String> {
270    let mut map = HashMap::new();
271    let Some(Object::Dictionary(enc)) = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o))
272    else {
273        return map;
274    };
275    let Some(Object::Array(diffs)) = enc.get(b"Differences").ok().and_then(|o| deref(doc, o))
276    else {
277        return map;
278    };
279    let mut code = 0u8;
280    for el in diffs {
281        match el {
282            Object::Integer(i) => code = *i as u8,
283            Object::Name(name) => {
284                if glyph_name_to_char(name).is_none() && is_gid_name(name) {
285                    map.insert(code, String::from_utf8_lossy(name).into_owned());
286                }
287                code = code.wrapping_add(1);
288            }
289            _ => {}
290        }
291    }
292    map
293}
294
295/// Parse the embedded Type1 font program's built-in `/Encoding` vector
296/// (`dup <code> /<glyphname> put` entries in the clear-text header before
297/// `eexec`) into `code → char`. This is how docling recovers glyphs from
298/// standard TeX math fonts (CMMI/CMSY/…) that carry no PDF `/Encoding` and no
299/// ToUnicode — e.g. CMMI's `dup 21 /lambda` or CMSY's `dup 20 /lessequal`.
300/// Only `FontFile` (Type1) is parsed; CFF (`FontFile3`) and TrueType
301/// (`FontFile2`) store their encoding in a binary table and are left alone.
302fn type1_program_encoding(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
303    let mut map = HashMap::new();
304    let Some(desc) = fdict
305        .get(b"FontDescriptor")
306        .ok()
307        .and_then(|o| deref(doc, o))
308        .and_then(|o| o.as_dict().ok())
309    else {
310        return map;
311    };
312    let Some(data) = desc
313        .get(b"FontFile")
314        .ok()
315        .and_then(|o| deref(doc, o))
316        .and_then(|o| o.as_stream().ok())
317        .and_then(|s| s.decompressed_content().ok())
318    else {
319        return map;
320    };
321    // The clear-text header (PostScript) ends at `eexec`; the rest is encrypted.
322    let head_end = data
323        .windows(5)
324        .position(|w| w == b"eexec")
325        .unwrap_or(data.len());
326    let head = String::from_utf8_lossy(&data[..head_end]);
327    // Scan for `dup <code> /<name> put` tokens.
328    let toks: Vec<&str> = head.split_whitespace().collect();
329    for w in toks.windows(4) {
330        if w[0] == "dup" && w[3] == "put" {
331            if let (Ok(code), Some(name)) = (w[1].parse::<u32>(), w[2].strip_prefix('/')) {
332                if code <= 255 {
333                    if let Some(ch) = glyph_name_to_char(name.as_bytes()) {
334                        map.insert(code as u8, ch);
335                    }
336                }
337            }
338        }
339    }
340    map
341}
342
343/// A glyph name that is a synthetic placeholder, not a real Adobe name:
344/// `g115`, `cid42`, `glyph7`, `index9`, `G12`, or a short-prefix code name like
345/// `SM590000` (IBM BookMaster). These carry no Unicode meaning, and docling-parse
346/// emits them verbatim (`/SM590000`). `afii####` / `uni####` are real Adobe names
347/// and excluded. The restriction keeps genuine glyph names on the Unicode path.
348fn is_gid_name(name: &[u8]) -> bool {
349    let Ok(s) = std::str::from_utf8(name) else {
350        return false;
351    };
352    if s.starts_with("afii") || s.starts_with("uni") {
353        return false;
354    }
355    for prefix in ["g", "G", "cid", "CID", "glyph", "index"] {
356        if let Some(rest) = s.strip_prefix(prefix) {
357            if !rest.is_empty() && rest.bytes().all(|b| b.is_ascii_digit()) {
358                return true;
359            }
360        }
361    }
362    // Short alpha prefix (≤3 letters) followed by a run of ≥3 digits — synthetic
363    // code names like `SM590000`, distinct from real Adobe names (whole words or
364    // letter+`.suffix` variants).
365    let alpha = s.bytes().take_while(|b| b.is_ascii_alphabetic()).count();
366    let digits = s.len() - alpha;
367    (1..=3).contains(&alpha)
368        && digits >= 3
369        && s.as_bytes()[alpha..].iter().all(|b| b.is_ascii_digit())
370}
371
372fn font_ascent_descent(doc: &Document, fdict: &Dictionary, two_byte: bool) -> (f64, f64) {
373    // For Type0, the descriptor lives on the descendant CIDFont.
374    let descr_owner = if two_byte {
375        fdict
376            .get(b"DescendantFonts")
377            .ok()
378            .and_then(|o| deref(doc, o))
379            .and_then(|o| match o {
380                Object::Array(a) => a.first(),
381                _ => None,
382            })
383            .and_then(|o| as_dict(doc, o))
384    } else {
385        Some(fdict)
386    };
387    let fd = descr_owner
388        .and_then(|d| d.get(b"FontDescriptor").ok())
389        .and_then(|o| as_dict(doc, o));
390    let asc = fd
391        .and_then(|d| d.get(b"Ascent").ok())
392        .and_then(|o| {
393            o.as_float()
394                .ok()
395                .or_else(|| o.as_i64().ok().map(|i| i as f32))
396        })
397        .unwrap_or(750.0) as f64;
398    let desc = fd
399        .and_then(|d| d.get(b"Descent").ok())
400        .and_then(|o| {
401            o.as_float()
402                .ok()
403                .or_else(|| o.as_i64().ok().map(|i| i as f32))
404        })
405        .unwrap_or(-250.0) as f64;
406    // Some subsetted fonts carry a degenerate FontDescriptor (`/Ascent 0
407    // /Descent 0`) — the real metrics live in the font program. That collapses
408    // the loose box to zero height, so the line cells get zero area and the
409    // layout's region/text assignment drops them (2305's References list lost
410    // every prose line, keeping only the URLs). Fall back to typical text metrics
411    // so the box has height.
412    if asc - desc <= 1.0 {
413        return (750.0, -250.0);
414    }
415    (asc, desc)
416}
417
418/// The `/BaseFont` name with any `ABCDEF+` subset prefix stripped (#187).
419fn base_font_name(fdict: &Dictionary) -> Option<Vec<u8>> {
420    let name = fdict.get(b"BaseFont").ok()?.as_name().ok()?;
421    let stripped = match name.iter().position(|&b| b == b'+') {
422        Some(i) if i == 6 => &name[i + 1..],
423        _ => name,
424    };
425    Some(stripped.to_vec())
426}
427
428/// Simple-font widths: `/FirstChar` + `/Widths` array, `/MissingWidth` default.
429fn simple_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
430    let mut map = HashMap::new();
431    let first = fdict
432        .get(b"FirstChar")
433        .ok()
434        .and_then(|o| o.as_i64().ok())
435        .unwrap_or(0) as u32;
436    if let Some(Object::Array(arr)) = fdict.get(b"Widths").ok().and_then(|o| deref(doc, o)) {
437        for (i, w) in arr.iter().enumerate() {
438            if let Some(w) = num(w) {
439                map.insert(first + i as u32, w);
440            }
441        }
442    }
443    let dw = fdict
444        .get(b"FontDescriptor")
445        .ok()
446        .and_then(|o| as_dict(doc, o))
447        .and_then(|d| d.get(b"MissingWidth").ok())
448        .and_then(num)
449        .unwrap_or(0.0);
450    (map, dw)
451}
452
453/// CIDFont widths: the `/W` array on the descendant font (`/DW` default = 1000).
454fn cid_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
455    let mut map = HashMap::new();
456    let Some(desc) = fdict
457        .get(b"DescendantFonts")
458        .ok()
459        .and_then(|o| deref(doc, o))
460        .and_then(|o| match o {
461            Object::Array(a) => a.first(),
462            _ => None,
463        })
464        .and_then(|o| as_dict(doc, o))
465    else {
466        return (map, 1000.0);
467    };
468    let dw = desc.get(b"DW").ok().and_then(num).unwrap_or(1000.0);
469    if let Some(Object::Array(w)) = desc.get(b"W").ok().and_then(|o| deref(doc, o)) {
470        let mut i = 0;
471        while i < w.len() {
472            let c = w.get(i).and_then(num);
473            match (c, w.get(i + 1)) {
474                // `c [w1 w2 ...]`: consecutive CIDs starting at c.
475                (Some(c), Some(Object::Array(list))) => {
476                    for (k, wv) in list.iter().enumerate() {
477                        if let Some(wv) = num(wv) {
478                            map.insert(c as u32 + k as u32, wv);
479                        }
480                    }
481                    i += 2;
482                }
483                // `c_first c_last w`: a run all of width w.
484                (Some(c1), Some(o2)) => {
485                    if let (Some(c2), Some(wv)) = (num(o2), w.get(i + 2).and_then(num)) {
486                        for cid in c1 as u32..=c2 as u32 {
487                            map.insert(cid, wv);
488                        }
489                    }
490                    i += 3;
491                }
492                _ => break,
493            }
494        }
495    }
496    (map, dw)
497}
498
499fn num(o: &Object) -> Option<f64> {
500    match o {
501        Object::Integer(i) => Some(*i as f64),
502        Object::Real(r) => Some(*r as f64),
503        _ => None,
504    }
505}
506
507/// Parse a ToUnicode CMap's `bfchar` / `bfrange` sections into code→string.
508fn parse_tounicode(data: &[u8]) -> HashMap<u32, String> {
509    let text = String::from_utf8_lossy(data);
510    let mut map = HashMap::new();
511    let hex = |s: &str| -> Option<Vec<u16>> {
512        let s = s.trim();
513        if !s.starts_with('<') || !s.ends_with('>') {
514            return None;
515        }
516        let h = &s[1..s.len() - 1];
517        let bytes: Vec<u8> = (0..h.len())
518            .step_by(2)
519            .filter_map(|i| u8::from_str_radix(h.get(i..i + 2)?, 16).ok())
520            .collect();
521        Some(
522            bytes
523                .chunks(2)
524                .map(|c| {
525                    if c.len() == 2 {
526                        u16::from_be_bytes([c[0], c[1]])
527                    } else {
528                        c[0] as u16
529                    }
530                })
531                .collect(),
532        )
533    };
534    let u16s_to_string = |u: &[u16]| String::from_utf16_lossy(u);
535    let code_of = |u: &[u16]| u.iter().fold(0u32, |acc, &x| (acc << 16) | x as u32);
536
537    // Tokenize by structure, not whitespace: CMap hex groups are often written
538    // back-to-back with no separators (`<21><21><0054>`), so scan for `<…>`
539    // groups, `[`/`]` brackets, and bareword keywords.
540    let tokens: Vec<String> = {
541        let bytes = text.as_bytes();
542        let mut toks = Vec::new();
543        let mut i = 0;
544        while i < bytes.len() {
545            let c = bytes[i];
546            if c.is_ascii_whitespace() {
547                i += 1;
548            } else if c == b'<' {
549                let start = i;
550                while i < bytes.len() && bytes[i] != b'>' {
551                    i += 1;
552                }
553                i += 1; // include '>'
554                toks.push(String::from_utf8_lossy(&bytes[start..i.min(bytes.len())]).into_owned());
555            } else if c == b'[' || c == b']' {
556                toks.push((c as char).to_string());
557                i += 1;
558            } else {
559                let start = i;
560                while i < bytes.len()
561                    && !bytes[i].is_ascii_whitespace()
562                    && bytes[i] != b'<'
563                    && bytes[i] != b'['
564                    && bytes[i] != b']'
565                {
566                    i += 1;
567                }
568                toks.push(String::from_utf8_lossy(&bytes[start..i]).into_owned());
569            }
570        }
571        toks
572    };
573    let tokens: Vec<&str> = tokens.iter().map(|s| s.as_str()).collect();
574    let mut i = 0;
575    while i < tokens.len() {
576        match tokens[i] {
577            "beginbfchar" => {
578                i += 1;
579                while i + 1 < tokens.len() && tokens[i] != "endbfchar" {
580                    if let (Some(src), Some(dst)) = (hex(tokens[i]), hex(tokens[i + 1])) {
581                        map.insert(code_of(&src), u16s_to_string(&dst));
582                    }
583                    i += 2;
584                }
585            }
586            "beginbfrange" => {
587                i += 1;
588                while i + 2 < tokens.len() && tokens[i] != "endbfrange" {
589                    let (Some(lo), Some(hi)) = (hex(tokens[i]), hex(tokens[i + 1])) else {
590                        i += 1;
591                        continue;
592                    };
593                    let lo = code_of(&lo);
594                    let hi = code_of(&hi);
595                    if tokens[i + 2] == "[" {
596                        // `<lo> <hi> [ <d0> <d1> ... ]`: one dst per code in the range.
597                        let mut j = i + 3;
598                        let mut code = lo;
599                        while j < tokens.len() && tokens[j] != "]" {
600                            if let Some(dst) = hex(tokens[j]) {
601                                map.insert(code, u16s_to_string(&dst));
602                            }
603                            code += 1;
604                            j += 1;
605                        }
606                        i = j + 1;
607                    } else if let Some(dst) = hex(tokens[i + 2]) {
608                        // `<lo> <hi> <dst>`: consecutive Unicode from a base.
609                        let base = code_of(&dst);
610                        for (k, code) in (lo..=hi).enumerate() {
611                            if let Some(ch) = char::from_u32(base + k as u32) {
612                                map.insert(code, ch.to_string());
613                            }
614                        }
615                        i += 3;
616                    } else {
617                        i += 1;
618                    }
619                }
620            }
621            _ => i += 1,
622        }
623    }
624    map
625}
626
627/// Decode a PDF string literal in a Tj/TJ operand into raw code units.
628fn codes(font: &Font, bytes: &[u8]) -> Vec<u32> {
629    if font.two_byte {
630        bytes
631            .chunks(2)
632            .map(|c| {
633                if c.len() == 2 {
634                    ((c[0] as u32) << 8) | c[1] as u32
635                } else {
636                    c[0] as u32
637                }
638            })
639            .collect()
640    } else {
641        bytes.iter().map(|&b| b as u32).collect()
642    }
643}
644
645/// Page size (width, height) in PDF points from the MediaBox.
646fn page_size(doc: &Document, page_id: lopdf::ObjectId) -> (f32, f32) {
647    let mb = doc
648        .get_object(page_id)
649        .ok()
650        .and_then(|o| o.as_dict().ok())
651        .and_then(|d| {
652            // MediaBox may be inherited; lopdf resolves via get_page... fall back to a guess.
653            d.get(b"MediaBox").ok().cloned()
654        })
655        .or_else(|| {
656            doc.get_dictionary(page_id)
657                .ok()
658                .and_then(|d| d.get(b"MediaBox").ok().cloned())
659        });
660    if let Some(Object::Array(a)) = mb {
661        let v: Vec<f32> = a.iter().filter_map(|o| num(o).map(|x| x as f32)).collect();
662        if v.len() == 4 {
663            return ((v[2] - v[0]).abs(), (v[3] - v[1]).abs());
664        }
665    }
666    (612.0, 792.0)
667}
668
669/// Localize where a page's text is lost, for the `text_layer` diagnostic.
670/// Extraction can come up empty at three different points — no content stream
671/// reached the parser, the stream did not decode into operators, or it ran but
672/// produced no glyphs (fonts/encodings) — and from the outside all three look
673/// the same. Report them per page.
674pub fn content_diagnosis(bytes: &[u8]) -> String {
675    let Some(doc) = load_document(bytes) else {
676        return "document does not load".into();
677    };
678    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
679    pages.sort_by_key(|(n, _)| *n);
680    let mut out = String::new();
681    let mut caches = DocCaches::default();
682    for (n, pid) in pages.into_iter().take(4) {
683        let content_bytes = doc.get_page_content(pid);
684        let ops = lopdf::content::Content::decode(&content_bytes)
685            .map(|c| c.operations.len())
686            .ok();
687        let res = page_res(&doc, pid);
688        let fonts = res.map(|r| fonts_from_res(&doc, r, &mut caches).len());
689        let glyphs = page_glyphs_cached(&doc, pid, &mut caches).len();
690        out.push_str(&format!(
691            "\n   page {n}: content {} B, ops {}, resources {}, fonts {}, glyphs {}",
692            content_bytes.len(),
693            ops.map_or("UNDECODABLE".to_string(), |n| n.to_string()),
694            if res.is_some() { "ok" } else { "MISSING" },
695            fonts.map_or("-".to_string(), |n| n.to_string()),
696            glyphs,
697        ));
698    }
699    out
700}
701
702/// Is this "text layer" a vestige rather than the document's text?
703///
704/// Scanned forms often carry a handful of typed-in strings — a date filled
705/// into three form fields, say — on top of pages that are otherwise images.
706/// Treating that as a real text layer is the worst of both worlds: the text
707/// path proudly extracts thirteen characters, and no OCR ever runs on the
708/// letter the pages actually show. The reported form did exactly this (3
709/// lines, 13 chars, 3 pages).
710///
711/// The rule is deliberately tight so genuinely sparse *digital* documents are
712/// not misrouted into OCR: only a document averaging at most one line per page
713/// **and** totalling fewer than 32 characters is called vestigial.
714pub fn text_layer_is_vestigial(pages: &[crate::pdfium_backend::PdfPage]) -> bool {
715    let lines: usize = pages.iter().map(|p| p.cells.len()).sum();
716    if lines == 0 {
717        return true;
718    }
719    let chars: usize = pages
720        .iter()
721        .flat_map(|p| &p.cells)
722        .map(|c| c.text.chars().count())
723        .sum();
724    lines <= pages.len() && chars < 32
725}
726
727/// Why the cross-reference repair did or did not fire, for the `text_layer`
728/// diagnostic. A PDF that will not load is indistinguishable from a scan in
729/// production (both convert to nothing), so the reason has to be askable.
730pub fn xref_repair_status(bytes: &[u8]) -> String {
731    if Document::load_mem(bytes).is_ok() {
732        return "loads unaided; no repair needed".into();
733    }
734    match pad_short_xref_entries(bytes) {
735        Ok(fixed) => match Document::load_mem(&fixed) {
736            Ok(_) => "repaired: cross-reference entries padded to 20 bytes".into(),
737            Err(e) => format!("padded the entries, but it still will not load: {e}"),
738        },
739        Err(why) => format!("repair declined — {why}"),
740    }
741}
742
743/// Load a PDF, repairing the one malformation that otherwise costs us the whole
744/// document: **19-byte cross-reference entries**.
745///
746/// The spec fixes an xref entry at 20 bytes — `nnnnnnnnnn ggggg n` plus a
747/// *two*-byte EOL. Some generators (an Austrian telecom's invoices, for one)
748/// emit a bare LF instead, making each entry 19 bytes. lopdf rejects the file
749/// outright (`invalid file trailer`) where pdfium reads it happily, so a
750/// perfectly good text layer looked to the browser exactly like a scan and cost
751/// ten seconds of OCR.
752///
753/// Padding is only attempted when it cannot move anything the xref points at:
754/// a single `xref` section that begins after the last object. The repair then
755/// has to prove itself — the padded bytes are used only if they load — so a
756/// mis-repair degrades to today's behaviour rather than to silent garbage.
757fn load_document(bytes: &[u8]) -> Option<Document> {
758    // Try progressively more repair, and accept a candidate only once the pages
759    // actually carry content — a document whose streams were dropped still
760    // "loads", so loading alone is not evidence the repair helped. A
761    // well-formed file returns on the first attempt and pays for nothing.
762    let mut fallback = None;
763    if let Some(doc) = best_effort_load(bytes, &mut fallback) {
764        return Some(doc);
765    }
766    let xref_fixed = pad_short_xref_entries(bytes).ok();
767    if let Some(fixed) = &xref_fixed {
768        if let Some(doc) = best_effort_load(fixed, &mut fallback) {
769            return Some(doc);
770        }
771    }
772    // Both defects can coexist, and the second only becomes visible once the
773    // first is repaired, so build on whatever the previous step produced.
774    let lengths_fixed = fix_stream_lengths(xref_fixed.as_deref().unwrap_or(bytes));
775    if let Some(doc) = best_effort_load(&lengths_fixed, &mut fallback) {
776        return Some(doc);
777    }
778    fallback
779}
780
781/// Load `data`, returning it only when its pages carry content; a document that
782/// merely parses is remembered as the fallback for when nothing does better.
783fn best_effort_load(data: &[u8], fallback: &mut Option<Document>) -> Option<Document> {
784    match Document::load_mem(data) {
785        Ok(doc) if has_page_content(&doc) => Some(doc),
786        Ok(doc) => {
787            fallback.get_or_insert(doc);
788            None
789        }
790        Err(_) => None,
791    }
792}
793
794/// Does any page actually hand us a content stream? A document whose streams
795/// were dropped still parses — it simply has nothing to read — so this is what
796/// tells a successful repair from a pointless one.
797fn has_page_content(doc: &Document) -> bool {
798    doc.get_pages()
799        .into_values()
800        .take(4)
801        .any(|pid| !doc.get_page_content(pid).is_empty())
802}
803
804/// Correct `/Length` values that disagree with where `endstream` actually is.
805///
806/// The same generator that writes short xref entries also overstates its
807/// content-stream lengths by a byte or two. lopdf trusts `/Length`, reads past
808/// the data, fails to find `endstream` there and drops the stream — the object
809/// comes back as a bare dictionary, so the page has no content at all and the
810/// document looks like a scan. pdfium instead trusts `endstream`, which is what
811/// this does.
812///
813/// The rewrite is length-preserving: the corrected number is written over the
814/// old digits and padded with spaces, so every byte offset in the file — and
815/// therefore the whole cross-reference table — stays valid.
816fn fix_stream_lengths(bytes: &[u8]) -> Vec<u8> {
817    let mut out = bytes.to_vec();
818    let mut i = 0;
819    while let Some(rel) = find(&out[i..], b"stream") {
820        let kw = i + rel;
821        i = kw + 6;
822        // Skip `endstream` (the keyword we are measuring *to*).
823        if kw >= 3 && &out[kw - 3..kw] == b"end" {
824            continue;
825        }
826        // The stream data starts after the EOL that follows the keyword.
827        let mut data = kw + 6;
828        if out.get(data..data + 2) == Some(b"\r\n".as_slice()) {
829            data += 2;
830        } else if matches!(out.get(data), Some(b'\n' | b'\r')) {
831            data += 1;
832        }
833        let Some(end) = find(&out[data..], b"endstream").map(|r| data + r) else {
834            continue;
835        };
836        // `/Length <digits>` in the dictionary just before the keyword.
837        let dict_start = out[..kw].iter().rposition(|&c| c == b'<').unwrap_or(0);
838        let Some(lrel) = find(&out[dict_start..kw], b"/Length") else {
839            continue;
840        };
841        let mut d = dict_start + lrel + 7;
842        while matches!(out.get(d), Some(b' ')) {
843            d += 1;
844        }
845        let digits = out[d..].iter().take_while(|c| c.is_ascii_digit()).count();
846        if digits == 0 {
847            continue;
848        }
849        let declared: usize = match std::str::from_utf8(&out[d..d + digits])
850            .ok()
851            .and_then(|s| s.parse().ok())
852        {
853            Some(v) => v,
854            None => continue,
855        };
856        let actual = end - data;
857        // Only shrink, and only when the new value fits the space the old one
858        // occupied — growing the number would move every following byte.
859        let replacement = actual.to_string();
860        if actual == declared || replacement.len() > digits {
861            continue;
862        }
863        out[d..d + digits].fill(b' ');
864        out[d..d + replacement.len()].copy_from_slice(replacement.as_bytes());
865    }
866    out
867}
868
869fn find(haystack: &[u8], needle: &[u8]) -> Option<usize> {
870    haystack.windows(needle.len()).position(|w| w == needle)
871}
872
873/// Rewrite a classic cross-reference table's entries to the spec's 20 bytes,
874/// or `None` when the file's shape makes that unsafe (see [`load_document`]).
875fn pad_short_xref_entries(bytes: &[u8]) -> Result<Vec<u8>, &'static str> {
876    // Exactly one xref section, and it must start after every object, so that
877    // growing it shifts nothing the table's offsets refer to.
878    let is_boundary = |i: usize| i == 0 || matches!(bytes[i - 1], b'\n' | b'\r');
879    let mut starts = (0..bytes.len().saturating_sub(4))
880        .filter(|&i| &bytes[i..i + 4] == b"xref" && is_boundary(i));
881    let xref_at = starts
882        .next()
883        .ok_or("no classic `xref` section (an xref stream?)")?;
884    if starts.next().is_some() {
885        return Err("more than one xref section (incremental update)");
886    }
887    let last_obj = bytes
888        .windows(3)
889        .rposition(|w| w == b"obj")
890        .ok_or("no objects found")?;
891    if last_obj > xref_at {
892        return Err("an object follows the xref — padding would move it");
893    }
894
895    let mut out = bytes[..xref_at].to_vec();
896    out.extend_from_slice(b"xref\n");
897    let mut i = xref_at + 4;
898    let skip_ws = |i: &mut usize| {
899        while matches!(bytes.get(*i), Some(b'\r' | b'\n' | b' ')) {
900            *i += 1;
901        }
902    };
903    loop {
904        skip_ws(&mut i);
905        // Either the next subsection header ("first count") or the trailer.
906        if bytes[i..].starts_with(b"trailer") {
907            out.extend_from_slice(&bytes[i..]);
908            return Ok(out);
909        }
910        let header_end = i + bytes[i..]
911            .iter()
912            .position(|c| matches!(c, b'\n' | b'\r'))
913            .ok_or("subsection header runs off the end")?;
914        let header = std::str::from_utf8(&bytes[i..header_end])
915            .map_err(|_| "subsection header is not text")?
916            .trim();
917        let mut parts = header.split_whitespace();
918        let count: usize = parts
919            .nth(1)
920            .and_then(|c| c.parse().ok())
921            .ok_or("unparseable subsection header")?;
922        if parts.next().is_some() || count == 0 {
923            return Err("unexpected subsection header shape");
924        }
925        out.extend_from_slice(header.as_bytes());
926        out.push(b'\n');
927        i = header_end;
928        for _ in 0..count {
929            skip_ws(&mut i);
930            // `nnnnnnnnnn ggggg n` — the 18 bytes before whatever EOL follows.
931            let entry = bytes.get(i..i + 18).ok_or("xref entry runs off the end")?;
932            let well_formed = entry[..10].iter().all(u8::is_ascii_digit)
933                && entry[10] == b' '
934                && entry[11..16].iter().all(u8::is_ascii_digit)
935                && entry[16] == b' '
936                && matches!(entry[17], b'n' | b'f');
937            if !well_formed {
938                return Err("xref entry is not `nnnnnnnnnn ggggg n`");
939            }
940            out.extend_from_slice(entry);
941            out.extend_from_slice(b" \n"); // the spec's 2-byte EOL -> 20 bytes
942            i += 18;
943        }
944    }
945}
946
947/// Debug: raw glyph stream `(ch, ll, lr, lb, lt)` (native coords) for page
948/// `index`, before the sanitizer. For comparing char cells to docling-parse.
949pub fn debug_glyphs(bytes: &[u8], index: usize) -> Vec<(char, f32, f32, f32, f32)> {
950    let Some(doc) = load_document(bytes) else {
951        return Vec::new();
952    };
953    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
954    pages.sort_by_key(|(n, _)| *n);
955    let Some((_, pid)) = pages.get(index) else {
956        return Vec::new();
957    };
958    page_glyphs(&doc, *pid)
959        .into_iter()
960        .map(|g| (g.ch, g.ll, g.lr, g.lb, g.lt))
961        .collect()
962}
963
964/// Public entry: per-page (width, height, line cells) for a PDF, via the Rust
965/// text parser + the docling-parse line sanitizer. Used by the pipeline and the
966/// `textparse_dump` example.
967pub fn pdf_textlines(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
968    let Some(doc) = load_document(bytes) else {
969        return Vec::new();
970    };
971    let mut caches = DocCaches::default();
972    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
973    pages.sort_by_key(|(n, _)| *n);
974    pages
975        .into_iter()
976        .map(|(_, pid)| {
977            let (w, h) = page_size(&doc, pid);
978            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
979            let cells = crate::dp_lines::line_cells(&glyphs, h, true);
980            (w, h, cells)
981        })
982        .collect()
983}
984
985/// Debug/diagnostic entry: per-page (width, height, word cells) for a PDF, via
986/// the Rust parser glyphs run through the docling-parse word grouping. Used to
987/// compare parser word cells against docling-parse's `word_cells` oracle (roadmap
988/// item 6).
989pub fn pdf_words(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
990    let Some(doc) = load_document(bytes) else {
991        return Vec::new();
992    };
993    let mut caches = DocCaches::default();
994    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
995    pages.sort_by_key(|(n, _)| *n);
996    pages
997        .into_iter()
998        .map(|(_, pid)| {
999            let (w, h) = page_size(&doc, pid);
1000            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1001            let cells = crate::dp_lines::word_cells(&glyphs, h, true);
1002            (w, h, cells)
1003        })
1004        .collect()
1005}
1006
1007/// One page's text cells from the pure-Rust parser: prose line cells, per-word
1008/// cells, and code line cells — all from a single glyph parse. Replaces the
1009/// pdfium text path (roadmap item 6) when the parser drop is enabled.
1010#[derive(Default)]
1011pub struct PageParserCells {
1012    pub prose: Vec<crate::pdfium_backend::TextCell>,
1013    pub words: Vec<crate::pdfium_backend::TextCell>,
1014    pub code: Vec<crate::pdfium_backend::TextCell>,
1015}
1016
1017/// Full parser text layer: prose + word + code cells per page, glyphs parsed once.
1018/// `prose`/`words` come from the docling-parse contraction ([`crate::dp_lines`]);
1019/// `code` splits only at the parser's own space glyphs (monospace keeps its
1020/// source spacing). Used by the pipeline to retire pdfium's text path.
1021pub fn pdf_all_cells(bytes: &[u8]) -> Vec<PageParserCells> {
1022    let Some(doc) = load_document(bytes) else {
1023        return Vec::new();
1024    };
1025    let mut caches = DocCaches::default();
1026    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1027    pages.sort_by_key(|(n, _)| *n);
1028    pages
1029        .into_iter()
1030        .map(|(_, pid)| {
1031            let (_w, h) = page_size(&doc, pid);
1032            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1033            let (prose, words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1034            PageParserCells {
1035                prose,
1036                words,
1037                code: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1038            }
1039        })
1040        .collect()
1041}
1042
1043/// Whole pages for the text-layer-only conversion ([`crate::convert_text_layer`]):
1044/// the parser's prose/word/code cells plus page geometry, assembled into
1045/// [`PdfPage`]s with no rendered image and no link annotations. Everything here
1046/// is pure Rust (lopdf), so it compiles without the `ml` feature — including on
1047/// wasm32. A page the parser can't read (no text layer) comes back with empty
1048/// cells; there is no pdfium fallback on this path.
1049pub fn pdf_text_pages(bytes: &[u8]) -> Vec<crate::pdfium_backend::PdfPage> {
1050    let Some(doc) = load_document(bytes) else {
1051        return Vec::new();
1052    };
1053    let mut caches = DocCaches::default();
1054    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1055    pages.sort_by_key(|(n, _)| *n);
1056    pages
1057        .into_iter()
1058        .map(|(_, pid)| {
1059            let (w, h) = page_size(&doc, pid);
1060            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1061            let (mut prose, mut words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1062            drop_overpainted_cells(&mut prose);
1063            drop_overpainted_cells(&mut words);
1064            crate::pdfium_backend::PdfPage {
1065                #[cfg(feature = "ocr-prep")]
1066                image_layout: None,
1067                width: w,
1068                height: h,
1069                // Cells are native PDF points; there is no rendered bitmap.
1070                scale: 1.0,
1071                cells: prose,
1072                code_cells: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1073                word_cells: words,
1074                #[cfg(feature = "ocr-prep")]
1075                image: image::RgbImage::new(1, 1),
1076                links: Vec::new(),
1077            }
1078        })
1079        .collect()
1080}
1081
1082/// Drop line cells that are *painted over each other* — glyphs used as artwork.
1083///
1084/// Some generators draw their logo with a symbol font: on the reporting
1085/// invoice, a `TeleLogo` Type1 paints the T-Mobile mark by stacking the glyphs
1086/// encoded as `"` and `==` on top of one another, and the flat text-layer
1087/// output opened with that garbage. Nothing in the font metadata gives it away
1088/// (the *text* fonts in the same file are also flagged symbolic, and the logo
1089/// font names its glyphs `quotedbl` &c.), but the geometry does: two cells with
1090/// different text where one lies inside the other on the same line is
1091/// physically impossible for prose — ink from two words never occupies the
1092/// same box. Both cells of such a pair are paint, not text.
1093///
1094/// Containment (not mere overlap) keeps this narrow: adjacent words touch but
1095/// never contain each other, and a same-text near-duplicate (double-draw faux
1096/// bold) is left alone for the sanitizer's usual handling. Applied on the
1097/// flat/browser path only — the ML pipeline's text layer is byte-pinned by the
1098/// PDF corpus, and there the layout model already sinks logo marks into
1099/// `picture` regions.
1100fn drop_overpainted_cells(cells: &mut Vec<crate::pdfium_backend::TextCell>) {
1101    let mut paint = vec![false; cells.len()];
1102    for i in 0..cells.len() {
1103        for j in 0..cells.len() {
1104            if i == j || cells[i].text == cells[j].text {
1105                continue;
1106            }
1107            let (a, b) = (&cells[i], &cells[j]);
1108            // Same line band: the vertical overlap covers most of the shorter.
1109            let vo = (a.b.min(b.b) - a.t.max(b.t)).max(0.0);
1110            if vo < 0.6 * (a.b - a.t).min(b.b - b.t) {
1111                continue;
1112            }
1113            // `a` horizontally inside `b` (with a small tolerance).
1114            let ho = (a.r.min(b.r) - a.l.max(b.l)).max(0.0);
1115            if ho >= 0.8 * (a.r - a.l) && (a.r - a.l) <= (b.r - b.l) {
1116                paint[i] = true;
1117                paint[j] = true;
1118            }
1119        }
1120    }
1121    let mut keep = paint.iter().map(|p| !p);
1122    cells.retain(|_| keep.next().unwrap());
1123}
1124
1125/// The text-state scalars inherited by a Form XObject when it is invoked via
1126/// `Do` (the PDF graphics state includes the text parameters, but not the text
1127/// matrices, which a form re-establishes inside its own `BT`/`ET`).
1128#[derive(Clone, Copy)]
1129struct TextState {
1130    tc: f64,
1131    tw: f64,
1132    th: f64,
1133    tl: f64,
1134    trise: f64,
1135    fsize: f64,
1136}
1137
1138impl TextState {
1139    const INIT: TextState = TextState {
1140        tc: 0.0,
1141        tw: 0.0,
1142        th: 1.0,
1143        tl: 0.0,
1144        trise: 0.0,
1145        fsize: 0.0,
1146    };
1147}
1148
1149/// The effective `/Resources` dictionary for a page (inline or via reference,
1150/// falling back to an inherited one from a `/Parent`).
1151fn page_res(doc: &Document, page_id: lopdf::ObjectId) -> Option<&Dictionary> {
1152    let (inline, ids) = doc.get_page_resources(page_id).ok()?;
1153    if let Some(d) = inline {
1154        return Some(d);
1155    }
1156    ids.into_iter().find_map(|id| doc.get_dictionary(id).ok())
1157}
1158
1159/// Build the code→[`Font`] map for a resources dictionary's `/Font` sub-dict,
1160/// reusing the per-document cache for fonts referenced indirectly (the common
1161/// case — the same font objects recur on every page).
1162fn fonts_from_res(
1163    doc: &Document,
1164    res: &Dictionary,
1165    caches: &mut DocCaches,
1166) -> HashMap<Vec<u8>, Rc<Font>> {
1167    let mut map = HashMap::new();
1168    let font_dict = res
1169        .get(b"Font")
1170        .ok()
1171        .and_then(|o| deref(doc, o))
1172        .and_then(|o| o.as_dict().ok());
1173    if let Some(fd) = font_dict {
1174        for (name, value) in fd.iter() {
1175            let font = match value {
1176                Object::Reference(id) => {
1177                    let key = (*id, name.clone());
1178                    if let Some(f) = caches.fonts.get(&key) {
1179                        Rc::clone(f)
1180                    } else if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1181                        let f = Rc::new(parse_font(doc, name, fdict));
1182                        caches.fonts.insert(key, Rc::clone(&f));
1183                        f
1184                    } else {
1185                        continue;
1186                    }
1187                }
1188                _ => {
1189                    if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1190                        Rc::new(parse_font(doc, name, fdict))
1191                    } else {
1192                        continue;
1193                    }
1194                }
1195            };
1196            map.insert(name.clone(), font);
1197        }
1198    }
1199    map
1200}
1201
1202/// Extract every glyph on a page as a native-coordinate [`Glyph`].
1203pub(crate) fn page_glyphs(doc: &Document, page_id: lopdf::ObjectId) -> Vec<Glyph> {
1204    page_glyphs_cached(doc, page_id, &mut DocCaches::default())
1205}
1206
1207/// [`page_glyphs`] with an explicit per-document cache, so a multi-page walk
1208/// parses each font / decodes each form once instead of once per page.
1209fn page_glyphs_cached(
1210    doc: &Document,
1211    page_id: lopdf::ObjectId,
1212    caches: &mut DocCaches,
1213) -> Vec<Glyph> {
1214    let mut out = Vec::new();
1215    // lopdf 0.44: get_page_content returns the assembled content-stream bytes
1216    // directly (an empty Vec when the page has none).
1217    let content_bytes = doc.get_page_content(page_id);
1218    let Ok(content) = lopdf::content::Content::decode(&content_bytes) else {
1219        return out;
1220    };
1221    if let Some(res) = page_res(doc, page_id) {
1222        run_content(
1223            doc,
1224            res,
1225            &content,
1226            Mat::ID,
1227            TextState::INIT,
1228            0,
1229            caches,
1230            &mut out,
1231        );
1232    }
1233    out
1234}
1235
1236/// Run a content stream's operators, emitting glyphs into `out`. Recurses into
1237/// Form XObjects on `Do` (bulk body text in heavy PDFs lives inside a form, not
1238/// the page content stream). `res` is the resources dict in scope (the page's,
1239/// or the form's own); `base_ctm` is the CTM at the point of invocation.
1240#[allow(clippy::too_many_arguments)]
1241fn run_content(
1242    doc: &Document,
1243    res: &Dictionary,
1244    content: &lopdf::content::Content,
1245    base_ctm: Mat,
1246    init: TextState,
1247    depth: u32,
1248    caches: &mut DocCaches,
1249    out: &mut Vec<Glyph>,
1250) {
1251    let fonts = fonts_from_res(doc, res, caches);
1252    let xobjects = res
1253        .get(b"XObject")
1254        .ok()
1255        .and_then(|o| deref(doc, o))
1256        .and_then(|o| o.as_dict().ok());
1257
1258    // Graphics + text state. `q`/`Q` save and restore the whole graphics state,
1259    // which includes the text parameters (Tc, Tw, Tz, TL, Tfs, Trise, font) —
1260    // *not* the text matrix (that is reset by BT). Saving only the CTM let a Tc
1261    // set inside a `q…Q` block leak out and drift every later glyph.
1262    #[allow(clippy::type_complexity)]
1263    let mut gstate_stack: Vec<(Mat, f64, f64, f64, f64, f64, f64, Option<&Rc<Font>>)> = Vec::new();
1264    let mut ctm = base_ctm;
1265    let mut tm = Mat::ID;
1266    let mut tlm = Mat::ID;
1267    let mut font: Option<&Rc<Font>> = None;
1268    let mut fsize = init.fsize;
1269    let mut tc = init.tc; // char spacing
1270    let mut tw = init.tw; // word spacing
1271    let mut th = init.th; // horizontal scale (Tz/100)
1272    let mut tl = init.tl; // leading
1273    let mut trise = init.trise;
1274
1275    let op_f = |operands: &[Object], i: usize| operands.get(i).and_then(num).unwrap_or(0.0);
1276
1277    for op in &content.operations {
1278        let operands = &op.operands;
1279        match op.operator.as_str() {
1280            "q" => gstate_stack.push((ctm, tc, tw, th, tl, trise, fsize, font)),
1281            "Q" => {
1282                if let Some((c, a, b, h, l, r, fs, f)) = gstate_stack.pop() {
1283                    ctm = c;
1284                    tc = a;
1285                    tw = b;
1286                    th = h;
1287                    tl = l;
1288                    trise = r;
1289                    fsize = fs;
1290                    font = f;
1291                }
1292            }
1293            "cm" => {
1294                let m = Mat {
1295                    a: op_f(operands, 0),
1296                    b: op_f(operands, 1),
1297                    c: op_f(operands, 2),
1298                    d: op_f(operands, 3),
1299                    e: op_f(operands, 4),
1300                    f: op_f(operands, 5),
1301                };
1302                ctm = m.then(ctm);
1303            }
1304            "BT" => {
1305                tm = Mat::ID;
1306                tlm = Mat::ID;
1307            }
1308            "ET" => {}
1309            "Tf" => {
1310                if let Some(Object::Name(n)) = operands.first() {
1311                    font = fonts.get(n.as_slice());
1312                }
1313                fsize = op_f(operands, 1);
1314            }
1315            "Td" => {
1316                tlm = Mat {
1317                    a: 1.0,
1318                    b: 0.0,
1319                    c: 0.0,
1320                    d: 1.0,
1321                    e: op_f(operands, 0),
1322                    f: op_f(operands, 1),
1323                }
1324                .then(tlm);
1325                tm = tlm;
1326            }
1327            "TD" => {
1328                tl = -op_f(operands, 1);
1329                tlm = Mat {
1330                    a: 1.0,
1331                    b: 0.0,
1332                    c: 0.0,
1333                    d: 1.0,
1334                    e: op_f(operands, 0),
1335                    f: op_f(operands, 1),
1336                }
1337                .then(tlm);
1338                tm = tlm;
1339            }
1340            "Tm" => {
1341                tlm = Mat {
1342                    a: op_f(operands, 0),
1343                    b: op_f(operands, 1),
1344                    c: op_f(operands, 2),
1345                    d: op_f(operands, 3),
1346                    e: op_f(operands, 4),
1347                    f: op_f(operands, 5),
1348                };
1349                tm = tlm;
1350            }
1351            "T*" => {
1352                tlm = Mat {
1353                    a: 1.0,
1354                    b: 0.0,
1355                    c: 0.0,
1356                    d: 1.0,
1357                    e: 0.0,
1358                    f: -tl,
1359                }
1360                .then(tlm);
1361                tm = tlm;
1362            }
1363            "Tc" => tc = op_f(operands, 0),
1364            "Tw" => tw = op_f(operands, 0),
1365            "Tz" => th = op_f(operands, 0) / 100.0,
1366            "TL" => tl = op_f(operands, 0),
1367            "Ts" => trise = op_f(operands, 0),
1368            "Tj" | "'" | "\"" => {
1369                if op.operator == "'" || op.operator == "\"" {
1370                    // move to next line first
1371                    tlm = Mat {
1372                        a: 1.0,
1373                        b: 0.0,
1374                        c: 0.0,
1375                        d: 1.0,
1376                        e: 0.0,
1377                        f: -tl,
1378                    }
1379                    .then(tlm);
1380                    tm = tlm;
1381                }
1382                if op.operator == "\"" {
1383                    // `aw ac string "` sets word- and char-spacing before
1384                    // showing the string (PDF 32000-1 §9.4.3), persisting after.
1385                    tw = op_f(operands, 0);
1386                    tc = op_f(operands, 1);
1387                }
1388                if let (Some(f), Some(Object::String(s, _))) = (font, operands.last()) {
1389                    show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out);
1390                }
1391            }
1392            "TJ" => {
1393                if let (Some(f), Some(Object::Array(arr))) = (font, operands.first()) {
1394                    for el in arr {
1395                        match el {
1396                            Object::String(s, _) => {
1397                                show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out)
1398                            }
1399                            other => {
1400                                if let Some(adj) = num(other) {
1401                                    // negative number moves text right (PDF: subtract)
1402                                    let tx = -adj / 1000.0 * fsize * th;
1403                                    tm = Mat {
1404                                        a: 1.0,
1405                                        b: 0.0,
1406                                        c: 0.0,
1407                                        d: 1.0,
1408                                        e: tx,
1409                                        f: 0.0,
1410                                    }
1411                                    .then(tm);
1412                                }
1413                            }
1414                        }
1415                    }
1416                }
1417            }
1418            "Do" => {
1419                // Invoke a Form XObject: bulk body text in many PDFs lives inside
1420                // a form, reached only here. Image XObjects are skipped (no text).
1421                if depth >= 8 {
1422                    continue;
1423                }
1424                let Some(Object::Name(n)) = operands.first() else {
1425                    continue;
1426                };
1427                let obj = xobjects.and_then(|d| d.get(n.as_slice()).ok());
1428                let form_id = match obj {
1429                    Some(Object::Reference(id)) => Some(*id),
1430                    _ => None,
1431                };
1432                let stream = obj
1433                    .and_then(|o| deref(doc, o))
1434                    .and_then(|o| o.as_stream().ok());
1435                let Some(stream) = stream else { continue };
1436                let is_form = stream
1437                    .dict
1438                    .get(b"Subtype")
1439                    .ok()
1440                    .and_then(|o| o.as_name().ok())
1441                    == Some(b"Form".as_slice());
1442                if !is_form {
1443                    continue;
1444                }
1445                // Decode the form's content once per document (headers/footers
1446                // and bulk body text invoke the same form on every page).
1447                let cached = form_id.and_then(|id| caches.forms.get(&id).cloned());
1448                let form_content = match cached {
1449                    Some(c) => c,
1450                    None => {
1451                        let Ok(data) = stream.decompressed_content() else {
1452                            continue;
1453                        };
1454                        let Ok(c) = lopdf::content::Content::decode(&data) else {
1455                            continue;
1456                        };
1457                        let c = Rc::new(c);
1458                        if let Some(id) = form_id {
1459                            caches.forms.insert(id, Rc::clone(&c));
1460                        }
1461                        c
1462                    }
1463                };
1464                // The form's /Matrix maps form space into the CTM at invocation.
1465                let form_mat = match stream.dict.get(b"Matrix").ok() {
1466                    Some(Object::Array(a)) if a.len() == 6 => {
1467                        let v: Vec<f64> = a.iter().filter_map(num).collect();
1468                        if v.len() == 6 {
1469                            Mat {
1470                                a: v[0],
1471                                b: v[1],
1472                                c: v[2],
1473                                d: v[3],
1474                                e: v[4],
1475                                f: v[5],
1476                            }
1477                        } else {
1478                            Mat::ID
1479                        }
1480                    }
1481                    _ => Mat::ID,
1482                };
1483                // The form's own /Resources, falling back to the inherited ones.
1484                let form_res = stream
1485                    .dict
1486                    .get(b"Resources")
1487                    .ok()
1488                    .and_then(|o| deref(doc, o))
1489                    .and_then(|o| o.as_dict().ok())
1490                    .unwrap_or(res);
1491                let state = TextState {
1492                    tc,
1493                    tw,
1494                    th,
1495                    tl,
1496                    trise,
1497                    fsize,
1498                };
1499                run_content(
1500                    doc,
1501                    form_res,
1502                    &form_content,
1503                    form_mat.then(ctm),
1504                    state,
1505                    depth + 1,
1506                    caches,
1507                    out,
1508                );
1509            }
1510            _ => {}
1511        }
1512    }
1513}
1514
1515#[allow(clippy::too_many_arguments)]
1516fn show_text(
1517    font: &Font,
1518    bytes: &[u8],
1519    fsize: f64,
1520    tc: f64,
1521    tw: f64,
1522    th: f64,
1523    trise: f64,
1524    tm: &mut Mat,
1525    ctm: Mat,
1526    out: &mut Vec<Glyph>,
1527) {
1528    for code in codes(font, bytes) {
1529        let (text, w) = font.decode_code(code);
1530        let w0 = w / 1000.0; // advance in text-space (em) units
1531                             // The glyph→user transform: scale glyph space by font size, then Tm, CTM.
1532        let scale = Mat {
1533            a: fsize * th,
1534            b: 0.0,
1535            c: 0.0,
1536            d: fsize,
1537            e: 0.0,
1538            f: trise,
1539        };
1540        let trm = scale.then(*tm).then(ctm);
1541        // Box in glyph space (1000-unit em): x 0..w, y descent..ascent.
1542        let (x0, y0) = trm.apply(0.0, font.descent / 1000.0);
1543        let (x1, _y1) = trm.apply(w0, font.descent / 1000.0);
1544        let (_x2, y2) = trm.apply(0.0, font.ascent / 1000.0);
1545        let (left, right) = (x0.min(x1), x0.max(x1));
1546        let (bot, top) = (y0.min(y2), y0.max(y2));
1547        if let Some(s) = text {
1548            // A run may map one code to multiple chars (ligature/fraction); share box.
1549            for ch in s.chars() {
1550                if ch != '\u{0}' {
1551                    out.push(Glyph {
1552                        ch,
1553                        l: left as f32,
1554                        b: bot as f32,
1555                        r: right as f32,
1556                        t: top as f32,
1557                        ll: left as f32,
1558                        lb: bot as f32,
1559                        lr: right as f32,
1560                        lt: top as f32,
1561                        font: font.hash,
1562                    });
1563                }
1564            }
1565        }
1566        // Advance the text matrix. Word spacing applies to single-byte code 32.
1567        let is_space = !font.two_byte && code == 32;
1568        let tx = (w0 * fsize + tc + if is_space { tw } else { 0.0 }) * th;
1569        *tm = Mat {
1570            a: 1.0,
1571            b: 0.0,
1572            c: 0.0,
1573            d: 1.0,
1574            e: tx,
1575            f: 0.0,
1576        }
1577        .then(*tm);
1578    }
1579}
1580
1581/// Build a simple font's code→char table from its `/Encoding`: the base
1582/// encoding (WinAnsi / MacRoman) plus any `/Differences` overrides (glyph names
1583/// resolved through a small Adobe-glyph-name subset).
1584fn simple_encoding_table(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
1585    let enc = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o));
1586    let base_name = match enc {
1587        Some(Object::Name(n)) => n.clone(),
1588        Some(Object::Dictionary(d)) => d
1589            .get(b"BaseEncoding")
1590            .ok()
1591            .and_then(|o| o.as_name().ok())
1592            .map(|n| n.to_vec())
1593            .unwrap_or_default(),
1594        _ => Vec::new(),
1595    };
1596    let mut m = if base_name == b"MacRomanEncoding" {
1597        macroman_table()
1598    } else {
1599        winansi_table()
1600    };
1601    // Apply /Differences: `code /glyphname /glyphname ... code ...`.
1602    if let Some(Object::Dictionary(d)) = enc {
1603        if let Some(Object::Array(diffs)) = d.get(b"Differences").ok().and_then(|o| deref(doc, o)) {
1604            let mut code = 0u8;
1605            for el in diffs {
1606                match el {
1607                    Object::Integer(i) => code = *i as u8,
1608                    Object::Name(name) => {
1609                        if let Some(ch) = glyph_name_to_char(name) {
1610                            m.insert(code, ch);
1611                        }
1612                        code = code.wrapping_add(1);
1613                    }
1614                    _ => {}
1615                }
1616            }
1617        }
1618    }
1619    m
1620}
1621
1622/// Resolve an Adobe glyph name to Unicode: `uniXXXX`, single ASCII letters, the
1623/// digit/punctuation names from the Adobe Glyph List, and common typographic
1624/// names. A `.suffix` (`one.taboldstyle`, `a.sc`) is stripped and the base name
1625/// retried — docling renders these as the base character.
1626fn glyph_name_to_char(name: &[u8]) -> Option<char> {
1627    let s = std::str::from_utf8(name).ok()?;
1628    if let Some(hex) = s.strip_prefix("uni") {
1629        if let Ok(cp) = u32::from_str_radix(hex.get(0..4)?, 16) {
1630            return char::from_u32(cp);
1631        }
1632    }
1633    // Single ASCII letter names (`A`, `m`) map to themselves.
1634    if s.len() == 1 {
1635        let b = s.as_bytes()[0];
1636        if b.is_ascii_alphabetic() {
1637            return Some(b as char);
1638        }
1639    }
1640    let resolved = match s {
1641        "space" => ' ',
1642        "exclam" => '!',
1643        "quotedbl" => '"',
1644        "numbersign" => '#',
1645        "dollar" => '$',
1646        "percent" => '%',
1647        "ampersand" => '&',
1648        "quotesingle" => '\'',
1649        "parenleft" => '(',
1650        "parenright" => ')',
1651        "asterisk" => '*',
1652        "plus" => '+',
1653        "comma" => ',',
1654        "hyphen" => '-',
1655        "period" => '.',
1656        "slash" => '/',
1657        "zero" => '0',
1658        "one" => '1',
1659        "two" => '2',
1660        "three" => '3',
1661        "four" => '4',
1662        "five" => '5',
1663        "six" => '6',
1664        "seven" => '7',
1665        "eight" => '8',
1666        "nine" => '9',
1667        "colon" => ':',
1668        "semicolon" => ';',
1669        "less" => '<',
1670        "equal" => '=',
1671        "greater" => '>',
1672        "question" => '?',
1673        "at" => '@',
1674        "bracketleft" => '[',
1675        "backslash" => '\\',
1676        "bracketright" => ']',
1677        "asciicircum" => '^',
1678        "underscore" => '_',
1679        "grave" => '`',
1680        "braceleft" => '{',
1681        "bar" => '|',
1682        "braceright" => '}',
1683        "asciitilde" => '~',
1684        "bullet" => '\u{2022}',
1685        "periodcentered" => '\u{00B7}',
1686        "endash" => '\u{2013}',
1687        "emdash" => '\u{2014}',
1688        "quoteright" => '\u{2019}',
1689        "quoteleft" => '\u{2018}',
1690        "quotedblleft" => '\u{201C}',
1691        "quotedblright" => '\u{201D}',
1692        "quotedblbase" => '\u{201E}',
1693        "quotesinglbase" => '\u{201A}',
1694        // Latin f-ligatures named in `/Differences` (e.g. 2305's body font). These
1695        // map to the presentation-form code points, which `decompose_ligatures`
1696        // then spells back out (`ff`→"ff") — without them the glyph decodes to
1697        // nothing and the sanitizer fills the gap with a space (`di erences`).
1698        "ff" => '\u{FB00}',
1699        "fi" => '\u{FB01}',
1700        "fl" => '\u{FB02}',
1701        "ffi" => '\u{FB03}',
1702        "ffl" => '\u{FB04}',
1703        "ft" => '\u{FB05}',
1704        "st" => '\u{FB06}',
1705        "degree" => '\u{00B0}',
1706        "trademark" => '\u{2122}',
1707        "registered" => '\u{00AE}',
1708        "copyright" => '\u{00A9}',
1709        "ellipsis" => '\u{2026}',
1710        "minus" => '\u{2212}',
1711        "fraction" => '\u{2044}',
1712        "nbspace" => '\u{00A0}',
1713        // Greek + math glyph names (standard Adobe Glyph List). Standard TeX math
1714        // fonts (CMMI/CMSY/…) name their glyphs this way in the embedded font
1715        // program's `/Encoding`; without these a `λ`/`≤` decodes to nothing and is
1716        // dropped from body text (`and λ set to 0.5` → `and set to 0.5`).
1717        "alpha" => '\u{03B1}',
1718        "beta" => '\u{03B2}',
1719        "gamma" => '\u{03B3}',
1720        "delta" => '\u{03B4}',
1721        "epsilon" | "epsilon1" => '\u{03B5}',
1722        "zeta" => '\u{03B6}',
1723        "eta" => '\u{03B7}',
1724        "theta" | "theta1" => '\u{03B8}',
1725        "iota" => '\u{03B9}',
1726        "kappa" => '\u{03BA}',
1727        "lambda" => '\u{03BB}',
1728        "mu" => '\u{03BC}',
1729        "nu" => '\u{03BD}',
1730        "xi" => '\u{03BE}',
1731        "omicron" => '\u{03BF}',
1732        "pi" | "pi1" => '\u{03C0}',
1733        "rho" | "rho1" => '\u{03C1}',
1734        "sigma" => '\u{03C3}',
1735        "sigma1" => '\u{03C2}',
1736        "tau" => '\u{03C4}',
1737        "upsilon" => '\u{03C5}',
1738        "phi" | "phi1" => '\u{03C6}',
1739        "chi" => '\u{03C7}',
1740        "psi" => '\u{03C8}',
1741        "omega" | "omega1" => '\u{03C9}',
1742        "Gamma" => '\u{0393}',
1743        "Delta" => '\u{0394}',
1744        "Theta" => '\u{0398}',
1745        "Lambda" => '\u{039B}',
1746        "Xi" => '\u{039E}',
1747        "Pi" => '\u{03A0}',
1748        "Sigma" => '\u{03A3}',
1749        "Upsilon" => '\u{03A5}',
1750        "Phi" => '\u{03A6}',
1751        "Psi" => '\u{03A8}',
1752        "Omega" => '\u{03A9}',
1753        "lessequal" => '\u{2264}',
1754        "greaterequal" => '\u{2265}',
1755        "notequal" => '\u{2260}',
1756        "approxequal" => '\u{2248}',
1757        "equivalence" => '\u{2261}',
1758        "element" => '\u{2208}',
1759        "plusminus" => '\u{00B1}',
1760        "multiply" => '\u{00D7}',
1761        "divide" => '\u{00F7}',
1762        "infinity" => '\u{221E}',
1763        "partialdiff" => '\u{2202}',
1764        "gradient" => '\u{2207}',
1765        "summation" => '\u{2211}',
1766        "product" => '\u{220F}',
1767        "integral" => '\u{222B}',
1768        "radical" => '\u{221A}',
1769        "proportional" => '\u{221D}',
1770        "arrowright" => '\u{2192}',
1771        "arrowleft" => '\u{2190}',
1772        "arrowup" => '\u{2191}',
1773        "arrowdown" => '\u{2193}',
1774        "arrowboth" => '\u{2194}',
1775        "arrowdblright" => '\u{21D2}',
1776        "logicaland" => '\u{2227}',
1777        "logicalor" => '\u{2228}',
1778        "intersection" => '\u{2229}',
1779        "union" => '\u{222A}',
1780        "similar" => '\u{223C}',
1781        "congruent" => '\u{2245}',
1782        "dotmath" => '\u{22C5}',
1783        "asteriskmath" => '\u{2217}',
1784        _ => {
1785            // Strip an AGL `.suffix` (oldstyle/small-cap variant) and retry.
1786            if let Some((base, _)) = s.split_once('.') {
1787                if !base.is_empty() {
1788                    return glyph_name_to_char(base.as_bytes());
1789                }
1790            }
1791            return None;
1792        }
1793    };
1794    Some(resolved)
1795}
1796
1797/// Minimal WinAnsiEncoding (Latin-1-ish) for simple fonts lacking ToUnicode.
1798fn winansi_table() -> HashMap<u8, char> {
1799    let mut m = HashMap::new();
1800    for b in 0x20u8..=0x7e {
1801        m.insert(b, b as char);
1802    }
1803    // High range: Windows-1252 printable points that differ from Latin-1.
1804    let extra: &[(u8, char)] = &[
1805        (0x91, '\u{2018}'),
1806        (0x92, '\u{2019}'),
1807        (0x93, '\u{201C}'),
1808        (0x94, '\u{201D}'),
1809        (0x95, '\u{2022}'),
1810        (0x96, '\u{2013}'),
1811        (0x97, '\u{2014}'),
1812        (0x85, '\u{2026}'),
1813        (0xA0, '\u{00A0}'),
1814    ];
1815    for &(b, c) in extra {
1816        m.insert(b, c);
1817    }
1818    for b in 0xA1u8..=0xFF {
1819        m.entry(b).or_insert(b as char);
1820    }
1821    m
1822}
1823
1824/// Minimal MacRomanEncoding: ASCII plus the high-range points our corpus hits
1825/// (notably 0xA5 = bullet, used as a list marker).
1826fn macroman_table() -> HashMap<u8, char> {
1827    let mut m = HashMap::new();
1828    for b in 0x20u8..=0x7e {
1829        m.insert(b, b as char);
1830    }
1831    let high: &[(u8, char)] = &[
1832        (0xA5, '\u{2022}'), // bullet
1833        (0xD0, '\u{2013}'), // endash
1834        (0xD1, '\u{2014}'), // emdash
1835        (0xD2, '\u{201C}'),
1836        (0xD3, '\u{201D}'),
1837        (0xD4, '\u{2018}'),
1838        (0xD5, '\u{2019}'),
1839        (0xCA, '\u{00A0}'),
1840        (0xC9, '\u{2026}'),
1841        (0xDE, '\u{FB01}'),
1842        (0xDF, '\u{FB02}'),
1843    ];
1844    for &(b, c) in high {
1845        m.insert(b, c);
1846    }
1847    m
1848}
1849
1850#[cfg(test)]
1851mod xref_repair {
1852    /// Build a tiny one-page PDF whose cross-reference entries are either the
1853    /// spec's 20 bytes (`two_byte_eol`) or the 19-byte form some generators
1854    /// emit — everything else about the two files is identical.
1855    fn pdf_with_xref(two_byte_eol: bool) -> Vec<u8> {
1856        let content = b"BT /F1 12 Tf 72 700 Td (Invoice 922769430725) Tj ET\n";
1857        let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
1858        let objs: Vec<Vec<u8>> = vec![
1859            b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
1860            b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
1861            b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
1862               /Resources<</Font<</F1 5 0 R>>>>>>"
1863                .to_vec(),
1864            [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
1865            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".to_vec(),
1866        ];
1867
1868        let mut out = b"%PDF-1.4\n".to_vec();
1869        let mut offsets = Vec::new();
1870        for (i, body) in objs.iter().enumerate() {
1871            offsets.push(out.len());
1872            out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
1873            out.extend_from_slice(body);
1874            out.extend_from_slice(b"endobj\n");
1875        }
1876        let xref_at = out.len();
1877        let eol: &[u8] = if two_byte_eol { b" \n" } else { b"\n" };
1878        out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
1879        out.extend_from_slice(b"0000000000 65535 f");
1880        out.extend_from_slice(eol);
1881        for off in &offsets {
1882            out.extend_from_slice(format!("{off:010} 00000 n").as_bytes());
1883            out.extend_from_slice(eol);
1884        }
1885        out.extend_from_slice(
1886            format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
1887        );
1888        out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
1889        out
1890    }
1891
1892    /// A 19-byte cross-reference entry (a bare LF where the spec wants a
1893    /// two-byte EOL) makes lopdf reject the whole file, so a readable text
1894    /// layer used to look exactly like a scan — in the browser that meant ten
1895    /// seconds of OCR for nothing. The repair must recover *the same* parse the
1896    /// well-formed file gives.
1897    #[test]
1898    fn short_xref_entries_still_parse() {
1899        let good = pdf_with_xref(true);
1900        let broken = pdf_with_xref(false);
1901        assert!(
1902            broken.len() < good.len(),
1903            "the broken file is the shorter one"
1904        );
1905        assert!(
1906            lopdf::Document::load_mem(&good).is_ok(),
1907            "the control file must load unaided"
1908        );
1909        assert!(
1910            lopdf::Document::load_mem(&broken).is_err(),
1911            "lopdf rejects 19-byte entries — if this ever passes, drop the repair"
1912        );
1913
1914        let cells = |b: &[u8]| -> Vec<String> {
1915            super::pdf_textlines(b)
1916                .into_iter()
1917                .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
1918                .collect()
1919        };
1920        let from_good = cells(&good);
1921        assert!(
1922            from_good.iter().any(|t| t.contains("922769430725")),
1923            "control text: {from_good:?}"
1924        );
1925        assert_eq!(
1926            cells(&broken),
1927            from_good,
1928            "repair must match the good parse"
1929        );
1930    }
1931
1932    /// The same generator overstates `/Length`, so lopdf reads past the data,
1933    /// misses `endstream` and drops the stream — the object comes back as a
1934    /// bare dictionary and the page has no content at all. Trust `endstream`
1935    /// instead, and do it without moving a single byte.
1936    #[test]
1937    fn overstated_stream_length_still_yields_content() {
1938        let good = pdf_with_xref(true);
1939        // Inflate the content stream's /Length by one, exactly as the invoice
1940        // that prompted this does.
1941        let broken = {
1942            let at = good
1943                .windows(8)
1944                .position(|w| w == b"/Length ")
1945                .expect("a /Length")
1946                + 8;
1947            let digits = good[at..].iter().take_while(|c| c.is_ascii_digit()).count();
1948            let n: usize = std::str::from_utf8(&good[at..at + digits])
1949                .unwrap()
1950                .parse()
1951                .unwrap();
1952            let inflated = (n + 1).to_string();
1953            assert_eq!(inflated.len(), digits, "keep the digit count");
1954            let mut b = good.clone();
1955            b[at..at + digits].copy_from_slice(inflated.as_bytes());
1956            b
1957        };
1958        assert_eq!(broken.len(), good.len(), "the defect must not move bytes");
1959        // lopdf alone loses the stream: the page parses but carries no content.
1960        let raw = lopdf::Document::load_mem(&broken).expect("still loads");
1961        assert!(
1962            raw.get_pages()
1963                .into_values()
1964                .all(|p| raw.get_page_content(p).is_empty()),
1965            "lopdf should drop the stream — if it stops, drop this repair"
1966        );
1967        // Ours recovers the same text the well-formed file gives.
1968        let text = |b: &[u8]| -> Vec<String> {
1969            super::pdf_textlines(b)
1970                .into_iter()
1971                .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
1972                .collect()
1973        };
1974        let expected = text(&good);
1975        assert!(!expected.is_empty(), "control must produce text");
1976        assert_eq!(text(&broken), expected);
1977    }
1978
1979    /// The repair only fires where padding cannot move an object: it declines a
1980    /// file whose xref precedes an object (an incremental update), rather than
1981    /// shifting every offset the table records.
1982    #[test]
1983    fn repair_declines_when_padding_would_move_objects() {
1984        let mut incremental = pdf_with_xref(false);
1985        incremental.extend_from_slice(b"6 0 obj<</Type/Whatever>>endobj\n");
1986        let declined = super::pad_short_xref_entries(&incremental).unwrap_err();
1987        assert!(
1988            declined.contains("object follows the xref"),
1989            "reason: {declined}"
1990        );
1991    }
1992}
1993
1994/// #187: standard-14 fonts referenced without an embedded program (and thus
1995/// usually without `/Widths` or a `/FontDescriptor`) must decode with the
1996/// built-in Adobe Core 14 metrics instead of collapsing every cell to zero
1997/// width — the failure mode where a valid text layer was silently dropped
1998/// while pdfium read the same file fine.
1999#[cfg(test)]
2000mod base14_fonts {
2001    /// A one-page PDF whose single `Tj` uses `fontdict` (no embedded program).
2002    fn pdf_with_font(fontdict: &[u8], text: &[u8]) -> Vec<u8> {
2003        let content = [b"BT /F1 12 Tf 72 700 Td (".as_slice(), text, b") Tj ET\n"].concat();
2004        let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2005        let objs: Vec<Vec<u8>> = vec![
2006            b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2007            b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2008            b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2009               /Resources<</Font<</F1 5 0 R>>>>>>"
2010                .to_vec(),
2011            [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
2012            fontdict.to_vec(),
2013        ];
2014        let mut out = b"%PDF-1.4\n".to_vec();
2015        let mut offsets = Vec::new();
2016        for (i, body) in objs.iter().enumerate() {
2017            offsets.push(out.len());
2018            out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2019            out.extend_from_slice(body);
2020            out.extend_from_slice(b"endobj\n");
2021        }
2022        let xref_at = out.len();
2023        out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2024        out.extend_from_slice(b"0000000000 65535 f \n");
2025        for off in &offsets {
2026            out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2027        }
2028        out.extend_from_slice(
2029            format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2030        );
2031        out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2032        out
2033    }
2034
2035    /// The parsed cells of the only page.
2036    fn cells(pdf: &[u8]) -> Vec<crate::pdfium_backend::TextCell> {
2037        super::pdf_textlines(pdf)
2038            .into_iter()
2039            .flat_map(|(_, _, c)| c)
2040            .collect()
2041    }
2042
2043    /// Every standard-14 alias/style decodes with real (positive-width) boxes.
2044    #[test]
2045    fn standard14_faces_get_builtin_widths() {
2046        for fontdict in [
2047            // ReportLab's default: base-14 Helvetica, WinAnsi, nothing else.
2048            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>".as_slice(),
2049            // No /Encoding at all (StandardEncoding-ish default).
2050            b"<</Type/Font/Subtype/Type1/BaseFont/Times-BoldItalic>>",
2051            // Substitution aliases + a subset prefix.
2052            b"<</Type/Font/Subtype/TrueType/BaseFont/Arial,Bold>>",
2053            b"<</Type/Font/Subtype/Type1/BaseFont/ABCDEF+Courier-Oblique>>",
2054        ] {
2055            let pdf = pdf_with_font(fontdict, b"Words have width now");
2056            let cs = cells(&pdf);
2057            let text: String = cs
2058                .iter()
2059                .map(|c| c.text.as_str())
2060                .collect::<Vec<_>>()
2061                .join(" ");
2062            assert!(
2063                text.contains("Words have width now"),
2064                "{}: text lost: {text:?}",
2065                String::from_utf8_lossy(fontdict)
2066            );
2067            assert!(
2068                cs.iter().all(|c| c.r > c.l),
2069                "{}: zero-width cells: {cs:?}",
2070                String::from_utf8_lossy(fontdict)
2071            );
2072        }
2073    }
2074
2075    /// An explicit `/Widths` array always wins over the built-in metrics, and a
2076    /// non-standard face without `/Widths` stays as before (no invented boxes).
2077    #[test]
2078    fn explicit_widths_win_and_unknown_faces_are_untouched() {
2079        // Helvetica with explicit 100/1000-em widths: the word's box must be
2080        // ~4×100 units at 12pt = 4.8pt wide — far narrower than the ~2.7×
2081        // wider built-in Helvetica advances would make it.
2082        let explicit = pdf_with_font(
2083            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/FirstChar 65\
2084               /Widths[100 100 100 100]/Encoding/WinAnsiEncoding>>",
2085            b"ABBA",
2086        );
2087        let builtin = pdf_with_font(
2088            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>",
2089            b"ABBA",
2090        );
2091        let w = |pdf: &[u8]| {
2092            let cs = cells(pdf);
2093            assert_eq!(cs.len(), 1, "one word cell: {cs:?}");
2094            cs[0].r - cs[0].l
2095        };
2096        let (we, wb) = (w(&explicit), w(&builtin));
2097        assert!(
2098            (we - 4.8).abs() < 0.1,
2099            "explicit widths must win: got {we}, want 4×100×12/1000"
2100        );
2101        assert!(
2102            wb > 2.0 * we,
2103            "built-in Helvetica is much wider: {wb} vs {we}"
2104        );
2105
2106        // An unknown face with no /Widths: still parses (text kept), but no
2107        // built-in table applies — the old zero-width behavior is preserved
2108        // rather than inventing Helvetica metrics for an arbitrary font.
2109        let unknown = pdf_with_font(
2110            b"<</Type/Font/Subtype/Type1/BaseFont/FancyCorp-Display>>",
2111            b"Mystery",
2112        );
2113        let cs = cells(&unknown);
2114        let text: String = cs.iter().map(|c| c.text.as_str()).collect();
2115        assert!(text.contains("Mystery"), "text still decodes: {cs:?}");
2116    }
2117}
2118
2119#[cfg(test)]
2120mod overpainted {
2121    use crate::pdfium_backend::TextCell;
2122
2123    fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
2124        TextCell {
2125            text: text.into(),
2126            l,
2127            t,
2128            r,
2129            b,
2130        }
2131    }
2132
2133    /// The reporting invoice's logo: a `"` painted inside a `==` on one band —
2134    /// artwork drawn with glyphs. Both cells go; the real text on the next
2135    /// band stays.
2136    #[test]
2137    fn stacked_logo_glyphs_are_dropped() {
2138        let mut cells = vec![
2139            cell("\"", 72.7, 21.5, 86.4, 31.5),
2140            cell("==", 59.4, 21.5, 99.6, 31.5),
2141            cell("Herr", 65.2, 151.3, 81.7, 161.3),
2142        ];
2143        super::drop_overpainted_cells(&mut cells);
2144        assert_eq!(cells.len(), 1, "cells: {cells:?}");
2145        assert_eq!(cells[0].text, "Herr");
2146    }
2147
2148    /// Adjacent words on a line touch but never contain each other — prose is
2149    /// untouched, and so is a same-text near-duplicate (double-drawn faux
2150    /// bold), which is not evidence of artwork.
2151    #[test]
2152    fn prose_and_double_draw_are_kept() {
2153        let mut cells = vec![
2154            cell("Telefon", 354.3, 133.2, 381.5, 143.2),
2155            cell("0676/2000", 387.3, 133.2, 428.7, 143.2),
2156            cell("Bold", 100.0, 50.0, 130.0, 60.0),
2157            cell("Bold", 100.3, 50.0, 130.3, 60.0),
2158        ];
2159        super::drop_overpainted_cells(&mut cells);
2160        assert_eq!(cells.len(), 4);
2161    }
2162}
2163
2164#[cfg(test)]
2165mod vestigial_layer {
2166    use crate::pdfium_backend::{PdfPage, TextCell};
2167
2168    fn page_with(texts: &[&str]) -> PdfPage {
2169        let cells = texts
2170            .iter()
2171            .enumerate()
2172            .map(|(i, t)| TextCell {
2173                text: t.to_string(),
2174                l: 10.0,
2175                t: 10.0 + 12.0 * i as f32,
2176                r: 90.0,
2177                b: 20.0 + 12.0 * i as f32,
2178            })
2179            .collect();
2180        PdfPage::from_cells(595.0, 842.0, 1.0, cells)
2181    }
2182
2183    /// The reported scanned form: three typed-in field values ("03", "05",
2184    /// "2025") over three image pages. That must read as *no usable layer*,
2185    /// so the browser routes the document to OCR instead of extracting
2186    /// thirteen characters and skipping the letter entirely.
2187    #[test]
2188    fn typed_in_form_fields_are_not_a_text_layer() {
2189        let pages = vec![
2190            page_with(&["03", "05", "2025"]),
2191            page_with(&[]),
2192            page_with(&[]),
2193        ];
2194        assert!(super::text_layer_is_vestigial(&pages));
2195        assert!(super::text_layer_is_vestigial(&[page_with(&[])]));
2196    }
2197
2198    /// A short but genuine digital document — one page, a few real lines —
2199    /// keeps the fast text path.
2200    #[test]
2201    fn sparse_but_real_documents_pass() {
2202        let one_pager = vec![page_with(&[
2203            "Confidential briefing",
2204            "Prepared for the board meeting",
2205            "Do not distribute",
2206        ])];
2207        assert!(!super::text_layer_is_vestigial(&one_pager));
2208    }
2209}