Skip to main content

docling_pdf/
textparse.rs

1//! Pure-Rust PDF text extraction (replacing pdfium's glyph layer).
2//!
3//! pdfium reports *rendered* glyph boxes, which diverge from docling's
4//! `docling-parse` C++ parser at exactly the points that drive conformance:
5//! generated spaces get a zero-width box, combining diacritics get a real-width
6//! box, and ligature/fraction glyphs land at different x. This module instead
7//! reconstructs each glyph's box from the **font's own advance widths** and the
8//! PDF text/graphics matrices — the same information docling-parse uses — so a
9//! space is as wide as the font says and a combining mark has zero advance.
10//!
11//! The output is the same [`Glyph`] stream pdfium produces (native PDF
12//! coordinates, y-up), fed straight into the existing docling-parse line
13//! sanitizer ([`crate::dp_lines`]). Only the digital text layer is handled here;
14//! pages without one still fall back to OCR upstream.
15
16use std::collections::HashMap;
17use std::sync::Arc;
18
19use lopdf::{Dictionary, Document, Object};
20
21use crate::pdfium_backend::Glyph;
22
23/// Per-document caches for the content-stream interpreter. Fonts are indirect
24/// objects shared by many pages, but were fully re-parsed — ToUnicode CMap
25/// decompression + tokenization, embedded Type1 program scan, width tables —
26/// for **every page and every Form XObject invocation**; decoded form content
27/// streams were likewise re-inflated on every `Do`. Cached per document,
28/// keyed by the referenced object id (fonts also by resource name, which
29/// feeds the docling-parse font hash). Inline (non-reference) dicts are rare
30/// and stay uncached.
31#[derive(Default)]
32struct DocCaches {
33    fonts: HashMap<(lopdf::ObjectId, Vec<u8>), Arc<Font>>,
34    forms: HashMap<lopdf::ObjectId, Arc<lopdf::content::Content>>,
35}
36
37/// A 2×3 affine matrix `[a b c d e f]`: maps `(x,y)` → `(a·x+c·y+e, b·x+d·y+f)`.
38#[derive(Clone, Copy)]
39struct Mat {
40    a: f64,
41    b: f64,
42    c: f64,
43    d: f64,
44    e: f64,
45    f: f64,
46}
47
48impl Mat {
49    const ID: Mat = Mat {
50        a: 1.0,
51        b: 0.0,
52        c: 0.0,
53        d: 1.0,
54        e: 0.0,
55        f: 0.0,
56    };
57
58    /// `self ∘ m`: the matrix that applies `self` first, then `m`.
59    fn then(self, m: Mat) -> Mat {
60        Mat {
61            a: self.a * m.a + self.b * m.c,
62            b: self.a * m.b + self.b * m.d,
63            c: self.c * m.a + self.d * m.c,
64            d: self.c * m.b + self.d * m.d,
65            e: self.e * m.a + self.f * m.c + m.e,
66            f: self.e * m.b + self.f * m.d + m.f,
67        }
68    }
69
70    fn apply(self, x: f64, y: f64) -> (f64, f64) {
71        (
72            self.a * x + self.c * y + self.e,
73            self.b * x + self.d * y + self.f,
74        )
75    }
76}
77
78/// A parsed font: how to turn raw string bytes into (unicode, advance) pairs.
79struct Font {
80    /// 2-byte codes (Type0 / Identity-H) vs 1-byte (simple fonts).
81    two_byte: bool,
82    /// code → Unicode string (from ToUnicode; may be multi-char, e.g. ligatures).
83    to_unicode: HashMap<u32, String>,
84    /// code → glyph advance, in 1000-unit glyph space.
85    widths: HashMap<u32, f64>,
86    default_width: f64,
87    /// 1-byte fallback decoding when ToUnicode lacks a code (WinAnsi-ish).
88    simple_encoding: Option<HashMap<u8, char>>,
89    /// code → raw `/Differences` glyph name, for GID-style names (`g115`) that
90    /// have no Unicode mapping. docling-parse emits these verbatim as `/g115`
91    /// (see the redp5110 bulleted list); matching it keeps no text skipped.
92    fallback_names: HashMap<u8, String>,
93    /// code → char from the embedded Type1 font program's own `/Encoding` vector,
94    /// used only as a last resort for glyphs the base encoding leaves unmapped
95    /// (standard TeX math fonts: `λ`, `≤`, …).
96    program_encoding: HashMap<u8, char>,
97    ascent: f64,
98    descent: f64,
99    hash: u64,
100    /// Weight and slant read from the `/BaseFont` name (the heading
101    /// hierarchy's style signal, #302).
102    style: crate::font_style::FontStyle,
103}
104
105impl Font {
106    fn decode_code(&self, code: u32) -> (Option<String>, f64) {
107        let w = self
108            .widths
109            .get(&code)
110            .copied()
111            .unwrap_or(self.default_width);
112        if let Some(s) = self.to_unicode.get(&code) {
113            return (Some(decompose_ligatures(s)), w);
114        }
115        if !self.two_byte {
116            // A GID-style `/Differences` name (no Unicode) overrides the base
117            // encoding, matching docling's verbatim `/g115` fallback.
118            if let Some(name) = self.fallback_names.get(&(code as u8)) {
119                return (Some(format!("/{name}")), w);
120            }
121            if let Some(enc) = &self.simple_encoding {
122                if let Some(&ch) = enc.get(&(code as u8)) {
123                    return (Some(decompose_ligatures(&ch.to_string())), w);
124                }
125            }
126            // Last resort: the embedded Type1 font program's own `/Encoding`
127            // vector (`dup N /glyphname put`). Standard TeX math fonts (CMMI, CMSY,
128            // …) ship no PDF `/Encoding` and no ToUnicode, so a glyph like `λ`
129            // (CMMI code 21 → `/lambda`) or `≤` (CMSY code 20 → `/lessequal`) has
130            // no other mapping and would otherwise be silently dropped. docling
131            // recovers these from the same font program. This only fills codes the
132            // base encoding left unmapped, so it never changes an existing decode.
133            if let Some(&ch) = self.program_encoding.get(&(code as u8)) {
134                return (Some(decompose_ligatures(&ch.to_string())), w);
135            }
136        }
137        (None, w)
138    }
139}
140
141/// Spell out Latin presentation-form ligatures (`fi`→`fi`, `ffi`→`ffi`, …) the way
142/// docling does, so `configuration`/`difficult` don't keep the ligature glyph.
143/// The chars share the ligature's box, so the line sanitizer recomposes them.
144fn decompose_ligatures(s: &str) -> String {
145    if !s.chars().any(|c| ('\u{FB00}'..='\u{FB06}').contains(&c)) {
146        return s.to_string();
147    }
148    s.chars()
149        .map(|c| {
150            match c {
151                '\u{FB00}' => "ff",
152                '\u{FB01}' => "fi",
153                '\u{FB02}' => "fl",
154                '\u{FB03}' => "ffi",
155                '\u{FB04}' => "ffl",
156                '\u{FB05}' => "ft",
157                '\u{FB06}' => "st",
158                _ => return c.to_string(),
159            }
160            .to_string()
161        })
162        .collect()
163}
164
165fn hash_name(name: &[u8]) -> u64 {
166    use std::hash::{Hash, Hasher};
167    let mut h = std::collections::hash_map::DefaultHasher::new();
168    name.hash(&mut h);
169    h.finish()
170}
171
172/// Resolve a possibly-indirect object to a dictionary.
173fn as_dict<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Dictionary> {
174    match obj {
175        Object::Dictionary(d) => Some(d),
176        Object::Reference(id) => doc.get_object(*id).ok().and_then(|o| o.as_dict().ok()),
177        _ => None,
178    }
179}
180
181fn deref<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Object> {
182    match obj {
183        Object::Reference(id) => doc.get_object(*id).ok(),
184        other => Some(other),
185    }
186}
187
188/// Parse one font dictionary into a [`Font`].
189fn parse_font(doc: &Document, name: &[u8], fdict: &Dictionary) -> Font {
190    let subtype: &[u8] = fdict
191        .get(b"Subtype")
192        .ok()
193        .and_then(|o| o.as_name().ok())
194        .unwrap_or(&[]);
195    let two_byte = subtype == b"Type0".as_slice();
196
197    let to_unicode = fdict
198        .get(b"ToUnicode")
199        .ok()
200        .and_then(|o| deref(doc, o))
201        .and_then(|o| o.as_stream().ok())
202        .and_then(|s| s.decompressed_content().ok())
203        .map(|data| parse_tounicode(&data))
204        .unwrap_or_default();
205
206    let (mut widths, mut default_width) = if two_byte {
207        cid_widths(doc, fdict)
208    } else {
209        simple_widths(doc, fdict)
210    };
211
212    let simple_encoding = if two_byte {
213        None
214    } else {
215        Some(simple_encoding_table(doc, fdict))
216    };
217
218    // A standard-14 font referenced without an embedded program usually ships
219    // no `/Widths` and no `/FontDescriptor` either (ReportLab's default, #187).
220    // Every advance then resolved to 0, the cells collapsed to zero width, and
221    // the page's whole text layer was silently dropped — while pdfium, with
222    // its built-in metrics, reads the same file fine. Fill the widths from the
223    // built-in Adobe Core 14 AFM tables via the font's own code→char decode
224    // (base encoding + `/Differences`), so an explicit `/Widths` always wins.
225    if !two_byte && widths.is_empty() && default_width == 0.0 {
226        if let Some(std14) = base_font_name(fdict).and_then(|n| crate::std14::widths_for(&n)) {
227            if let Some(enc) = &simple_encoding {
228                for (&code, &ch) in enc {
229                    if let Some(w) = std14.width(ch) {
230                        widths.insert(u32::from(code), w);
231                    }
232                }
233            }
234            // Codes the table misses still advance a typical width instead of
235            // stacking at x=0 (the failure mode this whole branch fixes).
236            default_width = 500.0;
237        }
238    }
239    let fallback_names = if two_byte {
240        HashMap::new()
241    } else {
242        differences_gid_names(doc, fdict)
243    };
244    let program_encoding = if two_byte {
245        HashMap::new()
246    } else {
247        type1_program_encoding(doc, fdict)
248    };
249
250    let (ascent, descent) = font_ascent_descent(doc, fdict, two_byte);
251
252    Font {
253        two_byte,
254        to_unicode,
255        widths,
256        default_width,
257        simple_encoding,
258        fallback_names,
259        program_encoding,
260        ascent,
261        descent,
262        hash: hash_name(name),
263        style: crate::font_style::parse_font_style(&String::from_utf8_lossy(
264            &base_font_name(fdict).unwrap_or_else(|| name.to_vec()),
265        )),
266    }
267}
268
269/// Collect `/Differences` entries whose glyph name is a GID placeholder
270/// (`g115`, `cid42`, `glyph7`, `index9`) with no Unicode mapping. docling-parse
271/// emits such glyphs as the literal name `/g115`; mapping them here keeps the
272/// text from being silently dropped (subsetted fonts with no ToUnicode). The
273/// GID-name restriction keeps real Adobe glyph names on the normal path so this
274/// never invents garbage on the clean files.
275fn differences_gid_names(doc: &Document, fdict: &Dictionary) -> HashMap<u8, String> {
276    let mut map = HashMap::new();
277    let Some(Object::Dictionary(enc)) = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o))
278    else {
279        return map;
280    };
281    let Some(Object::Array(diffs)) = enc.get(b"Differences").ok().and_then(|o| deref(doc, o))
282    else {
283        return map;
284    };
285    let mut code = 0u8;
286    for el in diffs {
287        match el {
288            Object::Integer(i) => code = *i as u8,
289            Object::Name(name) => {
290                if glyph_name_to_char(name).is_none() && is_gid_name(name) {
291                    map.insert(code, String::from_utf8_lossy(name).into_owned());
292                }
293                code = code.wrapping_add(1);
294            }
295            _ => {}
296        }
297    }
298    map
299}
300
301/// Parse the embedded Type1 font program's built-in `/Encoding` vector
302/// (`dup <code> /<glyphname> put` entries in the clear-text header before
303/// `eexec`) into `code → char`. This is how docling recovers glyphs from
304/// standard TeX math fonts (CMMI/CMSY/…) that carry no PDF `/Encoding` and no
305/// ToUnicode — e.g. CMMI's `dup 21 /lambda` or CMSY's `dup 20 /lessequal`.
306/// Only `FontFile` (Type1) is parsed; CFF (`FontFile3`) and TrueType
307/// (`FontFile2`) store their encoding in a binary table and are left alone.
308fn type1_program_encoding(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
309    let mut map = HashMap::new();
310    let Some(desc) = fdict
311        .get(b"FontDescriptor")
312        .ok()
313        .and_then(|o| deref(doc, o))
314        .and_then(|o| o.as_dict().ok())
315    else {
316        return map;
317    };
318    let Some(data) = desc
319        .get(b"FontFile")
320        .ok()
321        .and_then(|o| deref(doc, o))
322        .and_then(|o| o.as_stream().ok())
323        .and_then(|s| s.decompressed_content().ok())
324    else {
325        return map;
326    };
327    // The clear-text header (PostScript) ends at `eexec`; the rest is encrypted.
328    let head_end = data
329        .windows(5)
330        .position(|w| w == b"eexec")
331        .unwrap_or(data.len());
332    let head = String::from_utf8_lossy(&data[..head_end]);
333    // Scan for `dup <code> /<name> put` tokens.
334    let toks: Vec<&str> = head.split_whitespace().collect();
335    for w in toks.windows(4) {
336        if w[0] == "dup" && w[3] == "put" {
337            if let (Ok(code), Some(name)) = (w[1].parse::<u32>(), w[2].strip_prefix('/')) {
338                if code <= 255 {
339                    if let Some(ch) = glyph_name_to_char(name.as_bytes()) {
340                        map.insert(code as u8, ch);
341                    }
342                }
343            }
344        }
345    }
346    map
347}
348
349/// A glyph name that is a synthetic placeholder, not a real Adobe name:
350/// `g115`, `cid42`, `glyph7`, `index9`, `G12`, or a short-prefix code name like
351/// `SM590000` (IBM BookMaster). These carry no Unicode meaning, and docling-parse
352/// emits them verbatim (`/SM590000`). `afii####` / `uni####` are real Adobe names
353/// and excluded. The restriction keeps genuine glyph names on the Unicode path.
354fn is_gid_name(name: &[u8]) -> bool {
355    let Ok(s) = std::str::from_utf8(name) else {
356        return false;
357    };
358    if s.starts_with("afii") || s.starts_with("uni") {
359        return false;
360    }
361    for prefix in ["g", "G", "cid", "CID", "glyph", "index"] {
362        if let Some(rest) = s.strip_prefix(prefix) {
363            if !rest.is_empty() && rest.bytes().all(|b| b.is_ascii_digit()) {
364                return true;
365            }
366        }
367    }
368    // Short alpha prefix (≤3 letters) followed by a run of ≥3 digits — synthetic
369    // code names like `SM590000`, distinct from real Adobe names (whole words or
370    // letter+`.suffix` variants).
371    let alpha = s.bytes().take_while(|b| b.is_ascii_alphabetic()).count();
372    let digits = s.len() - alpha;
373    (1..=3).contains(&alpha)
374        && digits >= 3
375        && s.as_bytes()[alpha..].iter().all(|b| b.is_ascii_digit())
376}
377
378fn font_ascent_descent(doc: &Document, fdict: &Dictionary, two_byte: bool) -> (f64, f64) {
379    // For Type0, the descriptor lives on the descendant CIDFont.
380    let descr_owner = if two_byte {
381        fdict
382            .get(b"DescendantFonts")
383            .ok()
384            .and_then(|o| deref(doc, o))
385            .and_then(|o| match o {
386                Object::Array(a) => a.first(),
387                _ => None,
388            })
389            .and_then(|o| as_dict(doc, o))
390    } else {
391        Some(fdict)
392    };
393    let fd = descr_owner
394        .and_then(|d| d.get(b"FontDescriptor").ok())
395        .and_then(|o| as_dict(doc, o));
396    let asc = fd
397        .and_then(|d| d.get(b"Ascent").ok())
398        .and_then(|o| {
399            o.as_float()
400                .ok()
401                .or_else(|| o.as_i64().ok().map(|i| i as f32))
402        })
403        .unwrap_or(750.0) as f64;
404    let desc = fd
405        .and_then(|d| d.get(b"Descent").ok())
406        .and_then(|o| {
407            o.as_float()
408                .ok()
409                .or_else(|| o.as_i64().ok().map(|i| i as f32))
410        })
411        .unwrap_or(-250.0) as f64;
412    // Some subsetted fonts carry a degenerate FontDescriptor (`/Ascent 0
413    // /Descent 0`) — the real metrics live in the font program. That collapses
414    // the loose box to zero height, so the line cells get zero area and the
415    // layout's region/text assignment drops them (2305's References list lost
416    // every prose line, keeping only the URLs). Fall back to typical text metrics
417    // so the box has height.
418    if asc - desc <= 1.0 {
419        return (750.0, -250.0);
420    }
421    (asc, desc)
422}
423
424/// The `/BaseFont` name with any `ABCDEF+` subset prefix stripped (#187).
425fn base_font_name(fdict: &Dictionary) -> Option<Vec<u8>> {
426    let name = fdict.get(b"BaseFont").ok()?.as_name().ok()?;
427    let stripped = match name.iter().position(|&b| b == b'+') {
428        Some(i) if i == 6 => &name[i + 1..],
429        _ => name,
430    };
431    Some(stripped.to_vec())
432}
433
434/// Simple-font widths: `/FirstChar` + `/Widths` array, `/MissingWidth` default.
435fn simple_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
436    let mut map = HashMap::new();
437    let first = fdict
438        .get(b"FirstChar")
439        .ok()
440        .and_then(|o| o.as_i64().ok())
441        .unwrap_or(0) as u32;
442    if let Some(Object::Array(arr)) = fdict.get(b"Widths").ok().and_then(|o| deref(doc, o)) {
443        for (i, w) in arr.iter().enumerate() {
444            if let Some(w) = num(w) {
445                map.insert(first + i as u32, w);
446            }
447        }
448    }
449    let dw = fdict
450        .get(b"FontDescriptor")
451        .ok()
452        .and_then(|o| as_dict(doc, o))
453        .and_then(|d| d.get(b"MissingWidth").ok())
454        .and_then(num)
455        .unwrap_or(0.0);
456    (map, dw)
457}
458
459/// CIDFont widths: the `/W` array on the descendant font (`/DW` default = 1000).
460fn cid_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
461    let mut map = HashMap::new();
462    let Some(desc) = fdict
463        .get(b"DescendantFonts")
464        .ok()
465        .and_then(|o| deref(doc, o))
466        .and_then(|o| match o {
467            Object::Array(a) => a.first(),
468            _ => None,
469        })
470        .and_then(|o| as_dict(doc, o))
471    else {
472        return (map, 1000.0);
473    };
474    let dw = desc.get(b"DW").ok().and_then(num).unwrap_or(1000.0);
475    if let Some(Object::Array(w)) = desc.get(b"W").ok().and_then(|o| deref(doc, o)) {
476        let mut i = 0;
477        while i < w.len() {
478            let c = w.get(i).and_then(num);
479            match (c, w.get(i + 1)) {
480                // `c [w1 w2 ...]`: consecutive CIDs starting at c.
481                (Some(c), Some(Object::Array(list))) => {
482                    for (k, wv) in list.iter().enumerate() {
483                        if let Some(wv) = num(wv) {
484                            map.insert(c as u32 + k as u32, wv);
485                        }
486                    }
487                    i += 2;
488                }
489                // `c_first c_last w`: a run all of width w.
490                (Some(c1), Some(o2)) => {
491                    if let (Some(c2), Some(wv)) = (num(o2), w.get(i + 2).and_then(num)) {
492                        for cid in c1 as u32..=c2 as u32 {
493                            map.insert(cid, wv);
494                        }
495                    }
496                    i += 3;
497                }
498                _ => break,
499            }
500        }
501    }
502    (map, dw)
503}
504
505fn num(o: &Object) -> Option<f64> {
506    match o {
507        Object::Integer(i) => Some(*i as f64),
508        Object::Real(r) => Some(*r as f64),
509        _ => None,
510    }
511}
512
513/// Parse a ToUnicode CMap's `bfchar` / `bfrange` sections into code→string.
514pub(crate) fn parse_tounicode(data: &[u8]) -> HashMap<u32, String> {
515    let text = String::from_utf8_lossy(data);
516    let mut map = HashMap::new();
517    let hex = |s: &str| -> Option<Vec<u16>> {
518        let s = s.trim();
519        if !s.starts_with('<') || !s.ends_with('>') {
520            return None;
521        }
522        let h = &s[1..s.len() - 1];
523        let bytes: Vec<u8> = (0..h.len())
524            .step_by(2)
525            .filter_map(|i| u8::from_str_radix(h.get(i..i + 2)?, 16).ok())
526            .collect();
527        Some(
528            bytes
529                .chunks(2)
530                .map(|c| {
531                    if c.len() == 2 {
532                        u16::from_be_bytes([c[0], c[1]])
533                    } else {
534                        c[0] as u16
535                    }
536                })
537                .collect(),
538        )
539    };
540    let u16s_to_string = |u: &[u16]| String::from_utf16_lossy(u);
541    let code_of = |u: &[u16]| u.iter().fold(0u32, |acc, &x| (acc << 16) | x as u32);
542
543    // Tokenize by structure, not whitespace: CMap hex groups are often written
544    // back-to-back with no separators (`<21><21><0054>`), so scan for `<…>`
545    // groups, `[`/`]` brackets, and bareword keywords.
546    let tokens: Vec<String> = {
547        let bytes = text.as_bytes();
548        let mut toks = Vec::new();
549        let mut i = 0;
550        while i < bytes.len() {
551            let c = bytes[i];
552            if c.is_ascii_whitespace() {
553                i += 1;
554            } else if c == b'<' {
555                let start = i;
556                while i < bytes.len() && bytes[i] != b'>' {
557                    i += 1;
558                }
559                i += 1; // include '>'
560                toks.push(String::from_utf8_lossy(&bytes[start..i.min(bytes.len())]).into_owned());
561            } else if c == b'[' || c == b']' {
562                toks.push((c as char).to_string());
563                i += 1;
564            } else {
565                let start = i;
566                while i < bytes.len()
567                    && !bytes[i].is_ascii_whitespace()
568                    && bytes[i] != b'<'
569                    && bytes[i] != b'['
570                    && bytes[i] != b']'
571                {
572                    i += 1;
573                }
574                toks.push(String::from_utf8_lossy(&bytes[start..i]).into_owned());
575            }
576        }
577        toks
578    };
579    let tokens: Vec<&str> = tokens.iter().map(|s| s.as_str()).collect();
580    let mut i = 0;
581    while i < tokens.len() {
582        match tokens[i] {
583            "beginbfchar" => {
584                i += 1;
585                while i + 1 < tokens.len() && tokens[i] != "endbfchar" {
586                    if let (Some(src), Some(dst)) = (hex(tokens[i]), hex(tokens[i + 1])) {
587                        map.insert(code_of(&src), u16s_to_string(&dst));
588                    }
589                    i += 2;
590                }
591            }
592            "beginbfrange" => {
593                i += 1;
594                while i + 2 < tokens.len() && tokens[i] != "endbfrange" {
595                    let (Some(lo), Some(hi)) = (hex(tokens[i]), hex(tokens[i + 1])) else {
596                        i += 1;
597                        continue;
598                    };
599                    let lo = code_of(&lo);
600                    let hi = code_of(&hi);
601                    if tokens[i + 2] == "[" {
602                        // `<lo> <hi> [ <d0> <d1> ... ]`: one dst per code in the range.
603                        let mut j = i + 3;
604                        let mut code = lo;
605                        while j < tokens.len() && tokens[j] != "]" {
606                            if let Some(dst) = hex(tokens[j]) {
607                                map.insert(code, u16s_to_string(&dst));
608                            }
609                            code += 1;
610                            j += 1;
611                        }
612                        i = j + 1;
613                    } else if let Some(dst) = hex(tokens[i + 2]) {
614                        // `<lo> <hi> <dst>`: consecutive Unicode from a base.
615                        let base = code_of(&dst);
616                        for (k, code) in (lo..=hi).enumerate() {
617                            if let Some(ch) = char::from_u32(base + k as u32) {
618                                map.insert(code, ch.to_string());
619                            }
620                        }
621                        i += 3;
622                    } else {
623                        i += 1;
624                    }
625                }
626            }
627            _ => i += 1,
628        }
629    }
630    map
631}
632
633/// Decode a PDF string literal in a Tj/TJ operand into raw code units.
634fn codes(font: &Font, bytes: &[u8]) -> Vec<u32> {
635    if font.two_byte {
636        bytes
637            .chunks(2)
638            .map(|c| {
639                if c.len() == 2 {
640                    ((c[0] as u32) << 8) | c[1] as u32
641                } else {
642                    c[0] as u32
643                }
644            })
645            .collect()
646    } else {
647        bytes.iter().map(|&b| b as u32).collect()
648    }
649}
650
651/// A page's display box, in PDF user space: the `/CropBox` clipped to the
652/// `/MediaBox`, both inherited through the page tree and normalized — a
653/// missing or empty MediaBox is US Letter, an empty CropBox is the MediaBox
654/// (pdfium's `CPDF_Page::UpdateDimensions`). pdfium reports the page size from
655/// this box, renders exactly it, and translates every content coordinate so
656/// its lower-left corner is the origin (`m_PageMatrix`); docling's backends
657/// inherit that frame, so text cells, `prov` boxes and destinations all count
658/// from the CropBox corner, not the MediaBox one. The parser used to flip
659/// glyphs with the MediaBox *height* and no translation at all, so a page
660/// whose boxes do not start at (0, 0) — a trimmed book page with
661/// `MediaBox [-56 -58 576 723]` / `CropBox [1 -0.6 519 666]`, or a LaTeX
662/// figure cropped to `[156 147 637 391]` — had its text displaced against the
663/// rendered bitmap by the box offset, the bottom lines pushed past the page
664/// edge and clamped to `t = b`.
665#[derive(Debug, Clone, Copy, PartialEq)]
666pub(crate) struct PageBox {
667    /// Left edge, user space.
668    pub l: f32,
669    /// Bottom edge, user space.
670    pub b: f32,
671    pub w: f32,
672    pub h: f32,
673}
674
675impl PageBox {
676    /// Top edge, user space — the y that becomes `0` in the y-down frame.
677    pub fn top(&self) -> f32 {
678        self.b + self.h
679    }
680}
681
682/// A page-tree rect attribute (`/MediaBox`, `/CropBox`), inherited from the
683/// nearest ancestor that sets it, as normalized `(l, b, r, t)`.
684fn inherited_rect(
685    doc: &Document,
686    page_id: lopdf::ObjectId,
687    key: &[u8],
688) -> Option<(f32, f32, f32, f32)> {
689    let mut id = page_id;
690    for _ in 0..32 {
691        let dict = doc.get_object(id).ok()?.as_dict().ok()?;
692        if let Some(Object::Array(a)) = dict.get(key).ok().and_then(|o| deref(doc, o)) {
693            let v: Vec<f32> = a.iter().filter_map(|o| num(o).map(|x| x as f32)).collect();
694            if v.len() == 4 && v.iter().all(|x| x.is_finite()) {
695                return Some((
696                    v[0].min(v[2]),
697                    v[1].min(v[3]),
698                    v[0].max(v[2]),
699                    v[1].max(v[3]),
700                ));
701            }
702            return None;
703        }
704        id = dict.get(b"Parent").ok()?.as_reference().ok()?;
705    }
706    None
707}
708
709pub(crate) fn page_box(doc: &Document, page_id: lopdf::ObjectId) -> PageBox {
710    let nonempty = |r: &(f32, f32, f32, f32)| r.2 > r.0 && r.3 > r.1;
711    let media = inherited_rect(doc, page_id, b"MediaBox")
712        .filter(nonempty)
713        .unwrap_or((0.0, 0.0, 612.0, 792.0));
714    let crop = inherited_rect(doc, page_id, b"CropBox")
715        .map(|c| {
716            (
717                c.0.max(media.0),
718                c.1.max(media.1),
719                c.2.min(media.2),
720                c.3.min(media.3),
721            )
722        })
723        .filter(nonempty)
724        .unwrap_or(media);
725    PageBox {
726        l: crop.0,
727        b: crop.1,
728        w: crop.2 - crop.0,
729        h: crop.3 - crop.1,
730    }
731}
732
733/// Page size (width, height) in PDF points — the display box's, like pdfium's
734/// `FPDF_GetPageWidthF/HeightF`.
735fn page_size(doc: &Document, page_id: lopdf::ObjectId) -> (f32, f32) {
736    let pb = page_box(doc, page_id);
737    (pb.w, pb.h)
738}
739
740/// Localize where a page's text is lost, for the `text_layer` diagnostic.
741/// Extraction can come up empty at three different points — no content stream
742/// reached the parser, the stream did not decode into operators, or it ran but
743/// produced no glyphs (fonts/encodings) — and from the outside all three look
744/// the same. Report them per page.
745pub fn content_diagnosis(bytes: &[u8]) -> String {
746    let Some(doc) = load_document(bytes) else {
747        return "document does not load".into();
748    };
749    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
750    pages.sort_by_key(|(n, _)| *n);
751    let mut out = String::new();
752    let mut caches = DocCaches::default();
753    for (n, pid) in pages.into_iter().take(4) {
754        let content_bytes = doc.get_page_content(pid);
755        let ops = lopdf::content::Content::decode(&content_bytes)
756            .map(|c| c.operations.len())
757            .ok();
758        let res = page_res(&doc, pid);
759        let fonts = res.map(|r| fonts_from_res(&doc, r, &mut caches).len());
760        let glyphs = page_glyphs_cached(&doc, pid, &mut caches).len();
761        out.push_str(&format!(
762            "\n   page {n}: content {} B, ops {}, resources {}, fonts {}, glyphs {}",
763            content_bytes.len(),
764            ops.map_or("UNDECODABLE".to_string(), |n| n.to_string()),
765            if res.is_some() { "ok" } else { "MISSING" },
766            fonts.map_or("-".to_string(), |n| n.to_string()),
767            glyphs,
768        ));
769    }
770    out
771}
772
773/// Is this "text layer" a vestige rather than the document's text?
774///
775/// Scanned forms often carry a handful of typed-in strings — a date filled
776/// into three form fields, say — on top of pages that are otherwise images.
777/// Treating that as a real text layer is the worst of both worlds: the text
778/// path proudly extracts thirteen characters, and no OCR ever runs on the
779/// letter the pages actually show. The reported form did exactly this (3
780/// lines, 13 chars, 3 pages).
781///
782/// The rule is deliberately tight so genuinely sparse *digital* documents are
783/// not misrouted into OCR: only a document averaging at most one line per page
784/// **and** totalling fewer than 32 characters is called vestigial.
785pub fn text_layer_is_vestigial(pages: &[crate::pdfium_backend::PdfPage]) -> bool {
786    let mut tally = TextLayerTally::default();
787    pages.iter().for_each(|p| tally.add(p));
788    tally.is_vestigial()
789}
790
791/// [`text_layer_is_vestigial`]'s counts, accumulated a page at a time so a
792/// streaming caller need not keep the pages around to decide.
793#[derive(Debug, Default, Clone, Copy)]
794pub struct TextLayerTally {
795    pages: usize,
796    lines: usize,
797    chars: usize,
798}
799
800impl TextLayerTally {
801    pub fn add(&mut self, page: &crate::pdfium_backend::PdfPage) {
802        self.pages += 1;
803        self.lines += page.cells.len();
804        self.chars += page
805            .cells
806            .iter()
807            .map(|c| c.text.chars().count())
808            .sum::<usize>();
809    }
810
811    /// The verdict over the pages added so far.
812    pub fn is_vestigial(&self) -> bool {
813        self.lines == 0 || (self.lines <= self.pages && self.chars < 32)
814    }
815
816    /// Whether the pages added so far already rule vestigial out for a
817    /// document of `total_pages`, whatever the remaining pages hold — both
818    /// counts only grow, so a caller may stop parsing pages it does not need.
819    pub fn proves_text(&self, total_pages: usize) -> bool {
820        self.lines > 0 && (self.lines > total_pages || self.chars >= 32)
821    }
822}
823
824/// Why the cross-reference repair did or did not fire, for the `text_layer`
825/// diagnostic. A PDF that will not load is indistinguishable from a scan in
826/// production (both convert to nothing), so the reason has to be askable.
827pub fn xref_repair_status(bytes: &[u8]) -> String {
828    if Document::load_mem(bytes).is_ok() {
829        return "loads unaided; no repair needed".into();
830    }
831    match pad_short_xref_entries(bytes) {
832        Ok(fixed) => match Document::load_mem(&fixed) {
833            Ok(_) => "repaired: cross-reference entries padded to 20 bytes".into(),
834            Err(e) => format!("padded the entries, but it still will not load: {e}"),
835        },
836        Err(why) => format!("repair declined — {why}"),
837    }
838}
839
840/// Load a PDF, repairing the one malformation that otherwise costs us the whole
841/// document: **19-byte cross-reference entries**.
842///
843/// The spec fixes an xref entry at 20 bytes — `nnnnnnnnnn ggggg n` plus a
844/// *two*-byte EOL. Some generators (an Austrian telecom's invoices, for one)
845/// emit a bare LF instead, making each entry 19 bytes. lopdf rejects the file
846/// outright (`invalid file trailer`) where pdfium reads it happily, so a
847/// perfectly good text layer looked to the browser exactly like a scan and cost
848/// ten seconds of OCR.
849///
850/// Padding is only attempted when it cannot move anything the xref points at:
851/// a single `xref` section that begins after the last object. The repair then
852/// has to prove itself — the padded bytes are used only if they load — so a
853/// mis-repair degrades to today's behaviour rather than to silent garbage.
854pub(crate) fn load_document(bytes: &[u8]) -> Option<Document> {
855    open_document(bytes, None).ok()
856}
857
858/// Why [`open_document`] could not hand back a readable document.
859#[derive(Debug, Clone, Copy, PartialEq, Eq)]
860pub(crate) enum OpenError {
861    /// lopdf cannot read the file even after the repairs.
862    Unreadable,
863    /// The file is encrypted and the password given (or none) does not open it.
864    Password,
865}
866
867/// Load `bytes` with the document's password. lopdf decrypts while reading —
868/// with `password`, or the empty user password most "protected" PDFs carry
869/// (what every viewer opens silently) — so every reader downstream sees plain
870/// streams; a file the password does not open loads with its streams still
871/// encrypted, which is reported as [`OpenError::Password`] rather than handed
872/// on as a document whose every stream decodes to nothing.
873pub(crate) fn open_document(bytes: &[u8], password: Option<&str>) -> Result<Document, OpenError> {
874    let Some(doc) = load_document_raw(bytes, password) else {
875        // lopdf refuses to load at all under a *wrong* password (a missing
876        // one loads the file with its streams still encrypted); tell the two
877        // apart by loading without it.
878        if password.is_some() && load_document_raw(bytes, None).is_some_and(|d| d.is_encrypted()) {
879            return Err(OpenError::Password);
880        }
881        return Err(OpenError::Unreadable);
882    };
883    if doc.is_encrypted() {
884        return Err(OpenError::Password);
885    }
886    Ok(doc)
887}
888
889fn load_options(password: Option<&str>) -> lopdf::LoadOptions {
890    lopdf::LoadOptions {
891        password: password.map(str::to_string),
892        ..lopdf::LoadOptions::default()
893    }
894}
895
896fn load_document_raw(bytes: &[u8], password: Option<&str>) -> Option<Document> {
897    // Try progressively more repair, and accept a candidate only once the pages
898    // actually carry content — a document whose streams were dropped still
899    // "loads", so loading alone is not evidence the repair helped. A
900    // well-formed file returns on the first attempt and pays for nothing.
901    let mut fallback = None;
902    if let Some(doc) = best_effort_load(bytes, password, &mut fallback) {
903        return Some(doc);
904    }
905    let xref_fixed = pad_short_xref_entries(bytes).ok();
906    if let Some(fixed) = &xref_fixed {
907        if let Some(doc) = best_effort_load(fixed, password, &mut fallback) {
908            return Some(doc);
909        }
910    }
911    // Both defects can coexist, and the second only becomes visible once the
912    // first is repaired, so build on whatever the previous step produced.
913    let lengths_fixed = fix_stream_lengths(xref_fixed.as_deref().unwrap_or(bytes));
914    if let Some(doc) = best_effort_load(&lengths_fixed, password, &mut fallback) {
915        return Some(doc);
916    }
917    fallback
918}
919
920/// Load `data`, returning it only when its pages carry content; a document that
921/// merely parses is remembered as the fallback for when nothing does better.
922fn best_effort_load(
923    data: &[u8],
924    password: Option<&str>,
925    fallback: &mut Option<Document>,
926) -> Option<Document> {
927    match Document::load_mem_with_options(data, load_options(password)) {
928        Ok(doc) if has_page_content(&doc) => Some(doc),
929        Ok(doc) => {
930            fallback.get_or_insert(doc);
931            None
932        }
933        Err(_) => None,
934    }
935}
936
937/// Does any page actually hand us a content stream? A document whose streams
938/// were dropped still parses — it simply has nothing to read — so this is what
939/// tells a successful repair from a pointless one.
940fn has_page_content(doc: &Document) -> bool {
941    doc.get_pages()
942        .into_values()
943        .take(4)
944        .any(|pid| !doc.get_page_content(pid).is_empty())
945}
946
947/// Correct `/Length` values that disagree with where `endstream` actually is.
948///
949/// The same generator that writes short xref entries also overstates its
950/// content-stream lengths by a byte or two. lopdf trusts `/Length`, reads past
951/// the data, fails to find `endstream` there and drops the stream — the object
952/// comes back as a bare dictionary, so the page has no content at all and the
953/// document looks like a scan. pdfium instead trusts `endstream`, which is what
954/// this does.
955///
956/// The rewrite is length-preserving: the corrected number is written over the
957/// old digits and padded with spaces, so every byte offset in the file — and
958/// therefore the whole cross-reference table — stays valid.
959fn fix_stream_lengths(bytes: &[u8]) -> Vec<u8> {
960    let mut out = bytes.to_vec();
961    let mut i = 0;
962    while let Some(rel) = find(&out[i..], b"stream") {
963        let kw = i + rel;
964        i = kw + 6;
965        // Skip `endstream` (the keyword we are measuring *to*).
966        if kw >= 3 && &out[kw - 3..kw] == b"end" {
967            continue;
968        }
969        // The stream data starts after the EOL that follows the keyword.
970        let mut data = kw + 6;
971        if out.get(data..data + 2) == Some(b"\r\n".as_slice()) {
972            data += 2;
973        } else if matches!(out.get(data), Some(b'\n' | b'\r')) {
974            data += 1;
975        }
976        let Some(end) = find(&out[data..], b"endstream").map(|r| data + r) else {
977            continue;
978        };
979        // `/Length <digits>` in the dictionary just before the keyword.
980        let dict_start = out[..kw].iter().rposition(|&c| c == b'<').unwrap_or(0);
981        let Some(lrel) = find(&out[dict_start..kw], b"/Length") else {
982            continue;
983        };
984        let mut d = dict_start + lrel + 7;
985        while matches!(out.get(d), Some(b' ')) {
986            d += 1;
987        }
988        let digits = out[d..].iter().take_while(|c| c.is_ascii_digit()).count();
989        if digits == 0 {
990            continue;
991        }
992        let declared: usize = match std::str::from_utf8(&out[d..d + digits])
993            .ok()
994            .and_then(|s| s.parse().ok())
995        {
996            Some(v) => v,
997            None => continue,
998        };
999        let actual = end - data;
1000        // Only shrink, and only when the new value fits the space the old one
1001        // occupied — growing the number would move every following byte.
1002        let replacement = actual.to_string();
1003        if actual == declared || replacement.len() > digits {
1004            continue;
1005        }
1006        out[d..d + digits].fill(b' ');
1007        out[d..d + replacement.len()].copy_from_slice(replacement.as_bytes());
1008    }
1009    out
1010}
1011
1012fn find(haystack: &[u8], needle: &[u8]) -> Option<usize> {
1013    haystack.windows(needle.len()).position(|w| w == needle)
1014}
1015
1016/// Rewrite a classic cross-reference table's entries to the spec's 20 bytes,
1017/// or `None` when the file's shape makes that unsafe (see [`load_document`]).
1018fn pad_short_xref_entries(bytes: &[u8]) -> Result<Vec<u8>, &'static str> {
1019    // Exactly one xref section, and it must start after every object, so that
1020    // growing it shifts nothing the table's offsets refer to.
1021    let is_boundary = |i: usize| i == 0 || matches!(bytes[i - 1], b'\n' | b'\r');
1022    let mut starts = (0..bytes.len().saturating_sub(4))
1023        .filter(|&i| &bytes[i..i + 4] == b"xref" && is_boundary(i));
1024    let xref_at = starts
1025        .next()
1026        .ok_or("no classic `xref` section (an xref stream?)")?;
1027    if starts.next().is_some() {
1028        return Err("more than one xref section (incremental update)");
1029    }
1030    let last_obj = bytes
1031        .windows(3)
1032        .rposition(|w| w == b"obj")
1033        .ok_or("no objects found")?;
1034    if last_obj > xref_at {
1035        return Err("an object follows the xref — padding would move it");
1036    }
1037
1038    let mut out = bytes[..xref_at].to_vec();
1039    out.extend_from_slice(b"xref\n");
1040    let mut i = xref_at + 4;
1041    let skip_ws = |i: &mut usize| {
1042        while matches!(bytes.get(*i), Some(b'\r' | b'\n' | b' ')) {
1043            *i += 1;
1044        }
1045    };
1046    loop {
1047        skip_ws(&mut i);
1048        // Either the next subsection header ("first count") or the trailer.
1049        if bytes[i..].starts_with(b"trailer") {
1050            out.extend_from_slice(&bytes[i..]);
1051            return Ok(out);
1052        }
1053        let header_end = i + bytes[i..]
1054            .iter()
1055            .position(|c| matches!(c, b'\n' | b'\r'))
1056            .ok_or("subsection header runs off the end")?;
1057        let header = std::str::from_utf8(&bytes[i..header_end])
1058            .map_err(|_| "subsection header is not text")?
1059            .trim();
1060        let mut parts = header.split_whitespace();
1061        let count: usize = parts
1062            .nth(1)
1063            .and_then(|c| c.parse().ok())
1064            .ok_or("unparseable subsection header")?;
1065        if parts.next().is_some() || count == 0 {
1066            return Err("unexpected subsection header shape");
1067        }
1068        out.extend_from_slice(header.as_bytes());
1069        out.push(b'\n');
1070        i = header_end;
1071        for _ in 0..count {
1072            skip_ws(&mut i);
1073            // `nnnnnnnnnn ggggg n` — the 18 bytes before whatever EOL follows.
1074            let entry = bytes.get(i..i + 18).ok_or("xref entry runs off the end")?;
1075            let well_formed = entry[..10].iter().all(u8::is_ascii_digit)
1076                && entry[10] == b' '
1077                && entry[11..16].iter().all(u8::is_ascii_digit)
1078                && entry[16] == b' '
1079                && matches!(entry[17], b'n' | b'f');
1080            if !well_formed {
1081                return Err("xref entry is not `nnnnnnnnnn ggggg n`");
1082            }
1083            out.extend_from_slice(entry);
1084            out.extend_from_slice(b" \n"); // the spec's 2-byte EOL -> 20 bytes
1085            i += 18;
1086        }
1087    }
1088}
1089
1090/// Debug: raw glyph stream `(ch, ll, lr, lb, lt)` (native coords) for page
1091/// `index`, before the sanitizer. For comparing char cells to docling-parse.
1092pub fn debug_glyphs(bytes: &[u8], index: usize) -> Vec<(char, f32, f32, f32, f32)> {
1093    let Some(doc) = load_document(bytes) else {
1094        return Vec::new();
1095    };
1096    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1097    pages.sort_by_key(|(n, _)| *n);
1098    let Some((_, pid)) = pages.get(index) else {
1099        return Vec::new();
1100    };
1101    page_glyphs(&doc, *pid)
1102        .into_iter()
1103        .map(|g| (g.ch, g.ll, g.lr, g.lb, g.lt))
1104        .collect()
1105}
1106
1107/// Public entry: per-page (width, height, line cells) for a PDF, via the Rust
1108/// text parser + the docling-parse line sanitizer. Used by the pipeline and the
1109/// `textparse_dump` example.
1110pub fn pdf_textlines(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
1111    let Some(doc) = load_document(bytes) else {
1112        return Vec::new();
1113    };
1114    let mut caches = DocCaches::default();
1115    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1116    pages.sort_by_key(|(n, _)| *n);
1117    pages
1118        .into_iter()
1119        .map(|(_, pid)| {
1120            let (w, h) = page_size(&doc, pid);
1121            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1122            let cells = crate::dp_lines::line_cells(&glyphs, h, true);
1123            (w, h, cells)
1124        })
1125        .collect()
1126}
1127
1128/// Debug/diagnostic entry: per-page (width, height, word cells) for a PDF, via
1129/// the Rust parser glyphs run through the docling-parse word grouping. Used to
1130/// compare parser word cells against docling-parse's `word_cells` oracle (roadmap
1131/// item 6).
1132pub fn pdf_words(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
1133    let Some(doc) = load_document(bytes) else {
1134        return Vec::new();
1135    };
1136    let mut caches = DocCaches::default();
1137    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1138    pages.sort_by_key(|(n, _)| *n);
1139    pages
1140        .into_iter()
1141        .map(|(_, pid)| {
1142            let (w, h) = page_size(&doc, pid);
1143            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1144            let cells = crate::dp_lines::word_cells(&glyphs, h, true);
1145            (w, h, cells)
1146        })
1147        .collect()
1148}
1149
1150/// One page's text cells from the pure-Rust parser: prose line cells, per-word
1151/// cells, and code line cells — all from a single glyph parse. Replaces the
1152/// pdfium text path (roadmap item 6) when the parser drop is enabled.
1153#[derive(Default)]
1154pub struct PageParserCells {
1155    pub prose: Vec<crate::pdfium_backend::TextCell>,
1156    pub words: Vec<crate::pdfium_backend::TextCell>,
1157    pub code: Vec<crate::pdfium_backend::TextCell>,
1158    /// Drawn checkbox squares (#609), in the cells' frame.
1159    pub checkboxes: Vec<crate::checkbox::CheckBox>,
1160}
1161
1162/// The parser text layer, driven one page at a time: the document is loaded
1163/// (and repaired, see [`load_document`]) once, the font/form caches persist
1164/// across pages, and each page's glyphs are parsed only when asked for.
1165///
1166/// The eager whole-document walk this replaces ran *before* the first page
1167/// was rendered, so on a long PDF it was a serial prefix the page-worker pool
1168/// sat idle through — 6.2 s on the 1913-page .NET reference, in front of a
1169/// pipeline that otherwise overlaps parsing with inference — and a `--pages`
1170/// window still paid for every page in the file. Pulling pages on demand
1171/// keeps the parse on the producer thread but interleaved with rendering,
1172/// and skips unselected pages entirely. Output per page is unchanged: same
1173/// glyph walk, same shared caches, same contraction.
1174pub struct PageTextParser {
1175    doc: Document,
1176    caches: DocCaches,
1177    /// Page object ids in document order (page 1 first).
1178    pages: Vec<lopdf::ObjectId>,
1179}
1180
1181impl PageTextParser {
1182    /// Load the document; `None` when lopdf cannot read it at all.
1183    pub fn open(bytes: &[u8]) -> Option<Self> {
1184        Self::open_with_password(bytes, None)
1185    }
1186
1187    /// [`open`](Self::open) with the document's password (an encrypted file
1188    /// the password does not open reads as `None` here; the pipeline reports
1189    /// it through [`crate::pdf_meta::PdfMeta::open_with_password`] first).
1190    pub fn open_with_password(bytes: &[u8], password: Option<&str>) -> Option<Self> {
1191        let doc = open_document(bytes, password).ok()?;
1192        let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1193        pages.sort_by_key(|(n, _)| *n);
1194        Some(Self {
1195            doc,
1196            caches: DocCaches::default(),
1197            pages: pages.into_iter().map(|(_, pid)| pid).collect(),
1198        })
1199    }
1200
1201    /// The glyph boxes and font styles of the 0-based page `index` — the
1202    /// heading-hierarchy stage's style signal (#302): every non-space glyph's
1203    /// box (font ascent + descent at its size — the font-size proxy pdfium's
1204    /// loose char box also gave) in top-left coordinates, with the weight
1205    /// class and slant its `/BaseFont` name declares. Empty for a page
1206    /// without a text layer (a scan), and the stage falls back to its other
1207    /// signals.
1208    pub(crate) fn glyph_styles(
1209        &mut self,
1210        index: usize,
1211    ) -> Vec<crate::heading_hierarchy::GlyphStyle> {
1212        let Some(&pid) = self.pages.get(index) else {
1213            return Vec::new();
1214        };
1215        let (_w, h) = page_size(&self.doc, pid);
1216        let glyphs = page_glyphs_cached(&self.doc, pid, &mut self.caches);
1217        // Font hash → style, over the fonts the walk just parsed (an inline,
1218        // uncached font dictionary reads as unstyled).
1219        let styles: HashMap<u64, crate::font_style::FontStyle> = self
1220            .caches
1221            .fonts
1222            .values()
1223            .map(|f| (f.hash, f.style))
1224            .collect();
1225        glyphs
1226            .iter()
1227            .filter(|g| !g.ch.is_whitespace() && g.ll.is_finite())
1228            .map(|g| {
1229                let st = styles.get(&g.font).copied().unwrap_or_default();
1230                crate::heading_hierarchy::GlyphStyle {
1231                    l: g.ll,
1232                    t: h - g.lt,
1233                    r: g.lr,
1234                    b: h - g.lb,
1235                    height: g.height(),
1236                    weight_cls: crate::font_style::weight_class(st.weight),
1237                    italic: st.italic,
1238                    styled: st.known,
1239                }
1240            })
1241            .collect()
1242    }
1243
1244    /// Prose, word and code cells of the 0-based page `index` — empty for an
1245    /// index the parser's page tree doesn't have (a damaged file whose page
1246    /// tree disagrees with the object model's count).
1247    pub fn cells(&mut self, index: usize) -> PageParserCells {
1248        let Some(&pid) = self.pages.get(index) else {
1249            return PageParserCells::default();
1250        };
1251        let (_w, h) = page_size(&self.doc, pid);
1252        let (glyphs, inks) = page_glyphs_and_inks(&self.doc, pid, &mut self.caches);
1253        let (prose, words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1254        PageParserCells {
1255            prose,
1256            words,
1257            code: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1258            checkboxes: crate::checkbox::find(&inks, h),
1259        }
1260    }
1261
1262    /// Pages in the document's page tree.
1263    pub fn page_count(&self) -> usize {
1264        self.pages.len()
1265    }
1266
1267    /// The 0-based page `index` as the text-layer-only conversion sees it —
1268    /// one element of [`pdf_text_pages`], parsed on demand so a caller can
1269    /// drop each page once it is assembled instead of holding the whole
1270    /// document's cells. A blank page (no cells) for an index the page tree
1271    /// doesn't have.
1272    pub fn text_page(&mut self, index: usize) -> crate::pdfium_backend::PdfPage {
1273        let (w, h, (glyphs, inks)) = match self.pages.get(index) {
1274            Some(&pid) => {
1275                let (w, h) = page_size(&self.doc, pid);
1276                (w, h, page_glyphs_and_inks(&self.doc, pid, &mut self.caches))
1277            }
1278            None => (0.0, 0.0, Default::default()),
1279        };
1280        let (mut prose, mut words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1281        drop_overpainted_cells(&mut prose);
1282        drop_overpainted_cells(&mut words);
1283        crate::pdfium_backend::PdfPage {
1284            #[cfg(feature = "ocr-prep")]
1285            image_layout: None,
1286            width: w,
1287            height: h,
1288            // Cells are native PDF points; there is no rendered bitmap.
1289            scale: 1.0,
1290            cells: prose,
1291            code_cells: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1292            word_cells: words,
1293            checkboxes: crate::checkbox::find(&inks, h),
1294            #[cfg(feature = "ocr-prep")]
1295            image: image::RgbImage::new(1, 1),
1296            links: Vec::new(),
1297            rotation: 0,
1298        }
1299    }
1300}
1301
1302/// Full parser text layer: prose + word + code cells per page, glyphs parsed once.
1303/// `prose`/`words` come from the docling-parse contraction ([`crate::dp_lines`]);
1304/// `code` splits only at the parser's own space glyphs (monospace keeps its
1305/// source spacing). The eager form of [`PageTextParser`].
1306pub fn pdf_all_cells(bytes: &[u8]) -> Vec<PageParserCells> {
1307    let Some(mut parser) = PageTextParser::open(bytes) else {
1308        return Vec::new();
1309    };
1310    (0..parser.pages.len()).map(|i| parser.cells(i)).collect()
1311}
1312
1313/// Whole pages for the text-layer-only conversion ([`crate::convert_text_layer`]):
1314/// the parser's prose/word/code cells plus page geometry, assembled into
1315/// [`PdfPage`]s with no rendered image and no link annotations. Everything here
1316/// is pure Rust (lopdf), so it compiles without the `ml` feature — including on
1317/// wasm32. A page the parser can't read (no text layer) comes back with empty
1318/// cells; there is no pdfium fallback on this path.
1319pub fn pdf_text_pages(bytes: &[u8]) -> Vec<crate::pdfium_backend::PdfPage> {
1320    let Some(mut parser) = PageTextParser::open(bytes) else {
1321        return Vec::new();
1322    };
1323    (0..parser.page_count())
1324        .map(|i| parser.text_page(i))
1325        .collect()
1326}
1327
1328/// Drop line cells that are *painted over each other* — glyphs used as artwork.
1329///
1330/// Some generators draw their logo with a symbol font: on the reporting
1331/// invoice, a `TeleLogo` Type1 paints the T-Mobile mark by stacking the glyphs
1332/// encoded as `"` and `==` on top of one another, and the flat text-layer
1333/// output opened with that garbage. Nothing in the font metadata gives it away
1334/// (the *text* fonts in the same file are also flagged symbolic, and the logo
1335/// font names its glyphs `quotedbl` &c.), but the geometry does: two cells with
1336/// different text where one lies inside the other on the same line is
1337/// physically impossible for prose — ink from two words never occupies the
1338/// same box. Both cells of such a pair are paint, not text.
1339///
1340/// Containment (not mere overlap) keeps this narrow: adjacent words touch but
1341/// never contain each other, and a same-text near-duplicate (double-draw faux
1342/// bold) is left alone for the sanitizer's usual handling. Applied on the
1343/// flat/browser path only — the ML pipeline's text layer is byte-pinned by the
1344/// PDF corpus, and there the layout model already sinks logo marks into
1345/// `picture` regions.
1346fn drop_overpainted_cells(cells: &mut Vec<crate::pdfium_backend::TextCell>) {
1347    let mut paint = vec![false; cells.len()];
1348    for i in 0..cells.len() {
1349        for j in 0..cells.len() {
1350            if i == j || cells[i].text == cells[j].text {
1351                continue;
1352            }
1353            let (a, b) = (&cells[i], &cells[j]);
1354            // Same line band: the vertical overlap covers most of the shorter.
1355            let vo = (a.b.min(b.b) - a.t.max(b.t)).max(0.0);
1356            if vo < 0.6 * (a.b - a.t).min(b.b - b.t) {
1357                continue;
1358            }
1359            // `a` horizontally inside `b` (with a small tolerance).
1360            let ho = (a.r.min(b.r) - a.l.max(b.l)).max(0.0);
1361            if ho >= 0.8 * (a.r - a.l) && (a.r - a.l) <= (b.r - b.l) {
1362                paint[i] = true;
1363                paint[j] = true;
1364            }
1365        }
1366    }
1367    let mut keep = paint.iter().map(|p| !p);
1368    cells.retain(|_| keep.next().unwrap());
1369}
1370
1371/// The text-state scalars inherited by a Form XObject when it is invoked via
1372/// `Do` (the PDF graphics state includes the text parameters, but not the text
1373/// matrices, which a form re-establishes inside its own `BT`/`ET`).
1374#[derive(Clone, Copy)]
1375struct TextState {
1376    tc: f64,
1377    tw: f64,
1378    th: f64,
1379    tl: f64,
1380    trise: f64,
1381    fsize: f64,
1382}
1383
1384impl TextState {
1385    const INIT: TextState = TextState {
1386        tc: 0.0,
1387        tw: 0.0,
1388        th: 1.0,
1389        tl: 0.0,
1390        trise: 0.0,
1391        fsize: 0.0,
1392    };
1393}
1394
1395/// The effective `/Resources` dictionary for a page (inline or via reference,
1396/// falling back to an inherited one from a `/Parent`).
1397fn page_res(doc: &Document, page_id: lopdf::ObjectId) -> Option<&Dictionary> {
1398    let (inline, ids) = doc.get_page_resources(page_id).ok()?;
1399    if let Some(d) = inline {
1400        return Some(d);
1401    }
1402    ids.into_iter().find_map(|id| doc.get_dictionary(id).ok())
1403}
1404
1405/// Build the code→[`Font`] map for a resources dictionary's `/Font` sub-dict,
1406/// reusing the per-document cache for fonts referenced indirectly (the common
1407/// case — the same font objects recur on every page).
1408fn fonts_from_res(
1409    doc: &Document,
1410    res: &Dictionary,
1411    caches: &mut DocCaches,
1412) -> HashMap<Vec<u8>, Arc<Font>> {
1413    let mut map = HashMap::new();
1414    let font_dict = res
1415        .get(b"Font")
1416        .ok()
1417        .and_then(|o| deref(doc, o))
1418        .and_then(|o| o.as_dict().ok());
1419    if let Some(fd) = font_dict {
1420        for (name, value) in fd.iter() {
1421            let font = match value {
1422                Object::Reference(id) => {
1423                    let key = (*id, name.clone());
1424                    if let Some(f) = caches.fonts.get(&key) {
1425                        Arc::clone(f)
1426                    } else if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1427                        let f = Arc::new(parse_font(doc, name, fdict));
1428                        caches.fonts.insert(key, Arc::clone(&f));
1429                        f
1430                    } else {
1431                        continue;
1432                    }
1433                }
1434                _ => {
1435                    if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1436                        Arc::new(parse_font(doc, name, fdict))
1437                    } else {
1438                        continue;
1439                    }
1440                }
1441            };
1442            map.insert(name.clone(), font);
1443        }
1444    }
1445    map
1446}
1447
1448/// Extract every glyph on a page as a native-coordinate [`Glyph`].
1449/// [`PageTextParser::glyph_styles`] for the given **1-based** pages of a
1450/// document, keyed by page number — a separate, on-demand pass (no
1451/// rendering), so the extraction pipeline stays byte-identical whether or not
1452/// the heading-hierarchy stage runs. Empty when lopdf cannot open the file.
1453pub(crate) fn glyph_styles(
1454    bytes: &[u8],
1455    pages: &[usize],
1456) -> HashMap<usize, Vec<crate::heading_hierarchy::GlyphStyle>> {
1457    let mut out = HashMap::new();
1458    let Some(mut parser) = PageTextParser::open(bytes) else {
1459        return out;
1460    };
1461    for &page_no in pages {
1462        if page_no == 0 {
1463            continue;
1464        }
1465        let styles = parser.glyph_styles(page_no - 1);
1466        if !styles.is_empty() {
1467            out.insert(page_no, styles);
1468        }
1469    }
1470    out
1471}
1472
1473pub(crate) fn page_glyphs(doc: &Document, page_id: lopdf::ObjectId) -> Vec<Glyph> {
1474    page_glyphs_cached(doc, page_id, &mut DocCaches::default())
1475}
1476
1477/// [`page_glyphs`] with an explicit per-document cache, so a multi-page walk
1478/// parses each font / decodes each form once instead of once per page.
1479fn page_glyphs_cached(
1480    doc: &Document,
1481    page_id: lopdf::ObjectId,
1482    caches: &mut DocCaches,
1483) -> Vec<Glyph> {
1484    page_glyphs_and_inks(doc, page_id, caches).0
1485}
1486
1487/// [`page_glyphs_cached`] plus the small painted path pieces the same walk
1488/// passed — what [`crate::checkbox::find`] looks for checkbox squares in
1489/// (#609) — in the glyphs' frame.
1490fn page_glyphs_and_inks(
1491    doc: &Document,
1492    page_id: lopdf::ObjectId,
1493    caches: &mut DocCaches,
1494) -> (Vec<Glyph>, Vec<crate::checkbox::Ink>) {
1495    let mut out = Vec::new();
1496    let mut inks = Vec::new();
1497    // lopdf 0.44: get_page_content returns the assembled content-stream bytes
1498    // directly (an empty Vec when the page has none).
1499    let content_bytes = doc.get_page_content(page_id);
1500    let Ok(content) = lopdf::content::Content::decode(&content_bytes) else {
1501        return (out, inks);
1502    };
1503    if let Some(res) = page_res(doc, page_id) {
1504        // pdfium's page matrix: user space translated so the display box's
1505        // lower-left corner is the origin (see [`PageBox`]).
1506        let pb = page_box(doc, page_id);
1507        let base = Mat {
1508            e: -(pb.l as f64),
1509            f: -(pb.b as f64),
1510            ..Mat::ID
1511        };
1512        run_content(
1513            doc,
1514            res,
1515            &content,
1516            base,
1517            TextState::INIT,
1518            0,
1519            caches,
1520            &mut out,
1521            &mut inks,
1522        );
1523        out.retain(|g| on_page(g, pb.w, pb.h));
1524    }
1525    (out, inks)
1526}
1527
1528/// Whether a glyph is on the page (#529): docling-parse keeps a character
1529/// only when its whole box lies inside the display box — the CropBox, or
1530/// the MediaBox without one — edges included, so a FrameMaker print slug
1531/// drawn beside the CropBox or a tiled page's neighbouring text beyond the
1532/// MediaBox never becomes a cell, and a line crossing the edge is cut at the
1533/// last glyph that fits (`Crossing the rig`). The box is docling-parse's
1534/// char box — the advance by the font's ascent/descent, the loose box here
1535/// (its axis-aligned extent for rotated text) — in the frame `page_glyphs`
1536/// already moved to the display box's corner, so the page is `[0, w] × [0,
1537/// h]`. A glyph without a finite box is kept, as before. (A standard-14 font
1538/// without a FontDescriptor gets the 750 / −250 default ascent / descent
1539/// here where docling-parse reads the AFM's — Helvetica's 718 / −207 — so
1540/// for those a glyph within ~0.04 em of the top or bottom edge can fall the
1541/// other way; horizontally the advance is the same.)
1542fn on_page(g: &Glyph, w: f32, h: f32) -> bool {
1543    // f32 noise only: docling-parse already drops a glyph 0.01 pt over.
1544    const EPS: f32 = 1e-3;
1545    let (l, b, r, t) = if [g.ll, g.lb, g.lr, g.lt].iter().all(|v| v.is_finite()) {
1546        (g.ll, g.lb, g.lr, g.lt)
1547    } else {
1548        (g.l, g.b, g.r, g.t)
1549    };
1550    if ![l, b, r, t].iter().all(|v| v.is_finite()) {
1551        return true;
1552    }
1553    l >= -EPS && b >= -EPS && r <= w + EPS && t <= h + EPS
1554}
1555
1556/// Run a content stream's operators, emitting glyphs into `out`. Recurses into
1557/// Form XObjects on `Do` (bulk body text in heavy PDFs lives inside a form, not
1558/// the page content stream). `res` is the resources dict in scope (the page's,
1559/// or the form's own); `base_ctm` is the CTM at the point of invocation.
1560#[allow(clippy::too_many_arguments)]
1561fn run_content(
1562    doc: &Document,
1563    res: &Dictionary,
1564    content: &lopdf::content::Content,
1565    base_ctm: Mat,
1566    init: TextState,
1567    depth: u32,
1568    caches: &mut DocCaches,
1569    out: &mut Vec<Glyph>,
1570    inks: &mut Vec<crate::checkbox::Ink>,
1571) {
1572    let fonts = fonts_from_res(doc, res, caches);
1573    let xobjects = res
1574        .get(b"XObject")
1575        .ok()
1576        .and_then(|o| deref(doc, o))
1577        .and_then(|o| o.as_dict().ok());
1578
1579    // Graphics + text state. `q`/`Q` save and restore the whole graphics state,
1580    // which includes the text parameters (Tc, Tw, Tz, TL, Tfs, Trise, font) —
1581    // *not* the text matrix (that is reset by BT). Saving only the CTM let a Tc
1582    // set inside a `q…Q` block leak out and drift every later glyph.
1583    #[allow(clippy::type_complexity)]
1584    let mut gstate_stack: Vec<(Mat, f64, f64, f64, f64, f64, f64, Option<&Arc<Font>>)> = Vec::new();
1585    let mut ctm = base_ctm;
1586    let mut tm = Mat::ID;
1587    let mut tlm = Mat::ID;
1588    let mut font: Option<&Arc<Font>> = None;
1589    let mut fsize = init.fsize;
1590    let mut tc = init.tc; // char spacing
1591    let mut tw = init.tw; // word spacing
1592    let mut th = init.th; // horizontal scale (Tz/100)
1593    let mut tl = init.tl; // leading
1594    let mut trise = init.trise;
1595
1596    let op_f = |operands: &[Object], i: usize| operands.get(i).and_then(num).unwrap_or(0.0);
1597    // The path under construction, already in page space (a path is built
1598    // under one CTM — `cm` is not allowed inside it), for the checkbox
1599    // squares (#609). Text extraction never reads it.
1600    let mut path: Vec<PathPiece> = Vec::new();
1601    let (mut cur, mut start) = ((0.0, 0.0), (0.0, 0.0));
1602    // Whether the fill colour is (near) white, saved/restored with `q`/`Q`:
1603    // a white-filled stroked box is a checkbox outline, a coloured one a
1604    // chart legend's swatch. The initial fill colour is black.
1605    let mut fill_light = false;
1606    let mut fill_stack: Vec<bool> = Vec::new();
1607
1608    for op in &content.operations {
1609        let operands = &op.operands;
1610        match op.operator.as_str() {
1611            "q" => {
1612                gstate_stack.push((ctm, tc, tw, th, tl, trise, fsize, font));
1613                fill_stack.push(fill_light);
1614            }
1615            "Q" => {
1616                fill_light = fill_stack.pop().unwrap_or(false);
1617                if let Some((c, a, b, h, l, r, fs, f)) = gstate_stack.pop() {
1618                    ctm = c;
1619                    tc = a;
1620                    tw = b;
1621                    th = h;
1622                    tl = l;
1623                    trise = r;
1624                    fsize = fs;
1625                    font = f;
1626                }
1627            }
1628            "cm" => {
1629                let m = Mat {
1630                    a: op_f(operands, 0),
1631                    b: op_f(operands, 1),
1632                    c: op_f(operands, 2),
1633                    d: op_f(operands, 3),
1634                    e: op_f(operands, 4),
1635                    f: op_f(operands, 5),
1636                };
1637                ctm = m.then(ctm);
1638            }
1639            "BT" => {
1640                tm = Mat::ID;
1641                tlm = Mat::ID;
1642            }
1643            "ET" => {}
1644            "Tf" => {
1645                if let Some(Object::Name(n)) = operands.first() {
1646                    font = fonts.get(n.as_slice());
1647                }
1648                fsize = op_f(operands, 1);
1649            }
1650            "Td" => {
1651                tlm = Mat {
1652                    a: 1.0,
1653                    b: 0.0,
1654                    c: 0.0,
1655                    d: 1.0,
1656                    e: op_f(operands, 0),
1657                    f: op_f(operands, 1),
1658                }
1659                .then(tlm);
1660                tm = tlm;
1661            }
1662            "TD" => {
1663                tl = -op_f(operands, 1);
1664                tlm = Mat {
1665                    a: 1.0,
1666                    b: 0.0,
1667                    c: 0.0,
1668                    d: 1.0,
1669                    e: op_f(operands, 0),
1670                    f: op_f(operands, 1),
1671                }
1672                .then(tlm);
1673                tm = tlm;
1674            }
1675            "Tm" => {
1676                tlm = Mat {
1677                    a: op_f(operands, 0),
1678                    b: op_f(operands, 1),
1679                    c: op_f(operands, 2),
1680                    d: op_f(operands, 3),
1681                    e: op_f(operands, 4),
1682                    f: op_f(operands, 5),
1683                };
1684                tm = tlm;
1685            }
1686            "T*" => {
1687                tlm = Mat {
1688                    a: 1.0,
1689                    b: 0.0,
1690                    c: 0.0,
1691                    d: 1.0,
1692                    e: 0.0,
1693                    f: -tl,
1694                }
1695                .then(tlm);
1696                tm = tlm;
1697            }
1698            "Tc" => tc = op_f(operands, 0),
1699            "Tw" => tw = op_f(operands, 0),
1700            "Tz" => th = op_f(operands, 0) / 100.0,
1701            "TL" => tl = op_f(operands, 0),
1702            "Ts" => trise = op_f(operands, 0),
1703            "Tj" | "'" | "\"" => {
1704                if op.operator == "'" || op.operator == "\"" {
1705                    // move to next line first
1706                    tlm = Mat {
1707                        a: 1.0,
1708                        b: 0.0,
1709                        c: 0.0,
1710                        d: 1.0,
1711                        e: 0.0,
1712                        f: -tl,
1713                    }
1714                    .then(tlm);
1715                    tm = tlm;
1716                }
1717                if op.operator == "\"" {
1718                    // `aw ac string "` sets word- and char-spacing before
1719                    // showing the string (PDF 32000-1 §9.4.3), persisting after.
1720                    tw = op_f(operands, 0);
1721                    tc = op_f(operands, 1);
1722                }
1723                if let (Some(f), Some(Object::String(s, _))) = (font, operands.last()) {
1724                    show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out);
1725                }
1726            }
1727            "TJ" => {
1728                if let (Some(f), Some(Object::Array(arr))) = (font, operands.first()) {
1729                    for el in arr {
1730                        match el {
1731                            Object::String(s, _) => {
1732                                show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out)
1733                            }
1734                            other => {
1735                                if let Some(adj) = num(other) {
1736                                    // negative number moves text right (PDF: subtract)
1737                                    let tx = -adj / 1000.0 * fsize * th;
1738                                    tm = Mat {
1739                                        a: 1.0,
1740                                        b: 0.0,
1741                                        c: 0.0,
1742                                        d: 1.0,
1743                                        e: tx,
1744                                        f: 0.0,
1745                                    }
1746                                    .then(tm);
1747                                }
1748                            }
1749                        }
1750                    }
1751                }
1752            }
1753            "m" => {
1754                cur = ctm.apply(op_f(operands, 0), op_f(operands, 1));
1755                start = cur;
1756            }
1757            "l" => {
1758                let p = ctm.apply(op_f(operands, 0), op_f(operands, 1));
1759                path.push(PathPiece::Line(cur, p));
1760                cur = p;
1761            }
1762            "c" | "v" | "y" => {
1763                let pts: Vec<(f64, f64)> = (0..operands.len() / 2)
1764                    .map(|k| ctm.apply(op_f(operands, 2 * k), op_f(operands, 2 * k + 1)))
1765                    .collect();
1766                if let Some(&end) = pts.last() {
1767                    path.push(PathPiece::Curve(
1768                        std::iter::once(cur).chain(pts.iter().copied()).collect(),
1769                    ));
1770                    cur = end;
1771                }
1772            }
1773            "h" => {
1774                if cur != start {
1775                    path.push(PathPiece::Line(cur, start));
1776                }
1777                cur = start;
1778            }
1779            "re" => {
1780                let (x, y, w, h) = (
1781                    op_f(operands, 0),
1782                    op_f(operands, 1),
1783                    op_f(operands, 2),
1784                    op_f(operands, 3),
1785                );
1786                path.push(PathPiece::Rect([
1787                    ctm.apply(x, y),
1788                    ctm.apply(x + w, y),
1789                    ctm.apply(x + w, y + h),
1790                    ctm.apply(x, y + h),
1791                ]));
1792                cur = ctm.apply(x, y);
1793                start = cur;
1794            }
1795            "S" | "s" | "f" | "F" | "f*" | "B" | "B*" | "b" | "b*" => {
1796                let op = op.operator.as_str();
1797                if matches!(op, "s" | "b" | "b*") && cur != start {
1798                    path.push(PathPiece::Line(cur, start));
1799                }
1800                let stroke = matches!(op, "S" | "s" | "B" | "B*" | "b" | "b*");
1801                let fill = !matches!(op, "S" | "s");
1802                paint_path(&path, stroke, fill, fill_light, inks);
1803                path.clear();
1804            }
1805            "n" => path.clear(),
1806            // The fill colour, by its components: gray `g`, `rg`, CMYK `k`,
1807            // and `sc`/`scn` by operand count (a pattern name is not light).
1808            "g" | "rg" | "k" | "sc" | "scn" => {
1809                let v: Vec<f64> = operands
1810                    .iter()
1811                    .map(num)
1812                    .collect::<Option<_>>()
1813                    .unwrap_or_default();
1814                fill_light = match v.as_slice() {
1815                    [g] => *g >= 0.9,
1816                    [r, g, b] => r.min(*g).min(*b) >= 0.9,
1817                    [c, m, y, k] => c.max(*m).max(*y).max(*k) <= 0.1,
1818                    _ => false,
1819                };
1820            }
1821            "cs" => fill_light = false,
1822            "Do" => {
1823                // Invoke a Form XObject: bulk body text in many PDFs lives inside
1824                // a form, reached only here. Image XObjects are skipped (no text).
1825                if depth >= 8 {
1826                    continue;
1827                }
1828                let Some(Object::Name(n)) = operands.first() else {
1829                    continue;
1830                };
1831                let obj = xobjects.and_then(|d| d.get(n.as_slice()).ok());
1832                let form_id = match obj {
1833                    Some(Object::Reference(id)) => Some(*id),
1834                    _ => None,
1835                };
1836                let stream = obj
1837                    .and_then(|o| deref(doc, o))
1838                    .and_then(|o| o.as_stream().ok());
1839                let Some(stream) = stream else { continue };
1840                let is_form = stream
1841                    .dict
1842                    .get(b"Subtype")
1843                    .ok()
1844                    .and_then(|o| o.as_name().ok())
1845                    == Some(b"Form".as_slice());
1846                if !is_form {
1847                    continue;
1848                }
1849                // Decode the form's content once per document (headers/footers
1850                // and bulk body text invoke the same form on every page).
1851                let cached = form_id.and_then(|id| caches.forms.get(&id).cloned());
1852                let form_content = match cached {
1853                    Some(c) => c,
1854                    None => {
1855                        let Ok(data) = stream.decompressed_content() else {
1856                            continue;
1857                        };
1858                        let Ok(c) = lopdf::content::Content::decode(&data) else {
1859                            continue;
1860                        };
1861                        let c = Arc::new(c);
1862                        if let Some(id) = form_id {
1863                            caches.forms.insert(id, Arc::clone(&c));
1864                        }
1865                        c
1866                    }
1867                };
1868                // The form's /Matrix maps form space into the CTM at invocation.
1869                let form_mat = match stream.dict.get(b"Matrix").ok() {
1870                    Some(Object::Array(a)) if a.len() == 6 => {
1871                        let v: Vec<f64> = a.iter().filter_map(num).collect();
1872                        if v.len() == 6 {
1873                            Mat {
1874                                a: v[0],
1875                                b: v[1],
1876                                c: v[2],
1877                                d: v[3],
1878                                e: v[4],
1879                                f: v[5],
1880                            }
1881                        } else {
1882                            Mat::ID
1883                        }
1884                    }
1885                    _ => Mat::ID,
1886                };
1887                // The form's own /Resources, falling back to the inherited ones.
1888                let form_res = stream
1889                    .dict
1890                    .get(b"Resources")
1891                    .ok()
1892                    .and_then(|o| deref(doc, o))
1893                    .and_then(|o| o.as_dict().ok())
1894                    .unwrap_or(res);
1895                let state = TextState {
1896                    tc,
1897                    tw,
1898                    th,
1899                    tl,
1900                    trise,
1901                    fsize,
1902                };
1903                run_content(
1904                    doc,
1905                    form_res,
1906                    &form_content,
1907                    form_mat.then(ctm),
1908                    state,
1909                    depth + 1,
1910                    caches,
1911                    out,
1912                    inks,
1913                );
1914            }
1915            _ => {}
1916        }
1917    }
1918}
1919
1920/// One piece of a path under construction, page space (see `run_content`).
1921enum PathPiece {
1922    Line((f64, f64), (f64, f64)),
1923    /// An `re`: its four corners, counter-clockwise from `(x, y)`.
1924    Rect([(f64, f64); 4]),
1925    /// A Bézier segment: its start and control/end points (their hull).
1926    Curve(Vec<(f64, f64)>),
1927}
1928
1929/// Record a painted path's pieces for the checkbox finder (#609): a stroked
1930/// line is a [`Seg`](crate::checkbox::Ink::Seg), a stroked rectangle that
1931/// stays axis-aligned a [`Rect`](crate::checkbox::Ink::Rect) (one filled
1932/// white too — still an outline), and a coloured fill or a curve only its
1933/// bounding box, a [`Blot`](crate::checkbox::Ink::Blot): a possible tick
1934/// mark, or a swatch. White fills draw nothing and are dropped. Only pieces small
1935/// enough to be a checkbox edge or a mark inside one are kept.
1936fn paint_path(
1937    path: &[PathPiece],
1938    stroke: bool,
1939    fill: bool,
1940    fill_light: bool,
1941    inks: &mut Vec<crate::checkbox::Ink>,
1942) {
1943    use crate::checkbox::Ink;
1944    let bbox = |pts: &mut dyn Iterator<Item = (f64, f64)>| {
1945        pts.fold(
1946            (
1947                f64::INFINITY,
1948                f64::INFINITY,
1949                f64::NEG_INFINITY,
1950                f64::NEG_INFINITY,
1951            ),
1952            |(l, b, r, t), (x, y)| (l.min(x), b.min(y), r.max(x), t.max(y)),
1953        )
1954    };
1955    let mut keep = |ink: Ink| {
1956        if ink.worth_keeping() {
1957            inks.push(ink);
1958        }
1959    };
1960    if fill && !fill_light {
1961        // A coloured fill is a blot — a tick mark, or (the size of a square)
1962        // a legend swatch, which `checkbox::find` then rules out. A white
1963        // fill is no ink on paper: it only leaves the stroke, if any, below.
1964        let (l, b, r, t) = bbox(&mut path.iter().flat_map(|p| match p {
1965            PathPiece::Line(a, z) => vec![*a, *z],
1966            PathPiece::Rect(c) => c.to_vec(),
1967            PathPiece::Curve(c) => c.clone(),
1968        }));
1969        if l.is_finite() {
1970            keep(Ink::Blot { l, b, r, t });
1971        }
1972        return;
1973    }
1974    if !stroke {
1975        return;
1976    }
1977    for piece in path {
1978        match piece {
1979            PathPiece::Line(a, z) => keep(Ink::Seg {
1980                x0: a.0,
1981                y0: a.1,
1982                x1: z.0,
1983                y1: z.1,
1984            }),
1985            PathPiece::Rect(c) => {
1986                let axis = ((c[0].1 - c[1].1).abs() < 1e-6 && (c[1].0 - c[2].0).abs() < 1e-6)
1987                    || ((c[0].0 - c[1].0).abs() < 1e-6 && (c[1].1 - c[2].1).abs() < 1e-6);
1988                if axis {
1989                    let (l, b, r, t) = bbox(&mut c.iter().copied());
1990                    keep(Ink::Rect { l, b, r, t });
1991                } else {
1992                    for k in 0..4 {
1993                        let (a, z) = (c[k], c[(k + 1) % 4]);
1994                        keep(Ink::Seg {
1995                            x0: a.0,
1996                            y0: a.1,
1997                            x1: z.0,
1998                            y1: z.1,
1999                        });
2000                    }
2001                }
2002            }
2003            PathPiece::Curve(c) => {
2004                let (l, b, r, t) = bbox(&mut c.iter().copied());
2005                keep(Ink::Blot { l, b, r, t });
2006            }
2007        }
2008    }
2009}
2010
2011#[allow(clippy::too_many_arguments)]
2012fn show_text(
2013    font: &Font,
2014    bytes: &[u8],
2015    fsize: f64,
2016    tc: f64,
2017    tw: f64,
2018    th: f64,
2019    trise: f64,
2020    tm: &mut Mat,
2021    ctm: Mat,
2022    out: &mut Vec<Glyph>,
2023) {
2024    for code in codes(font, bytes) {
2025        let (text, w) = font.decode_code(code);
2026        let w0 = w / 1000.0; // advance in text-space (em) units
2027                             // The glyph→user transform: scale glyph space by font size, then Tm, CTM.
2028        let scale = Mat {
2029            a: fsize * th,
2030            b: 0.0,
2031            c: 0.0,
2032            d: fsize,
2033            e: 0.0,
2034            f: trise,
2035        };
2036        let trm = scale.then(*tm).then(ctm);
2037        // Box in glyph space (1000-unit em): x 0..w, y descent..ascent.
2038        let (desc, asc) = (font.descent / 1000.0, font.ascent / 1000.0);
2039        let (x0, y0) = trm.apply(0.0, desc);
2040        let (x1, y1) = trm.apply(w0, desc);
2041        let (x2, y2) = trm.apply(w0, asc);
2042        let (x3, y3) = trm.apply(0.0, asc);
2043        // Upright = the baseline runs left-to-right along +x. Only then is the
2044        // loose box the rectangle spanned by the baseline's x and the em's y
2045        // (a synthetic oblique's skew is ignored, as it always was). A rotated
2046        // matrix — the `0 s -s 0 tx ty Tm` landscape pages are built with
2047        // (#528) — collapsed that rectangle to zero width, and every such run
2048        // vanished; those glyphs keep their real quad (docling-parse's char
2049        // rect) and its extent instead.
2050        let upright = trm.a > 0.0 && trm.b.abs() <= 1e-6 * trm.a;
2051        let (left, bot, right, top, quad) = if upright {
2052            (x0.min(x1), y0.min(y3), x0.max(x1), y0.max(y3), None)
2053        } else {
2054            let (xs, ys) = ([x0, x1, x2, x3], [y0, y1, y2, y3]);
2055            let fold = |v: [f64; 4], f: fn(f64, f64) -> f64| v.into_iter().reduce(f).unwrap();
2056            let q = [x0, y0, x1, y1, x2, y2, x3, y3].map(|v| v as f32);
2057            (
2058                fold(xs, f64::min),
2059                fold(ys, f64::min),
2060                fold(xs, f64::max),
2061                fold(ys, f64::max),
2062                Some(q),
2063            )
2064        };
2065        if let Some(s) = text {
2066            // A run may map one code to multiple chars (ligature/fraction); share box.
2067            for ch in s.chars() {
2068                if ch != '\u{0}' {
2069                    out.push(Glyph {
2070                        ch,
2071                        l: left as f32,
2072                        b: bot as f32,
2073                        r: right as f32,
2074                        t: top as f32,
2075                        ll: left as f32,
2076                        lb: bot as f32,
2077                        lr: right as f32,
2078                        lt: top as f32,
2079                        font: font.hash,
2080                        quad,
2081                    });
2082                }
2083            }
2084        }
2085        // Advance the text matrix. Word spacing applies to single-byte code 32.
2086        let is_space = !font.two_byte && code == 32;
2087        let tx = (w0 * fsize + tc + if is_space { tw } else { 0.0 }) * th;
2088        *tm = Mat {
2089            a: 1.0,
2090            b: 0.0,
2091            c: 0.0,
2092            d: 1.0,
2093            e: tx,
2094            f: 0.0,
2095        }
2096        .then(*tm);
2097    }
2098}
2099
2100/// Build a simple font's code→char table from its `/Encoding`: the base
2101/// encoding (WinAnsi / MacRoman) plus any `/Differences` overrides (glyph names
2102/// resolved through a small Adobe-glyph-name subset).
2103fn simple_encoding_table(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
2104    let enc = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o));
2105    let base_name = match enc {
2106        Some(Object::Name(n)) => n.clone(),
2107        Some(Object::Dictionary(d)) => d
2108            .get(b"BaseEncoding")
2109            .ok()
2110            .and_then(|o| o.as_name().ok())
2111            .map(|n| n.to_vec())
2112            .unwrap_or_default(),
2113        _ => Vec::new(),
2114    };
2115    let mut m = if base_name == b"MacRomanEncoding" {
2116        macroman_table()
2117    } else if base_name.is_empty() {
2118        // No PDF /Encoding at all: the font's *built-in* encoding applies. For
2119        // the standard TeX math fonts that is their fixed TeX layout — falling
2120        // back to StandardEncoding read CMSY's braces as `f`/`g`, `→` as `!`,
2121        // `∈` as `2` (2203's `{ahn,…}` author line). The font program (often
2122        // CFF, which this parser does not read) carries the same mapping;
2123        // docling-parse decodes it from there.
2124        tex_math_builtin(fdict).unwrap_or_else(winansi_table)
2125    } else {
2126        winansi_table()
2127    };
2128    // Apply /Differences: `code /glyphname /glyphname ... code ...`.
2129    if let Some(Object::Dictionary(d)) = enc {
2130        if let Some(Object::Array(diffs)) = d.get(b"Differences").ok().and_then(|o| deref(doc, o)) {
2131            let mut code = 0u8;
2132            for el in diffs {
2133                match el {
2134                    Object::Integer(i) => code = *i as u8,
2135                    Object::Name(name) => {
2136                        if let Some(ch) = glyph_name_to_char(name) {
2137                            m.insert(code, ch);
2138                        }
2139                        code = code.wrapping_add(1);
2140                    }
2141                    _ => {}
2142                }
2143            }
2144        }
2145    }
2146    m
2147}
2148
2149/// The fixed built-in encodings of the standard TeX math fonts (TeXbook
2150/// Appendix F), keyed off the base font name: `CMSY*` (symbols; `CMBSY` is its
2151/// bold) and `CMMI*` (math italic). These fonts ship no PDF `/Encoding` and no
2152/// ToUnicode, and their program is usually CFF — without this table the codes
2153/// fell through to StandardEncoding and rendered as the wrong ASCII.
2154fn tex_math_builtin(fdict: &Dictionary) -> Option<HashMap<u8, char>> {
2155    const CMSY: [char; 128] = [
2156        '−', '·', '×', '∗', '÷', '⋄', '±', '∓', '⊕', '⊖', '⊗', '⊘', '⊙', '◯', '∘', '•', '≍', '≡',
2157        '⊆', '⊇', '≤', '≥', '≼', '≽', '∼', '≈', '⊂', '⊃', '≪', '≫', '≺', '≻', '←', '→', '↑', '↓',
2158        '↔', '↗', '↘', '≃', '⇐', '⇒', '⇑', '⇓', '⇔', '↖', '↙', '∝', '′', '∞', '∈', '∋', '△', '▽',
2159        '\u{338}', '↦', '∀', '∃', '¬', '∅', 'ℜ', 'ℑ', '⊤', '⊥', 'ℵ', 'A', 'B', 'C', 'D', 'E', 'F',
2160        'G', 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X',
2161        'Y', 'Z', '∪', '∩', '⊎', '∧', '∨', '⊢', '⊣', '⌊', '⌋', '⌈', '⌉', '{', '}', '⟨', '⟩', '|',
2162        '∥', '↕', '⇕', '\\', '≀', '√', '∐', '∇', '∫', '⊔', '⊓', '⊑', '⊒', '§', '†', '‡', '¶', '♣',
2163        '♢', '♡', '♠',
2164    ];
2165    const CMMI: [char; 128] = [
2166        'Γ', 'Δ', 'Θ', 'Λ', 'Ξ', 'Π', 'Σ', 'Υ', 'Φ', 'Ψ', 'Ω', 'α', 'β', 'γ', 'δ', 'ε', 'ζ', 'η',
2167        'θ', 'ι', 'κ', 'λ', 'μ', 'ν', 'ξ', 'π', 'ρ', 'σ', 'τ', 'υ', 'φ', 'χ', 'ψ', 'ω', 'ϵ', 'ϑ',
2168        'ϖ', 'ϱ', 'ς', 'ϕ', '↼', '↽', '⇀', '⇁', '↩', '↪', '▷', '◁', '0', '1', '2', '3', '4', '5',
2169        '6', '7', '8', '9', '.', ',', '<', '/', '>', '⋆', '∂', 'A', 'B', 'C', 'D', 'E', 'F', 'G',
2170        'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y',
2171        'Z', '♭', '♮', '♯', '⌣', '⌢', 'ℓ', 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k',
2172        'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', 'ı', 'ȷ', '℘',
2173        '\u{20d7}', '⁀',
2174    ];
2175    let name = base_font_name(fdict)?;
2176    let up = name.to_ascii_uppercase();
2177    let table: &[char; 128] = if up.starts_with(b"CMSY") || up.starts_with(b"CMBSY") {
2178        &CMSY
2179    } else if up.starts_with(b"CMMI") {
2180        &CMMI
2181    } else {
2182        return None;
2183    };
2184    Some(
2185        table
2186            .iter()
2187            .enumerate()
2188            .map(|(i, &c)| (i as u8, c))
2189            .collect(),
2190    )
2191}
2192
2193/// Resolve an Adobe glyph name to Unicode: `uniXXXX`, single ASCII letters, the
2194/// digit/punctuation names from the Adobe Glyph List, and common typographic
2195/// names. A `.suffix` (`one.taboldstyle`, `a.sc`) is stripped and the base name
2196/// retried — docling renders these as the base character.
2197pub(crate) fn glyph_name_to_char(name: &[u8]) -> Option<char> {
2198    let s = std::str::from_utf8(name).ok()?;
2199    if let Some(hex) = s.strip_prefix("uni") {
2200        if let Ok(cp) = u32::from_str_radix(hex.get(0..4)?, 16) {
2201            return char::from_u32(cp);
2202        }
2203    }
2204    // Single ASCII letter names (`A`, `m`) map to themselves.
2205    if s.len() == 1 {
2206        let b = s.as_bytes()[0];
2207        if b.is_ascii_alphabetic() {
2208            return Some(b as char);
2209        }
2210    }
2211    let resolved = match s {
2212        "space" => ' ',
2213        "exclam" => '!',
2214        "quotedbl" => '"',
2215        "numbersign" => '#',
2216        "dollar" => '$',
2217        "percent" => '%',
2218        "ampersand" => '&',
2219        "quotesingle" => '\'',
2220        "parenleft" => '(',
2221        "parenright" => ')',
2222        "asterisk" => '*',
2223        "plus" => '+',
2224        "comma" => ',',
2225        "hyphen" => '-',
2226        "period" => '.',
2227        "slash" => '/',
2228        "zero" => '0',
2229        "one" => '1',
2230        "two" => '2',
2231        "three" => '3',
2232        "four" => '4',
2233        "five" => '5',
2234        "six" => '6',
2235        "seven" => '7',
2236        "eight" => '8',
2237        "nine" => '9',
2238        "colon" => ':',
2239        "semicolon" => ';',
2240        "less" => '<',
2241        "equal" => '=',
2242        "greater" => '>',
2243        "question" => '?',
2244        "at" => '@',
2245        "bracketleft" => '[',
2246        "backslash" => '\\',
2247        "bracketright" => ']',
2248        "asciicircum" => '^',
2249        "underscore" => '_',
2250        "grave" => '`',
2251        "braceleft" => '{',
2252        "bar" => '|',
2253        "braceright" => '}',
2254        "asciitilde" => '~',
2255        "bullet" => '\u{2022}',
2256        "periodcentered" => '\u{00B7}',
2257        "endash" => '\u{2013}',
2258        "emdash" => '\u{2014}',
2259        "quoteright" => '\u{2019}',
2260        "quoteleft" => '\u{2018}',
2261        "quotedblleft" => '\u{201C}',
2262        "quotedblright" => '\u{201D}',
2263        "quotedblbase" => '\u{201E}',
2264        "quotesinglbase" => '\u{201A}',
2265        // Latin f-ligatures named in `/Differences` (e.g. 2305's body font). These
2266        // map to the presentation-form code points, which `decompose_ligatures`
2267        // then spells back out (`ff`→"ff") — without them the glyph decodes to
2268        // nothing and the sanitizer fills the gap with a space (`di erences`).
2269        "ff" => '\u{FB00}',
2270        "fi" => '\u{FB01}',
2271        "fl" => '\u{FB02}',
2272        "ffi" => '\u{FB03}',
2273        "ffl" => '\u{FB04}',
2274        "ft" => '\u{FB05}',
2275        "st" => '\u{FB06}',
2276        "degree" => '\u{00B0}',
2277        "trademark" => '\u{2122}',
2278        "registered" => '\u{00AE}',
2279        "copyright" => '\u{00A9}',
2280        "ellipsis" => '\u{2026}',
2281        "minus" => '\u{2212}',
2282        "fraction" => '\u{2044}',
2283        "nbspace" => '\u{00A0}',
2284        // Greek + math glyph names (standard Adobe Glyph List). Standard TeX math
2285        // fonts (CMMI/CMSY/…) name their glyphs this way in the embedded font
2286        // program's `/Encoding`; without these a `λ`/`≤` decodes to nothing and is
2287        // dropped from body text (`and λ set to 0.5` → `and set to 0.5`).
2288        "alpha" => '\u{03B1}',
2289        "beta" => '\u{03B2}',
2290        "gamma" => '\u{03B3}',
2291        "delta" => '\u{03B4}',
2292        "epsilon" | "epsilon1" => '\u{03B5}',
2293        "zeta" => '\u{03B6}',
2294        "eta" => '\u{03B7}',
2295        "theta" | "theta1" => '\u{03B8}',
2296        "iota" => '\u{03B9}',
2297        "kappa" => '\u{03BA}',
2298        "lambda" => '\u{03BB}',
2299        "mu" => '\u{03BC}',
2300        "nu" => '\u{03BD}',
2301        "xi" => '\u{03BE}',
2302        "omicron" => '\u{03BF}',
2303        "pi" | "pi1" => '\u{03C0}',
2304        "rho" | "rho1" => '\u{03C1}',
2305        "sigma" => '\u{03C3}',
2306        "sigma1" => '\u{03C2}',
2307        "tau" => '\u{03C4}',
2308        "upsilon" => '\u{03C5}',
2309        "phi" | "phi1" => '\u{03C6}',
2310        "chi" => '\u{03C7}',
2311        "psi" => '\u{03C8}',
2312        "omega" | "omega1" => '\u{03C9}',
2313        "Gamma" => '\u{0393}',
2314        "Delta" => '\u{0394}',
2315        "Theta" => '\u{0398}',
2316        "Lambda" => '\u{039B}',
2317        "Xi" => '\u{039E}',
2318        "Pi" => '\u{03A0}',
2319        "Sigma" => '\u{03A3}',
2320        "Upsilon" => '\u{03A5}',
2321        "Phi" => '\u{03A6}',
2322        "Psi" => '\u{03A8}',
2323        "Omega" => '\u{03A9}',
2324        "lessequal" => '\u{2264}',
2325        "greaterequal" => '\u{2265}',
2326        "notequal" => '\u{2260}',
2327        "approxequal" => '\u{2248}',
2328        "equivalence" => '\u{2261}',
2329        "element" => '\u{2208}',
2330        "plusminus" => '\u{00B1}',
2331        "multiply" => '\u{00D7}',
2332        "divide" => '\u{00F7}',
2333        "infinity" => '\u{221E}',
2334        "partialdiff" => '\u{2202}',
2335        "gradient" => '\u{2207}',
2336        "summation" => '\u{2211}',
2337        "product" => '\u{220F}',
2338        "integral" => '\u{222B}',
2339        "radical" => '\u{221A}',
2340        "proportional" => '\u{221D}',
2341        "arrowright" => '\u{2192}',
2342        "arrowleft" => '\u{2190}',
2343        "arrowup" => '\u{2191}',
2344        "arrowdown" => '\u{2193}',
2345        "arrowboth" => '\u{2194}',
2346        "arrowdblright" => '\u{21D2}',
2347        "logicaland" => '\u{2227}',
2348        "logicalor" => '\u{2228}',
2349        "intersection" => '\u{2229}',
2350        "union" => '\u{222A}',
2351        "similar" => '\u{223C}',
2352        "congruent" => '\u{2245}',
2353        "dotmath" => '\u{22C5}',
2354        "asteriskmath" => '\u{2217}',
2355        _ => {
2356            // Strip an AGL `.suffix` (oldstyle/small-cap variant) and retry.
2357            if let Some((base, _)) = s.split_once('.') {
2358                if !base.is_empty() {
2359                    return glyph_name_to_char(base.as_bytes());
2360                }
2361            }
2362            return None;
2363        }
2364    };
2365    Some(resolved)
2366}
2367
2368/// Minimal WinAnsiEncoding (Latin-1-ish) for simple fonts lacking ToUnicode.
2369fn winansi_table() -> HashMap<u8, char> {
2370    let mut m = HashMap::new();
2371    for b in 0x20u8..=0x7e {
2372        m.insert(b, b as char);
2373    }
2374    // High range: Windows-1252 printable points that differ from Latin-1.
2375    let extra: &[(u8, char)] = &[
2376        (0x91, '\u{2018}'),
2377        (0x92, '\u{2019}'),
2378        (0x93, '\u{201C}'),
2379        (0x94, '\u{201D}'),
2380        (0x95, '\u{2022}'),
2381        (0x96, '\u{2013}'),
2382        (0x97, '\u{2014}'),
2383        (0x85, '\u{2026}'),
2384        (0xA0, '\u{00A0}'),
2385    ];
2386    for &(b, c) in extra {
2387        m.insert(b, c);
2388    }
2389    for b in 0xA1u8..=0xFF {
2390        m.entry(b).or_insert(b as char);
2391    }
2392    m
2393}
2394
2395/// Minimal MacRomanEncoding: ASCII plus the high-range points our corpus hits
2396/// (notably 0xA5 = bullet, used as a list marker).
2397fn macroman_table() -> HashMap<u8, char> {
2398    let mut m = HashMap::new();
2399    for b in 0x20u8..=0x7e {
2400        m.insert(b, b as char);
2401    }
2402    let high: &[(u8, char)] = &[
2403        (0xA5, '\u{2022}'), // bullet
2404        (0xD0, '\u{2013}'), // endash
2405        (0xD1, '\u{2014}'), // emdash
2406        (0xD2, '\u{201C}'),
2407        (0xD3, '\u{201D}'),
2408        (0xD4, '\u{2018}'),
2409        (0xD5, '\u{2019}'),
2410        (0xCA, '\u{00A0}'),
2411        (0xC9, '\u{2026}'),
2412        (0xDE, '\u{FB01}'),
2413        (0xDF, '\u{FB02}'),
2414    ];
2415    for &(b, c) in high {
2416        m.insert(b, c);
2417    }
2418    m
2419}
2420
2421#[cfg(test)]
2422mod page_box_frame {
2423    use super::*;
2424
2425    /// One page, `boxes` spliced into the page dictionary verbatim, one text
2426    /// run at user-space `(x, y)`.
2427    fn pdf(boxes: &str, x: f32, y: f32) -> Vec<u8> {
2428        pdf_content(
2429            boxes,
2430            &format!("BT /F1 12 Tf {x} {y} Td (First printing) Tj ET\n"),
2431        )
2432    }
2433
2434    /// One page, `boxes` spliced into the page dictionary, `content` as its
2435    /// content stream (font `/F1` = Helvetica).
2436    fn pdf_content(boxes: &str, content: &str) -> Vec<u8> {
2437        let objs: Vec<String> = vec![
2438            "<</Type/Catalog/Pages 2 0 R>>".into(),
2439            format!("<</Type/Pages/Kids[3 0 R]/Count 1{boxes}>>"),
2440            "<</Type/Page/Parent 2 0 R/Contents 4 0 R/Resources<</Font<</F1 5 0 R>>>>>>".into(),
2441            format!("<</Length {}>>stream\n{content}endstream", content.len()),
2442            "<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".into(),
2443        ];
2444        let mut out = b"%PDF-1.4\n".to_vec();
2445        let mut offsets = Vec::new();
2446        for (i, body) in objs.iter().enumerate() {
2447            offsets.push(out.len());
2448            out.extend_from_slice(format!("{} 0 obj{body}endobj\n", i + 1).as_bytes());
2449        }
2450        let xref_at = out.len();
2451        out.extend_from_slice(
2452            format!("xref\n0 {}\n0000000000 65535 f \n", objs.len() + 1).as_bytes(),
2453        );
2454        for off in &offsets {
2455            out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2456        }
2457        out.extend_from_slice(
2458            format!(
2459                "trailer<</Size {}/Root 1 0 R>>\nstartxref\n{xref_at}\n%%EOF\n",
2460                objs.len() + 1
2461            )
2462            .as_bytes(),
2463        );
2464        out
2465    }
2466
2467    fn only_page(bytes: &[u8]) -> (PageBox, Vec<Glyph>) {
2468        let doc = load_document(bytes).expect("loads");
2469        let pid = *doc.get_pages().values().next().expect("one page");
2470        (page_box(&doc, pid), page_glyphs(&doc, pid))
2471    }
2472
2473    /// The trimmed-book-page shape: MediaBox and CropBox both away from the
2474    /// origin (inherited from `/Pages`). The display box is the CropBox, and a
2475    /// glyph's coordinates count from its lower-left corner — the same numbers
2476    /// the same text gets on a page whose CropBox *is* `[0 0 w h]`.
2477    #[test]
2478    fn glyphs_count_from_the_cropbox_corner_like_pdfium() {
2479        let (pb, shifted) = only_page(&pdf(
2480            "/MediaBox[-56.505 -58.25 576.303 723.31]/CropBox[1.095 -0.65 518.703 665.71]",
2481            37.0 + 1.095,
2482            58.0 - 0.65,
2483        ));
2484        assert!((pb.l - 1.095).abs() < 1e-3 && (pb.b + 0.65).abs() < 1e-3);
2485        assert!(
2486            (pb.w - 517.608).abs() < 1e-3 && (pb.h - 666.36).abs() < 1e-3,
2487            "{pb:?}"
2488        );
2489        let (pb0, plain) = only_page(&pdf("/MediaBox[0 0 517.608 666.36]", 37.0, 58.0));
2490        assert!((pb0.w - pb.w).abs() < 1e-3 && (pb0.h - pb.h).abs() < 1e-3);
2491        assert_eq!(shifted.len(), plain.len());
2492        assert!(!plain.is_empty());
2493        for (a, b) in shifted.iter().zip(&plain) {
2494            assert!(
2495                (a.l - b.l).abs() < 1e-3 && (a.b - b.b).abs() < 1e-3,
2496                "{:?} vs {:?}",
2497                (a.l, a.b),
2498                (b.l, b.b)
2499            );
2500        }
2501        assert!((plain[0].l - 37.0).abs() < 1e-3, "{}", plain[0].l);
2502    }
2503
2504    fn text(glyphs: &[Glyph]) -> String {
2505        glyphs.iter().map(|g| g.ch).collect()
2506    }
2507
2508    /// #529: text drawn outside the display box — a print slug beside the
2509    /// CropBox, a tiled page's neighbour beyond the MediaBox — is dropped,
2510    /// glyph by glyph like docling-parse, instead of being clamped onto the
2511    /// page edge; a line crossing the edge keeps the glyphs that fit.
2512    #[test]
2513    fn glyphs_outside_the_display_box_are_dropped() {
2514        let (_, g) = only_page(&pdf_content(
2515            "/MediaBox[0 0 400 400]/CropBox[100 100 400 400]",
2516            "BT /F1 14 Tf 120 300 Td (Visible.) Tj ET\nBT /F1 8 Tf 5 40 Td (Slug) Tj ET\n",
2517        ));
2518        assert_eq!(text(&g), "Visible.");
2519        let (_, g) = only_page(&pdf_content(
2520            "/MediaBox[0 0 300 400]",
2521            "BT /F1 14 Tf 40 300 Td (On page) Tj ET\nBT /F1 14 Tf 400 250 Td (Beyond) Tj ET\n",
2522        ));
2523        assert_eq!(text(&g), "On page");
2524        // docling-parse: "Crossing the right edge" on a 300 pt page → "Crossing the rig".
2525        let (_, g) = only_page(&pdf_content(
2526            "/MediaBox[0 0 300 400]",
2527            "BT /F1 14 Tf 200 250 Td (Crossing the right edge) Tj ET\n",
2528        ));
2529        assert_eq!(text(&g), "Crossing the rig");
2530    }
2531
2532    /// The containment test is docling-parse's: the whole char box (advance
2533    /// × the font's ascent / descent) inside the display box, edges
2534    /// included — a Helvetica `W` at 50 pt (47.2 wide) ending exactly on the
2535    /// right edge stays, 0.01 pt further it goes; likewise on the left.
2536    #[test]
2537    fn a_glyph_on_the_edge_stays_one_over_it_goes() {
2538        let w_at = |x: f32, y: f32| {
2539            let (_, g) = only_page(&pdf_content(
2540                "/MediaBox[0 0 300 400]",
2541                &format!("BT /F1 50 Tf {x} {y} Td (W) Tj ET\n"),
2542            ));
2543            text(&g)
2544        };
2545        assert_eq!(w_at(252.8, 200.0), "W");
2546        assert_eq!(w_at(252.81, 200.0), "");
2547        assert_eq!(w_at(0.0, 200.0), "W");
2548        assert_eq!(w_at(-0.01, 200.0), "");
2549        // Vertically the box is the font's ascent / descent: a baseline far
2550        // enough up keeps the descender on the page, one at 0 does not.
2551        assert_eq!(w_at(100.0, 20.0), "W");
2552        assert_eq!(w_at(100.0, 0.0), "");
2553        assert_eq!(w_at(100.0, 380.0), "");
2554    }
2555
2556    /// pdfium's fallbacks: no MediaBox → Letter; a CropBox is clipped to the
2557    /// MediaBox, and one that misses it entirely is ignored.
2558    #[test]
2559    fn page_box_follows_pdfium_fallbacks() {
2560        let (pb, _) = only_page(&pdf("", 10.0, 10.0));
2561        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 612.0, 792.0));
2562        let (pb, _) = only_page(&pdf(
2563            "/MediaBox[0 0 500 700]/CropBox[-100 100 600 900]",
2564            10.0,
2565            10.0,
2566        ));
2567        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 100.0, 500.0, 600.0));
2568        let (pb, _) = only_page(&pdf(
2569            "/MediaBox[0 0 500 700]/CropBox[800 800 900 900]",
2570            10.0,
2571            10.0,
2572        ));
2573        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 500.0, 700.0));
2574        // Reversed corners normalize.
2575        let (pb, _) = only_page(&pdf("/MediaBox[500 700 0 0]", 10.0, 10.0));
2576        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 500.0, 700.0));
2577    }
2578}
2579
2580#[cfg(test)]
2581mod xref_repair {
2582    /// Build a tiny one-page PDF whose cross-reference entries are either the
2583    /// spec's 20 bytes (`two_byte_eol`) or the 19-byte form some generators
2584    /// emit — everything else about the two files is identical.
2585    fn pdf_with_xref(two_byte_eol: bool) -> Vec<u8> {
2586        let content = b"BT /F1 12 Tf 72 700 Td (Invoice 922769430725) Tj ET\n";
2587        let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2588        let objs: Vec<Vec<u8>> = vec![
2589            b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2590            b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2591            b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2592               /Resources<</Font<</F1 5 0 R>>>>>>"
2593                .to_vec(),
2594            [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
2595            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".to_vec(),
2596        ];
2597
2598        let mut out = b"%PDF-1.4\n".to_vec();
2599        let mut offsets = Vec::new();
2600        for (i, body) in objs.iter().enumerate() {
2601            offsets.push(out.len());
2602            out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2603            out.extend_from_slice(body);
2604            out.extend_from_slice(b"endobj\n");
2605        }
2606        let xref_at = out.len();
2607        let eol: &[u8] = if two_byte_eol { b" \n" } else { b"\n" };
2608        out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2609        out.extend_from_slice(b"0000000000 65535 f");
2610        out.extend_from_slice(eol);
2611        for off in &offsets {
2612            out.extend_from_slice(format!("{off:010} 00000 n").as_bytes());
2613            out.extend_from_slice(eol);
2614        }
2615        out.extend_from_slice(
2616            format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2617        );
2618        out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2619        out
2620    }
2621
2622    /// A 19-byte cross-reference entry (a bare LF where the spec wants a
2623    /// two-byte EOL) makes lopdf reject the whole file, so a readable text
2624    /// layer used to look exactly like a scan — in the browser that meant ten
2625    /// seconds of OCR for nothing. The repair must recover *the same* parse the
2626    /// well-formed file gives.
2627    #[test]
2628    fn short_xref_entries_still_parse() {
2629        let good = pdf_with_xref(true);
2630        let broken = pdf_with_xref(false);
2631        assert!(
2632            broken.len() < good.len(),
2633            "the broken file is the shorter one"
2634        );
2635        assert!(
2636            lopdf::Document::load_mem(&good).is_ok(),
2637            "the control file must load unaided"
2638        );
2639        assert!(
2640            lopdf::Document::load_mem(&broken).is_err(),
2641            "lopdf rejects 19-byte entries — if this ever passes, drop the repair"
2642        );
2643
2644        let cells = |b: &[u8]| -> Vec<String> {
2645            super::pdf_textlines(b)
2646                .into_iter()
2647                .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2648                .collect()
2649        };
2650        let from_good = cells(&good);
2651        assert!(
2652            from_good.iter().any(|t| t.contains("922769430725")),
2653            "control text: {from_good:?}"
2654        );
2655        assert_eq!(
2656            cells(&broken),
2657            from_good,
2658            "repair must match the good parse"
2659        );
2660    }
2661
2662    /// The same generator overstates `/Length`, so lopdf reads past the data,
2663    /// misses `endstream` and drops the stream — the object comes back as a
2664    /// bare dictionary and the page has no content at all. Trust `endstream`
2665    /// instead, and do it without moving a single byte.
2666    #[test]
2667    fn overstated_stream_length_still_yields_content() {
2668        let good = pdf_with_xref(true);
2669        // Inflate the content stream's /Length by one, exactly as the invoice
2670        // that prompted this does.
2671        let broken = {
2672            let at = good
2673                .windows(8)
2674                .position(|w| w == b"/Length ")
2675                .expect("a /Length")
2676                + 8;
2677            let digits = good[at..].iter().take_while(|c| c.is_ascii_digit()).count();
2678            let n: usize = std::str::from_utf8(&good[at..at + digits])
2679                .unwrap()
2680                .parse()
2681                .unwrap();
2682            let inflated = (n + 1).to_string();
2683            assert_eq!(inflated.len(), digits, "keep the digit count");
2684            let mut b = good.clone();
2685            b[at..at + digits].copy_from_slice(inflated.as_bytes());
2686            b
2687        };
2688        assert_eq!(broken.len(), good.len(), "the defect must not move bytes");
2689        // lopdf alone loses the stream: the page parses but carries no content.
2690        let raw = lopdf::Document::load_mem(&broken).expect("still loads");
2691        assert!(
2692            raw.get_pages()
2693                .into_values()
2694                .all(|p| raw.get_page_content(p).is_empty()),
2695            "lopdf should drop the stream — if it stops, drop this repair"
2696        );
2697        // Ours recovers the same text the well-formed file gives.
2698        let text = |b: &[u8]| -> Vec<String> {
2699            super::pdf_textlines(b)
2700                .into_iter()
2701                .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2702                .collect()
2703        };
2704        let expected = text(&good);
2705        assert!(!expected.is_empty(), "control must produce text");
2706        assert_eq!(text(&broken), expected);
2707    }
2708
2709    /// The repair only fires where padding cannot move an object: it declines a
2710    /// file whose xref precedes an object (an incremental update), rather than
2711    /// shifting every offset the table records.
2712    #[test]
2713    fn repair_declines_when_padding_would_move_objects() {
2714        let mut incremental = pdf_with_xref(false);
2715        incremental.extend_from_slice(b"6 0 obj<</Type/Whatever>>endobj\n");
2716        let declined = super::pad_short_xref_entries(&incremental).unwrap_err();
2717        assert!(
2718            declined.contains("object follows the xref"),
2719            "reason: {declined}"
2720        );
2721    }
2722}
2723
2724/// #187: standard-14 fonts referenced without an embedded program (and thus
2725/// usually without `/Widths` or a `/FontDescriptor`) must decode with the
2726/// built-in Adobe Core 14 metrics instead of collapsing every cell to zero
2727/// width — the failure mode where a valid text layer was silently dropped
2728/// while pdfium read the same file fine.
2729#[cfg(test)]
2730mod base14_fonts {
2731    /// A one-page PDF whose single `Tj` uses `fontdict` (no embedded program).
2732    fn pdf_with_font(fontdict: &[u8], text: &[u8]) -> Vec<u8> {
2733        let content = [b"BT /F1 12 Tf 72 700 Td (".as_slice(), text, b") Tj ET\n"].concat();
2734        pdf_with_content(fontdict, &content)
2735    }
2736
2737    /// A one-page A4 PDF drawing `content` with `fontdict` as `/F1`.
2738    fn pdf_with_content(fontdict: &[u8], content: &[u8]) -> Vec<u8> {
2739        let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2740        let objs: Vec<Vec<u8>> = vec![
2741            b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2742            b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2743            b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2744               /Resources<</Font<</F1 5 0 R>>>>>>"
2745                .to_vec(),
2746            [stream.as_slice(), content, b"endstream"].concat(),
2747            fontdict.to_vec(),
2748        ];
2749        let mut out = b"%PDF-1.4\n".to_vec();
2750        let mut offsets = Vec::new();
2751        for (i, body) in objs.iter().enumerate() {
2752            offsets.push(out.len());
2753            out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2754            out.extend_from_slice(body);
2755            out.extend_from_slice(b"endobj\n");
2756        }
2757        let xref_at = out.len();
2758        out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2759        out.extend_from_slice(b"0000000000 65535 f \n");
2760        for off in &offsets {
2761            out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2762        }
2763        out.extend_from_slice(
2764            format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2765        );
2766        out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2767        out
2768    }
2769
2770    /// The parsed cells of the only page.
2771    fn cells(pdf: &[u8]) -> Vec<crate::pdfium_backend::TextCell> {
2772        super::pdf_textlines(pdf)
2773            .into_iter()
2774            .flat_map(|(_, _, c)| c)
2775            .collect()
2776    }
2777
2778    /// Text drawn with a rotated text matrix reads as one line in its own
2779    /// reading order, boxed by the run's axis-aligned extent (#528). The
2780    /// landscape case is `0 s -s 0 tx ty Tm` on a `/Rotate 90` page; every
2781    /// such run used to collapse to zero width and vanish (90°/270°/tilted)
2782    /// or read backwards (180°). Expected boxes are docling-parse 7.22's line
2783    /// cells for the same runs (y-up, PDF points).
2784    #[test]
2785    fn rotated_text_matrix_reads_in_order() {
2786        const HELV: &[u8] = b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>";
2787        const TEXT: &str = "Upright text on a rotated page.";
2788        for (tm, [l, b, r, t]) in [
2789            ("0 14 -14 0 200 100", [189.9, 100.0, 202.9, 289.1]), // 90°
2790            ("-14 0 0 -14 350 300", [160.9, 289.9, 350.0, 302.9]), // 180°
2791            ("0 -14 14 0 200 500", [197.1, 310.9, 210.1, 500.0]), // 270°
2792            (
2793                "9.8995 9.8995 -9.8995 9.8995 100 100",
2794                [92.9, 98.0, 235.8, 240.8],
2795            ), // 45°
2796        ] {
2797            let content = format!("BT /F1 1 Tf {tm} Tm ({TEXT}) Tj ET\n");
2798            let pdf = pdf_with_content(HELV, content.as_bytes());
2799            let cs = cells(&pdf);
2800            assert_eq!(cs.len(), 1, "{tm}: {cs:?}");
2801            let c = &cs[0];
2802            assert_eq!(c.text, TEXT, "{tm}");
2803            // `cells` is top-left-origin on the 842 pt page. The descriptor-
2804            // less base-14 face gets the parser's 1-em box where docling-parse
2805            // reads Helvetica's AFM (718/−207) — the same ≤ 0.6 pt
2806            // across-the-baseline offset upright text has — so allow 1 pt.
2807            let got = [c.l, 842.0 - c.b, c.r, 842.0 - c.t];
2808            for (g, w) in got.iter().zip([l, b, r, t]) {
2809                assert!(
2810                    (g - w).abs() < 1.0,
2811                    "{tm}: box {got:?} vs docling-parse {:?}",
2812                    [l, b, r, t]
2813                );
2814            }
2815            let words: Vec<String> = super::pdf_words(&pdf)
2816                .into_iter()
2817                .flat_map(|(_, _, c)| c)
2818                .map(|c| c.text)
2819                .collect();
2820            assert_eq!(words, TEXT.split(' ').collect::<Vec<_>>(), "{tm}");
2821        }
2822    }
2823
2824    /// A rotated run past the page edge (a cover's spine title) is dropped
2825    /// by the on-page test (#529) — measured on the quad's extent, not the
2826    /// zero-width box rotated glyphs used to get — instead of clamping onto
2827    /// the edge as a zero-width cell. One inside the page is kept.
2828    #[test]
2829    fn rotated_text_off_the_page_is_dropped() {
2830        const HELV: &[u8] = b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>";
2831        let off = pdf_with_content(
2832            HELV,
2833            b"BT /F1 1 Tf 0 14 -14 0 -4 200 Tm (Spine title) Tj ET\n",
2834        );
2835        assert!(cells(&off).is_empty(), "{:?}", cells(&off));
2836        let on = pdf_with_content(
2837            HELV,
2838            b"BT /F1 1 Tf 0 14 -14 0 30 200 Tm (Margin note) Tj ET\n",
2839        );
2840        assert_eq!(cells(&on).len(), 1);
2841    }
2842
2843    /// Upright text keeps the plain loose rectangle — no quad, and the
2844    /// sanitizer path is the one every pinned PDF baseline was made with.
2845    #[test]
2846    fn upright_glyphs_carry_no_quad() {
2847        let only_page = |pdf: &[u8]| {
2848            let doc = super::load_document(pdf).expect("loads");
2849            let pid = *doc.get_pages().values().next().expect("one page");
2850            super::page_glyphs(&doc, pid)
2851        };
2852        let pdf = pdf_with_font(b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>", b"Plain");
2853        let glyphs = only_page(&pdf);
2854        assert!(!glyphs.is_empty());
2855        assert!(glyphs.iter().all(|g| g.quad.is_none() && g.lr > g.ll));
2856        // A 90° glyph's style height is its font size, not its advance.
2857        let pdf = pdf_with_content(
2858            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>",
2859            b"BT /F1 1 Tf 0 14 -14 0 200 100 Tm (W) Tj ET\n",
2860        );
2861        let glyphs = only_page(&pdf);
2862        let g = &glyphs[0];
2863        assert!(g.quad.is_some());
2864        // 14 pt × the face's 1-em box (no descriptor: ascent − descent = 1).
2865        assert!((g.height() - 14.0).abs() < 0.05, "{}", g.height());
2866    }
2867
2868    /// #609: ReportLab's chart axis sets each tick label with the *same* `Tm`
2869    /// inside its own `q … cm … Q`, so only the CTM stacks `3`…`8` one under
2870    /// another at one x. Each tick is its own line cell with its own box (as
2871    /// docling-parse reads it), not one `345678` cell boxed like the first.
2872    #[test]
2873    fn ticks_stacked_by_cm_are_separate_line_cells() {
2874        let mut content = String::new();
2875        for (k, tick) in "345678".chars().enumerate() {
2876            let y = 300.0 + 25.92 * k as f64;
2877            content.push_str(&format!(
2878                "q 1 0 0 1 125 {y} cm BT /F1 10 Tf 1 0 0 1 -5 -4 Tm ({tick}) Tj ET Q\n"
2879            ));
2880        }
2881        let pdf = pdf_with_content(
2882            b"<</Type/Font/Subtype/Type1/BaseFont/Times-Roman>>",
2883            content.as_bytes(),
2884        );
2885        let cells = cells(&pdf);
2886        let texts: Vec<&str> = cells.iter().map(|c| c.text.as_str()).collect();
2887        assert_eq!(texts, ["3", "4", "5", "6", "7", "8"]);
2888        for (k, c) in cells.iter().enumerate() {
2889            // Top-left y of a tick whose baseline is 4 pt under its cm origin.
2890            let base = 842.0 - (300.0 + 25.92 * k as f32 - 4.0);
2891            assert!((c.l - 120.0).abs() < 0.1, "{k}: l {}", c.l);
2892            assert!(
2893                c.t < base && base - c.t < 10.0,
2894                "{k}: t {} base {base}",
2895                c.t
2896            );
2897        }
2898    }
2899
2900    /// #609: the path walk finds drawn checkbox squares — ReportLab's four
2901    /// `m … l S` edges under a `cm`, and a stroked `re` with a tick inside
2902    /// (checked) — in the text cells' top-left frame, and a stroked box
2903    /// filled white; a filled bar, a big frame and a colour-filled legend
2904    /// swatch are not checkboxes.
2905    #[test]
2906    fn drawn_checkbox_squares_are_found() {
2907        let content = b"q 1 0 0 1 119.52 449.04 cm \
2908            n 0 12.96 m 12.96 12.96 l S n 0 0 m 12.96 0 l S \
2909            n 0 0 m 0 12.96 l S n 12.96 0 m 12.96 12.96 l S Q\n\
2910            200 400 10 10 re S 202 405 m 204.5 402 l 208.5 408.5 l S\n\
2911            300 400 12 12 re f 50 50 100 100 re S\n\
2912            q 0.2 0.4 0.8 rg 400 400 12 12 re B Q q 1 g 500 400 12 12 re B Q\n";
2913        let pdf = pdf_with_content(b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>", content);
2914        let mut parser = super::PageTextParser::open(&pdf).expect("parses");
2915        let mut boxes = parser.cells(0).checkboxes;
2916        boxes.sort_by(|a, b| a.l.total_cmp(&b.l));
2917        let got: Vec<([i32; 4], bool)> = boxes
2918            .iter()
2919            .map(|c| {
2920                let r = |v: f32| (v * 100.0).round() as i32;
2921                ([r(c.l), r(c.t), r(c.r), r(c.b)], c.checked)
2922            })
2923            .collect();
2924        // Page height 842: y-up 449.04..462.0 → top-left 380.0..392.96.
2925        assert_eq!(
2926            got,
2927            [
2928                ([11952, 38000, 13248, 39296], false),
2929                ([20000, 43200, 21000, 44200], true),
2930                ([50000, 43000, 51200, 44200], false),
2931            ]
2932        );
2933    }
2934
2935    /// Every standard-14 alias/style decodes with real (positive-width) boxes.
2936    #[test]
2937    fn standard14_faces_get_builtin_widths() {
2938        for fontdict in [
2939            // ReportLab's default: base-14 Helvetica, WinAnsi, nothing else.
2940            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>".as_slice(),
2941            // No /Encoding at all (StandardEncoding-ish default).
2942            b"<</Type/Font/Subtype/Type1/BaseFont/Times-BoldItalic>>",
2943            // Substitution aliases + a subset prefix.
2944            b"<</Type/Font/Subtype/TrueType/BaseFont/Arial,Bold>>",
2945            b"<</Type/Font/Subtype/Type1/BaseFont/ABCDEF+Courier-Oblique>>",
2946        ] {
2947            let pdf = pdf_with_font(fontdict, b"Words have width now");
2948            let cs = cells(&pdf);
2949            let text: String = cs
2950                .iter()
2951                .map(|c| c.text.as_str())
2952                .collect::<Vec<_>>()
2953                .join(" ");
2954            assert!(
2955                text.contains("Words have width now"),
2956                "{}: text lost: {text:?}",
2957                String::from_utf8_lossy(fontdict)
2958            );
2959            assert!(
2960                cs.iter().all(|c| c.r > c.l),
2961                "{}: zero-width cells: {cs:?}",
2962                String::from_utf8_lossy(fontdict)
2963            );
2964        }
2965    }
2966
2967    /// An explicit `/Widths` array always wins over the built-in metrics, and a
2968    /// non-standard face without `/Widths` stays as before (no invented boxes).
2969    #[test]
2970    fn explicit_widths_win_and_unknown_faces_are_untouched() {
2971        // Helvetica with explicit 100/1000-em widths: the word's box must be
2972        // ~4×100 units at 12pt = 4.8pt wide — far narrower than the ~2.7×
2973        // wider built-in Helvetica advances would make it.
2974        let explicit = pdf_with_font(
2975            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/FirstChar 65\
2976               /Widths[100 100 100 100]/Encoding/WinAnsiEncoding>>",
2977            b"ABBA",
2978        );
2979        let builtin = pdf_with_font(
2980            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>",
2981            b"ABBA",
2982        );
2983        let w = |pdf: &[u8]| {
2984            let cs = cells(pdf);
2985            assert_eq!(cs.len(), 1, "one word cell: {cs:?}");
2986            cs[0].r - cs[0].l
2987        };
2988        let (we, wb) = (w(&explicit), w(&builtin));
2989        assert!(
2990            (we - 4.8).abs() < 0.1,
2991            "explicit widths must win: got {we}, want 4×100×12/1000"
2992        );
2993        assert!(
2994            wb > 2.0 * we,
2995            "built-in Helvetica is much wider: {wb} vs {we}"
2996        );
2997
2998        // An unknown face with no /Widths: still parses (text kept), but no
2999        // built-in table applies — the old zero-width behavior is preserved
3000        // rather than inventing Helvetica metrics for an arbitrary font.
3001        let unknown = pdf_with_font(
3002            b"<</Type/Font/Subtype/Type1/BaseFont/FancyCorp-Display>>",
3003            b"Mystery",
3004        );
3005        let cs = cells(&unknown);
3006        let text: String = cs.iter().map(|c| c.text.as_str()).collect();
3007        assert!(text.contains("Mystery"), "text still decodes: {cs:?}");
3008    }
3009}
3010
3011#[cfg(test)]
3012mod overpainted {
3013    use crate::pdfium_backend::TextCell;
3014
3015    fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
3016        TextCell {
3017            text: text.into(),
3018            l,
3019            t,
3020            r,
3021            b,
3022        }
3023    }
3024
3025    /// The reporting invoice's logo: a `"` painted inside a `==` on one band —
3026    /// artwork drawn with glyphs. Both cells go; the real text on the next
3027    /// band stays.
3028    #[test]
3029    fn stacked_logo_glyphs_are_dropped() {
3030        let mut cells = vec![
3031            cell("\"", 72.7, 21.5, 86.4, 31.5),
3032            cell("==", 59.4, 21.5, 99.6, 31.5),
3033            cell("Herr", 65.2, 151.3, 81.7, 161.3),
3034        ];
3035        super::drop_overpainted_cells(&mut cells);
3036        assert_eq!(cells.len(), 1, "cells: {cells:?}");
3037        assert_eq!(cells[0].text, "Herr");
3038    }
3039
3040    /// Adjacent words on a line touch but never contain each other — prose is
3041    /// untouched, and so is a same-text near-duplicate (double-drawn faux
3042    /// bold), which is not evidence of artwork.
3043    #[test]
3044    fn prose_and_double_draw_are_kept() {
3045        let mut cells = vec![
3046            cell("Telefon", 354.3, 133.2, 381.5, 143.2),
3047            cell("0676/2000", 387.3, 133.2, 428.7, 143.2),
3048            cell("Bold", 100.0, 50.0, 130.0, 60.0),
3049            cell("Bold", 100.3, 50.0, 130.3, 60.0),
3050        ];
3051        super::drop_overpainted_cells(&mut cells);
3052        assert_eq!(cells.len(), 4);
3053    }
3054}
3055
3056#[cfg(test)]
3057mod vestigial_layer {
3058    use crate::pdfium_backend::{PdfPage, TextCell};
3059
3060    fn page_with(texts: &[&str]) -> PdfPage {
3061        let cells = texts
3062            .iter()
3063            .enumerate()
3064            .map(|(i, t)| TextCell {
3065                text: t.to_string(),
3066                l: 10.0,
3067                t: 10.0 + 12.0 * i as f32,
3068                r: 90.0,
3069                b: 20.0 + 12.0 * i as f32,
3070            })
3071            .collect();
3072        PdfPage::from_cells(595.0, 842.0, 1.0, cells)
3073    }
3074
3075    /// The reported scanned form: three typed-in field values ("03", "05",
3076    /// "2025") over three image pages. That must read as *no usable layer*,
3077    /// so the browser routes the document to OCR instead of extracting
3078    /// thirteen characters and skipping the letter entirely.
3079    #[test]
3080    fn typed_in_form_fields_are_not_a_text_layer() {
3081        let pages = vec![
3082            page_with(&["03", "05", "2025"]),
3083            page_with(&[]),
3084            page_with(&[]),
3085        ];
3086        assert!(super::text_layer_is_vestigial(&pages));
3087        assert!(super::text_layer_is_vestigial(&[page_with(&[])]));
3088    }
3089
3090    /// A short but genuine digital document — one page, a few real lines —
3091    /// keeps the fast text path.
3092    #[test]
3093    fn sparse_but_real_documents_pass() {
3094        let one_pager = vec![page_with(&[
3095            "Confidential briefing",
3096            "Prepared for the board meeting",
3097            "Do not distribute",
3098        ])];
3099        assert!(!super::text_layer_is_vestigial(&one_pager));
3100    }
3101}