Skip to main content

docling_pdf/
textparse.rs

1//! Pure-Rust PDF text extraction (replacing pdfium's glyph layer).
2//!
3//! pdfium reports *rendered* glyph boxes, which diverge from docling's
4//! `docling-parse` C++ parser at exactly the points that drive conformance:
5//! generated spaces get a zero-width box, combining diacritics get a real-width
6//! box, and ligature/fraction glyphs land at different x. This module instead
7//! reconstructs each glyph's box from the **font's own advance widths** and the
8//! PDF text/graphics matrices — the same information docling-parse uses — so a
9//! space is as wide as the font says and a combining mark has zero advance.
10//!
11//! The output is the same [`Glyph`] stream pdfium produces (native PDF
12//! coordinates, y-up), fed straight into the existing docling-parse line
13//! sanitizer ([`crate::dp_lines`]). Only the digital text layer is handled here;
14//! pages without one still fall back to OCR upstream.
15
16use std::collections::HashMap;
17use std::sync::Arc;
18
19use lopdf::{Dictionary, Document, Object};
20
21use crate::pdfium_backend::Glyph;
22
23/// Per-document caches for the content-stream interpreter. Fonts are indirect
24/// objects shared by many pages, but were fully re-parsed — ToUnicode CMap
25/// decompression + tokenization, embedded Type1 program scan, width tables —
26/// for **every page and every Form XObject invocation**; decoded form content
27/// streams were likewise re-inflated on every `Do`. Cached per document,
28/// keyed by the referenced object id (fonts also by resource name, which
29/// feeds the docling-parse font hash). Inline (non-reference) dicts are rare
30/// and stay uncached.
31#[derive(Default)]
32struct DocCaches {
33    fonts: HashMap<(lopdf::ObjectId, Vec<u8>), Arc<Font>>,
34    forms: HashMap<lopdf::ObjectId, Arc<lopdf::content::Content>>,
35}
36
37/// A 2×3 affine matrix `[a b c d e f]`: maps `(x,y)` → `(a·x+c·y+e, b·x+d·y+f)`.
38#[derive(Clone, Copy)]
39struct Mat {
40    a: f64,
41    b: f64,
42    c: f64,
43    d: f64,
44    e: f64,
45    f: f64,
46}
47
48impl Mat {
49    const ID: Mat = Mat {
50        a: 1.0,
51        b: 0.0,
52        c: 0.0,
53        d: 1.0,
54        e: 0.0,
55        f: 0.0,
56    };
57
58    /// `self ∘ m`: the matrix that applies `self` first, then `m`.
59    fn then(self, m: Mat) -> Mat {
60        Mat {
61            a: self.a * m.a + self.b * m.c,
62            b: self.a * m.b + self.b * m.d,
63            c: self.c * m.a + self.d * m.c,
64            d: self.c * m.b + self.d * m.d,
65            e: self.e * m.a + self.f * m.c + m.e,
66            f: self.e * m.b + self.f * m.d + m.f,
67        }
68    }
69
70    fn apply(self, x: f64, y: f64) -> (f64, f64) {
71        (
72            self.a * x + self.c * y + self.e,
73            self.b * x + self.d * y + self.f,
74        )
75    }
76}
77
78/// A parsed font: how to turn raw string bytes into (unicode, advance) pairs.
79struct Font {
80    /// 2-byte codes (Type0 / Identity-H) vs 1-byte (simple fonts).
81    two_byte: bool,
82    /// code → Unicode string (from ToUnicode; may be multi-char, e.g. ligatures).
83    to_unicode: HashMap<u32, String>,
84    /// code → glyph advance, in 1000-unit glyph space.
85    widths: HashMap<u32, f64>,
86    default_width: f64,
87    /// 1-byte fallback decoding when ToUnicode lacks a code (WinAnsi-ish).
88    simple_encoding: Option<HashMap<u8, char>>,
89    /// code → raw `/Differences` glyph name, for GID-style names (`g115`) that
90    /// have no Unicode mapping. docling-parse emits these verbatim as `/g115`
91    /// (see the redp5110 bulleted list); matching it keeps no text skipped.
92    fallback_names: HashMap<u8, String>,
93    /// code → char from the embedded Type1 font program's own `/Encoding` vector,
94    /// used only as a last resort for glyphs the base encoding leaves unmapped
95    /// (standard TeX math fonts: `λ`, `≤`, …).
96    program_encoding: HashMap<u8, char>,
97    ascent: f64,
98    descent: f64,
99    hash: u64,
100    /// Weight and slant read from the `/BaseFont` name (the heading
101    /// hierarchy's style signal, #302).
102    style: crate::font_style::FontStyle,
103}
104
105impl Font {
106    fn decode_code(&self, code: u32) -> (Option<String>, f64) {
107        let w = self
108            .widths
109            .get(&code)
110            .copied()
111            .unwrap_or(self.default_width);
112        if let Some(s) = self.to_unicode.get(&code) {
113            return (Some(decompose_ligatures(s)), w);
114        }
115        if !self.two_byte {
116            // A GID-style `/Differences` name (no Unicode) overrides the base
117            // encoding, matching docling's verbatim `/g115` fallback.
118            if let Some(name) = self.fallback_names.get(&(code as u8)) {
119                return (Some(format!("/{name}")), w);
120            }
121            if let Some(enc) = &self.simple_encoding {
122                if let Some(&ch) = enc.get(&(code as u8)) {
123                    return (Some(decompose_ligatures(&ch.to_string())), w);
124                }
125            }
126            // Last resort: the embedded Type1 font program's own `/Encoding`
127            // vector (`dup N /glyphname put`). Standard TeX math fonts (CMMI, CMSY,
128            // …) ship no PDF `/Encoding` and no ToUnicode, so a glyph like `λ`
129            // (CMMI code 21 → `/lambda`) or `≤` (CMSY code 20 → `/lessequal`) has
130            // no other mapping and would otherwise be silently dropped. docling
131            // recovers these from the same font program. This only fills codes the
132            // base encoding left unmapped, so it never changes an existing decode.
133            if let Some(&ch) = self.program_encoding.get(&(code as u8)) {
134                return (Some(decompose_ligatures(&ch.to_string())), w);
135            }
136        }
137        (None, w)
138    }
139}
140
141/// Spell out Latin presentation-form ligatures (`fi`→`fi`, `ffi`→`ffi`, …) the way
142/// docling does, so `configuration`/`difficult` don't keep the ligature glyph.
143/// The chars share the ligature's box, so the line sanitizer recomposes them.
144fn decompose_ligatures(s: &str) -> String {
145    if !s.chars().any(|c| ('\u{FB00}'..='\u{FB06}').contains(&c)) {
146        return s.to_string();
147    }
148    s.chars()
149        .map(|c| {
150            match c {
151                '\u{FB00}' => "ff",
152                '\u{FB01}' => "fi",
153                '\u{FB02}' => "fl",
154                '\u{FB03}' => "ffi",
155                '\u{FB04}' => "ffl",
156                '\u{FB05}' => "ft",
157                '\u{FB06}' => "st",
158                _ => return c.to_string(),
159            }
160            .to_string()
161        })
162        .collect()
163}
164
165fn hash_name(name: &[u8]) -> u64 {
166    use std::hash::{Hash, Hasher};
167    let mut h = std::collections::hash_map::DefaultHasher::new();
168    name.hash(&mut h);
169    h.finish()
170}
171
172/// Resolve a possibly-indirect object to a dictionary.
173fn as_dict<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Dictionary> {
174    match obj {
175        Object::Dictionary(d) => Some(d),
176        Object::Reference(id) => doc.get_object(*id).ok().and_then(|o| o.as_dict().ok()),
177        _ => None,
178    }
179}
180
181fn deref<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Object> {
182    match obj {
183        Object::Reference(id) => doc.get_object(*id).ok(),
184        other => Some(other),
185    }
186}
187
188/// Parse one font dictionary into a [`Font`].
189fn parse_font(doc: &Document, name: &[u8], fdict: &Dictionary) -> Font {
190    let subtype: &[u8] = fdict
191        .get(b"Subtype")
192        .ok()
193        .and_then(|o| o.as_name().ok())
194        .unwrap_or(&[]);
195    let two_byte = subtype == b"Type0".as_slice();
196
197    let to_unicode = fdict
198        .get(b"ToUnicode")
199        .ok()
200        .and_then(|o| deref(doc, o))
201        .and_then(|o| o.as_stream().ok())
202        .and_then(|s| s.decompressed_content().ok())
203        .map(|data| parse_tounicode(&data))
204        .unwrap_or_default();
205
206    let (mut widths, mut default_width) = if two_byte {
207        cid_widths(doc, fdict)
208    } else {
209        simple_widths(doc, fdict)
210    };
211
212    let simple_encoding = if two_byte {
213        None
214    } else {
215        Some(simple_encoding_table(doc, fdict))
216    };
217
218    // A standard-14 font referenced without an embedded program usually ships
219    // no `/Widths` and no `/FontDescriptor` either (ReportLab's default, #187).
220    // Every advance then resolved to 0, the cells collapsed to zero width, and
221    // the page's whole text layer was silently dropped — while pdfium, with
222    // its built-in metrics, reads the same file fine. Fill the widths from the
223    // built-in Adobe Core 14 AFM tables via the font's own code→char decode
224    // (base encoding + `/Differences`), so an explicit `/Widths` always wins.
225    if !two_byte && widths.is_empty() && default_width == 0.0 {
226        if let Some(std14) = base_font_name(fdict).and_then(|n| crate::std14::widths_for(&n)) {
227            if let Some(enc) = &simple_encoding {
228                for (&code, &ch) in enc {
229                    if let Some(w) = std14.width(ch) {
230                        widths.insert(u32::from(code), w);
231                    }
232                }
233            }
234            // Codes the table misses still advance a typical width instead of
235            // stacking at x=0 (the failure mode this whole branch fixes).
236            default_width = 500.0;
237        }
238    }
239    let fallback_names = if two_byte {
240        HashMap::new()
241    } else {
242        differences_gid_names(doc, fdict)
243    };
244    let program_encoding = if two_byte {
245        HashMap::new()
246    } else {
247        type1_program_encoding(doc, fdict)
248    };
249
250    let (ascent, descent) = font_ascent_descent(doc, fdict, two_byte);
251
252    Font {
253        two_byte,
254        to_unicode,
255        widths,
256        default_width,
257        simple_encoding,
258        fallback_names,
259        program_encoding,
260        ascent,
261        descent,
262        hash: hash_name(name),
263        style: crate::font_style::parse_font_style(&String::from_utf8_lossy(
264            &base_font_name(fdict).unwrap_or_else(|| name.to_vec()),
265        )),
266    }
267}
268
269/// Collect `/Differences` entries whose glyph name is a GID placeholder
270/// (`g115`, `cid42`, `glyph7`, `index9`) with no Unicode mapping. docling-parse
271/// emits such glyphs as the literal name `/g115`; mapping them here keeps the
272/// text from being silently dropped (subsetted fonts with no ToUnicode). The
273/// GID-name restriction keeps real Adobe glyph names on the normal path so this
274/// never invents garbage on the clean files.
275fn differences_gid_names(doc: &Document, fdict: &Dictionary) -> HashMap<u8, String> {
276    let mut map = HashMap::new();
277    let Some(Object::Dictionary(enc)) = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o))
278    else {
279        return map;
280    };
281    let Some(Object::Array(diffs)) = enc.get(b"Differences").ok().and_then(|o| deref(doc, o))
282    else {
283        return map;
284    };
285    let mut code = 0u8;
286    for el in diffs {
287        match el {
288            Object::Integer(i) => code = *i as u8,
289            Object::Name(name) => {
290                if glyph_name_to_char(name).is_none() && is_gid_name(name) {
291                    map.insert(code, String::from_utf8_lossy(name).into_owned());
292                }
293                code = code.wrapping_add(1);
294            }
295            _ => {}
296        }
297    }
298    map
299}
300
301/// Parse the embedded Type1 font program's built-in `/Encoding` vector
302/// (`dup <code> /<glyphname> put` entries in the clear-text header before
303/// `eexec`) into `code → char`. This is how docling recovers glyphs from
304/// standard TeX math fonts (CMMI/CMSY/…) that carry no PDF `/Encoding` and no
305/// ToUnicode — e.g. CMMI's `dup 21 /lambda` or CMSY's `dup 20 /lessequal`.
306/// Only `FontFile` (Type1) is parsed; CFF (`FontFile3`) and TrueType
307/// (`FontFile2`) store their encoding in a binary table and are left alone.
308fn type1_program_encoding(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
309    let mut map = HashMap::new();
310    let Some(desc) = fdict
311        .get(b"FontDescriptor")
312        .ok()
313        .and_then(|o| deref(doc, o))
314        .and_then(|o| o.as_dict().ok())
315    else {
316        return map;
317    };
318    let Some(data) = desc
319        .get(b"FontFile")
320        .ok()
321        .and_then(|o| deref(doc, o))
322        .and_then(|o| o.as_stream().ok())
323        .and_then(|s| s.decompressed_content().ok())
324    else {
325        return map;
326    };
327    // The clear-text header (PostScript) ends at `eexec`; the rest is encrypted.
328    let head_end = data
329        .windows(5)
330        .position(|w| w == b"eexec")
331        .unwrap_or(data.len());
332    let head = String::from_utf8_lossy(&data[..head_end]);
333    // Scan for `dup <code> /<name> put` tokens.
334    let toks: Vec<&str> = head.split_whitespace().collect();
335    for w in toks.windows(4) {
336        if w[0] == "dup" && w[3] == "put" {
337            if let (Ok(code), Some(name)) = (w[1].parse::<u32>(), w[2].strip_prefix('/')) {
338                if code <= 255 {
339                    if let Some(ch) = glyph_name_to_char(name.as_bytes()) {
340                        map.insert(code as u8, ch);
341                    }
342                }
343            }
344        }
345    }
346    map
347}
348
349/// A glyph name that is a synthetic placeholder, not a real Adobe name:
350/// `g115`, `cid42`, `glyph7`, `index9`, `G12`, or a short-prefix code name like
351/// `SM590000` (IBM BookMaster). These carry no Unicode meaning, and docling-parse
352/// emits them verbatim (`/SM590000`). `afii####` / `uni####` are real Adobe names
353/// and excluded. The restriction keeps genuine glyph names on the Unicode path.
354fn is_gid_name(name: &[u8]) -> bool {
355    let Ok(s) = std::str::from_utf8(name) else {
356        return false;
357    };
358    if s.starts_with("afii") || s.starts_with("uni") {
359        return false;
360    }
361    for prefix in ["g", "G", "cid", "CID", "glyph", "index"] {
362        if let Some(rest) = s.strip_prefix(prefix) {
363            if !rest.is_empty() && rest.bytes().all(|b| b.is_ascii_digit()) {
364                return true;
365            }
366        }
367    }
368    // Short alpha prefix (≤3 letters) followed by a run of ≥3 digits — synthetic
369    // code names like `SM590000`, distinct from real Adobe names (whole words or
370    // letter+`.suffix` variants).
371    let alpha = s.bytes().take_while(|b| b.is_ascii_alphabetic()).count();
372    let digits = s.len() - alpha;
373    (1..=3).contains(&alpha)
374        && digits >= 3
375        && s.as_bytes()[alpha..].iter().all(|b| b.is_ascii_digit())
376}
377
378fn font_ascent_descent(doc: &Document, fdict: &Dictionary, two_byte: bool) -> (f64, f64) {
379    // For Type0, the descriptor lives on the descendant CIDFont.
380    let descr_owner = if two_byte {
381        fdict
382            .get(b"DescendantFonts")
383            .ok()
384            .and_then(|o| deref(doc, o))
385            .and_then(|o| match o {
386                Object::Array(a) => a.first(),
387                _ => None,
388            })
389            .and_then(|o| as_dict(doc, o))
390    } else {
391        Some(fdict)
392    };
393    let fd = descr_owner
394        .and_then(|d| d.get(b"FontDescriptor").ok())
395        .and_then(|o| as_dict(doc, o));
396    let asc = fd
397        .and_then(|d| d.get(b"Ascent").ok())
398        .and_then(|o| {
399            o.as_float()
400                .ok()
401                .or_else(|| o.as_i64().ok().map(|i| i as f32))
402        })
403        .unwrap_or(750.0) as f64;
404    let desc = fd
405        .and_then(|d| d.get(b"Descent").ok())
406        .and_then(|o| {
407            o.as_float()
408                .ok()
409                .or_else(|| o.as_i64().ok().map(|i| i as f32))
410        })
411        .unwrap_or(-250.0) as f64;
412    // Some subsetted fonts carry a degenerate FontDescriptor (`/Ascent 0
413    // /Descent 0`) — the real metrics live in the font program. That collapses
414    // the loose box to zero height, so the line cells get zero area and the
415    // layout's region/text assignment drops them (2305's References list lost
416    // every prose line, keeping only the URLs). Fall back to typical text metrics
417    // so the box has height.
418    if asc - desc <= 1.0 {
419        return (750.0, -250.0);
420    }
421    (asc, desc)
422}
423
424/// The `/BaseFont` name with any `ABCDEF+` subset prefix stripped (#187).
425fn base_font_name(fdict: &Dictionary) -> Option<Vec<u8>> {
426    let name = fdict.get(b"BaseFont").ok()?.as_name().ok()?;
427    let stripped = match name.iter().position(|&b| b == b'+') {
428        Some(i) if i == 6 => &name[i + 1..],
429        _ => name,
430    };
431    Some(stripped.to_vec())
432}
433
434/// Simple-font widths: `/FirstChar` + `/Widths` array, `/MissingWidth` default.
435fn simple_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
436    let mut map = HashMap::new();
437    let first = fdict
438        .get(b"FirstChar")
439        .ok()
440        .and_then(|o| o.as_i64().ok())
441        .unwrap_or(0) as u32;
442    if let Some(Object::Array(arr)) = fdict.get(b"Widths").ok().and_then(|o| deref(doc, o)) {
443        for (i, w) in arr.iter().enumerate() {
444            if let Some(w) = num(w) {
445                map.insert(first + i as u32, w);
446            }
447        }
448    }
449    let dw = fdict
450        .get(b"FontDescriptor")
451        .ok()
452        .and_then(|o| as_dict(doc, o))
453        .and_then(|d| d.get(b"MissingWidth").ok())
454        .and_then(num)
455        .unwrap_or(0.0);
456    (map, dw)
457}
458
459/// CIDFont widths: the `/W` array on the descendant font (`/DW` default = 1000).
460fn cid_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
461    let mut map = HashMap::new();
462    let Some(desc) = fdict
463        .get(b"DescendantFonts")
464        .ok()
465        .and_then(|o| deref(doc, o))
466        .and_then(|o| match o {
467            Object::Array(a) => a.first(),
468            _ => None,
469        })
470        .and_then(|o| as_dict(doc, o))
471    else {
472        return (map, 1000.0);
473    };
474    let dw = desc.get(b"DW").ok().and_then(num).unwrap_or(1000.0);
475    if let Some(Object::Array(w)) = desc.get(b"W").ok().and_then(|o| deref(doc, o)) {
476        let mut i = 0;
477        while i < w.len() {
478            let c = w.get(i).and_then(num);
479            match (c, w.get(i + 1)) {
480                // `c [w1 w2 ...]`: consecutive CIDs starting at c.
481                (Some(c), Some(Object::Array(list))) => {
482                    for (k, wv) in list.iter().enumerate() {
483                        if let Some(wv) = num(wv) {
484                            map.insert(c as u32 + k as u32, wv);
485                        }
486                    }
487                    i += 2;
488                }
489                // `c_first c_last w`: a run all of width w.
490                (Some(c1), Some(o2)) => {
491                    if let (Some(c2), Some(wv)) = (num(o2), w.get(i + 2).and_then(num)) {
492                        for cid in c1 as u32..=c2 as u32 {
493                            map.insert(cid, wv);
494                        }
495                    }
496                    i += 3;
497                }
498                _ => break,
499            }
500        }
501    }
502    (map, dw)
503}
504
505fn num(o: &Object) -> Option<f64> {
506    match o {
507        Object::Integer(i) => Some(*i as f64),
508        Object::Real(r) => Some(*r as f64),
509        _ => None,
510    }
511}
512
513/// Parse a ToUnicode CMap's `bfchar` / `bfrange` sections into code→string.
514pub(crate) fn parse_tounicode(data: &[u8]) -> HashMap<u32, String> {
515    let text = String::from_utf8_lossy(data);
516    let mut map = HashMap::new();
517    let hex = |s: &str| -> Option<Vec<u16>> {
518        let s = s.trim();
519        if !s.starts_with('<') || !s.ends_with('>') {
520            return None;
521        }
522        let h = &s[1..s.len() - 1];
523        let bytes: Vec<u8> = (0..h.len())
524            .step_by(2)
525            .filter_map(|i| u8::from_str_radix(h.get(i..i + 2)?, 16).ok())
526            .collect();
527        Some(
528            bytes
529                .chunks(2)
530                .map(|c| {
531                    if c.len() == 2 {
532                        u16::from_be_bytes([c[0], c[1]])
533                    } else {
534                        c[0] as u16
535                    }
536                })
537                .collect(),
538        )
539    };
540    let u16s_to_string = |u: &[u16]| String::from_utf16_lossy(u);
541    let code_of = |u: &[u16]| u.iter().fold(0u32, |acc, &x| (acc << 16) | x as u32);
542
543    // Tokenize by structure, not whitespace: CMap hex groups are often written
544    // back-to-back with no separators (`<21><21><0054>`), so scan for `<…>`
545    // groups, `[`/`]` brackets, and bareword keywords.
546    let tokens: Vec<String> = {
547        let bytes = text.as_bytes();
548        let mut toks = Vec::new();
549        let mut i = 0;
550        while i < bytes.len() {
551            let c = bytes[i];
552            if c.is_ascii_whitespace() {
553                i += 1;
554            } else if c == b'<' {
555                let start = i;
556                while i < bytes.len() && bytes[i] != b'>' {
557                    i += 1;
558                }
559                i += 1; // include '>'
560                toks.push(String::from_utf8_lossy(&bytes[start..i.min(bytes.len())]).into_owned());
561            } else if c == b'[' || c == b']' {
562                toks.push((c as char).to_string());
563                i += 1;
564            } else {
565                let start = i;
566                while i < bytes.len()
567                    && !bytes[i].is_ascii_whitespace()
568                    && bytes[i] != b'<'
569                    && bytes[i] != b'['
570                    && bytes[i] != b']'
571                {
572                    i += 1;
573                }
574                toks.push(String::from_utf8_lossy(&bytes[start..i]).into_owned());
575            }
576        }
577        toks
578    };
579    let tokens: Vec<&str> = tokens.iter().map(|s| s.as_str()).collect();
580    let mut i = 0;
581    while i < tokens.len() {
582        match tokens[i] {
583            "beginbfchar" => {
584                i += 1;
585                while i + 1 < tokens.len() && tokens[i] != "endbfchar" {
586                    if let (Some(src), Some(dst)) = (hex(tokens[i]), hex(tokens[i + 1])) {
587                        map.insert(code_of(&src), u16s_to_string(&dst));
588                    }
589                    i += 2;
590                }
591            }
592            "beginbfrange" => {
593                i += 1;
594                while i + 2 < tokens.len() && tokens[i] != "endbfrange" {
595                    let (Some(lo), Some(hi)) = (hex(tokens[i]), hex(tokens[i + 1])) else {
596                        i += 1;
597                        continue;
598                    };
599                    let lo = code_of(&lo);
600                    let hi = code_of(&hi);
601                    if tokens[i + 2] == "[" {
602                        // `<lo> <hi> [ <d0> <d1> ... ]`: one dst per code in the range.
603                        let mut j = i + 3;
604                        let mut code = lo;
605                        while j < tokens.len() && tokens[j] != "]" {
606                            if let Some(dst) = hex(tokens[j]) {
607                                map.insert(code, u16s_to_string(&dst));
608                            }
609                            code += 1;
610                            j += 1;
611                        }
612                        i = j + 1;
613                    } else if let Some(dst) = hex(tokens[i + 2]) {
614                        // `<lo> <hi> <dst>`: consecutive Unicode from a base.
615                        let base = code_of(&dst);
616                        for (k, code) in (lo..=hi).enumerate() {
617                            if let Some(ch) = char::from_u32(base + k as u32) {
618                                map.insert(code, ch.to_string());
619                            }
620                        }
621                        i += 3;
622                    } else {
623                        i += 1;
624                    }
625                }
626            }
627            _ => i += 1,
628        }
629    }
630    map
631}
632
633/// Decode a PDF string literal in a Tj/TJ operand into raw code units.
634fn codes(font: &Font, bytes: &[u8]) -> Vec<u32> {
635    if font.two_byte {
636        bytes
637            .chunks(2)
638            .map(|c| {
639                if c.len() == 2 {
640                    ((c[0] as u32) << 8) | c[1] as u32
641                } else {
642                    c[0] as u32
643                }
644            })
645            .collect()
646    } else {
647        bytes.iter().map(|&b| b as u32).collect()
648    }
649}
650
651/// A page's display box, in PDF user space: the `/CropBox` clipped to the
652/// `/MediaBox`, both inherited through the page tree and normalized — a
653/// missing or empty MediaBox is US Letter, an empty CropBox is the MediaBox
654/// (pdfium's `CPDF_Page::UpdateDimensions`). pdfium reports the page size from
655/// this box, renders exactly it, and translates every content coordinate so
656/// its lower-left corner is the origin (`m_PageMatrix`); docling's backends
657/// inherit that frame, so text cells, `prov` boxes and destinations all count
658/// from the CropBox corner, not the MediaBox one. The parser used to flip
659/// glyphs with the MediaBox *height* and no translation at all, so a page
660/// whose boxes do not start at (0, 0) — a trimmed book page with
661/// `MediaBox [-56 -58 576 723]` / `CropBox [1 -0.6 519 666]`, or a LaTeX
662/// figure cropped to `[156 147 637 391]` — had its text displaced against the
663/// rendered bitmap by the box offset, the bottom lines pushed past the page
664/// edge and clamped to `t = b`.
665#[derive(Debug, Clone, Copy, PartialEq)]
666pub(crate) struct PageBox {
667    /// Left edge, user space.
668    pub l: f32,
669    /// Bottom edge, user space.
670    pub b: f32,
671    pub w: f32,
672    pub h: f32,
673}
674
675impl PageBox {
676    /// Top edge, user space — the y that becomes `0` in the y-down frame.
677    pub fn top(&self) -> f32 {
678        self.b + self.h
679    }
680}
681
682/// A page-tree rect attribute (`/MediaBox`, `/CropBox`), inherited from the
683/// nearest ancestor that sets it, as normalized `(l, b, r, t)`.
684fn inherited_rect(
685    doc: &Document,
686    page_id: lopdf::ObjectId,
687    key: &[u8],
688) -> Option<(f32, f32, f32, f32)> {
689    let mut id = page_id;
690    for _ in 0..32 {
691        let dict = doc.get_object(id).ok()?.as_dict().ok()?;
692        if let Some(Object::Array(a)) = dict.get(key).ok().and_then(|o| deref(doc, o)) {
693            let v: Vec<f32> = a.iter().filter_map(|o| num(o).map(|x| x as f32)).collect();
694            if v.len() == 4 && v.iter().all(|x| x.is_finite()) {
695                return Some((
696                    v[0].min(v[2]),
697                    v[1].min(v[3]),
698                    v[0].max(v[2]),
699                    v[1].max(v[3]),
700                ));
701            }
702            return None;
703        }
704        id = dict.get(b"Parent").ok()?.as_reference().ok()?;
705    }
706    None
707}
708
709pub(crate) fn page_box(doc: &Document, page_id: lopdf::ObjectId) -> PageBox {
710    let nonempty = |r: &(f32, f32, f32, f32)| r.2 > r.0 && r.3 > r.1;
711    let media = inherited_rect(doc, page_id, b"MediaBox")
712        .filter(nonempty)
713        .unwrap_or((0.0, 0.0, 612.0, 792.0));
714    let crop = inherited_rect(doc, page_id, b"CropBox")
715        .map(|c| {
716            (
717                c.0.max(media.0),
718                c.1.max(media.1),
719                c.2.min(media.2),
720                c.3.min(media.3),
721            )
722        })
723        .filter(nonempty)
724        .unwrap_or(media);
725    PageBox {
726        l: crop.0,
727        b: crop.1,
728        w: crop.2 - crop.0,
729        h: crop.3 - crop.1,
730    }
731}
732
733/// Page size (width, height) in PDF points — the display box's, like pdfium's
734/// `FPDF_GetPageWidthF/HeightF`.
735fn page_size(doc: &Document, page_id: lopdf::ObjectId) -> (f32, f32) {
736    let pb = page_box(doc, page_id);
737    (pb.w, pb.h)
738}
739
740/// Localize where a page's text is lost, for the `text_layer` diagnostic.
741/// Extraction can come up empty at three different points — no content stream
742/// reached the parser, the stream did not decode into operators, or it ran but
743/// produced no glyphs (fonts/encodings) — and from the outside all three look
744/// the same. Report them per page.
745pub fn content_diagnosis(bytes: &[u8]) -> String {
746    let Some(doc) = load_document(bytes) else {
747        return "document does not load".into();
748    };
749    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
750    pages.sort_by_key(|(n, _)| *n);
751    let mut out = String::new();
752    let mut caches = DocCaches::default();
753    for (n, pid) in pages.into_iter().take(4) {
754        let content_bytes = doc.get_page_content(pid);
755        let ops = lopdf::content::Content::decode(&content_bytes)
756            .map(|c| c.operations.len())
757            .ok();
758        let res = page_res(&doc, pid);
759        let fonts = res.map(|r| fonts_from_res(&doc, r, &mut caches).len());
760        let glyphs = page_glyphs_cached(&doc, pid, &mut caches).len();
761        out.push_str(&format!(
762            "\n   page {n}: content {} B, ops {}, resources {}, fonts {}, glyphs {}",
763            content_bytes.len(),
764            ops.map_or("UNDECODABLE".to_string(), |n| n.to_string()),
765            if res.is_some() { "ok" } else { "MISSING" },
766            fonts.map_or("-".to_string(), |n| n.to_string()),
767            glyphs,
768        ));
769    }
770    out
771}
772
773/// Is this "text layer" a vestige rather than the document's text?
774///
775/// Scanned forms often carry a handful of typed-in strings — a date filled
776/// into three form fields, say — on top of pages that are otherwise images.
777/// Treating that as a real text layer is the worst of both worlds: the text
778/// path proudly extracts thirteen characters, and no OCR ever runs on the
779/// letter the pages actually show. The reported form did exactly this (3
780/// lines, 13 chars, 3 pages).
781///
782/// The rule is deliberately tight so genuinely sparse *digital* documents are
783/// not misrouted into OCR: only a document averaging at most one line per page
784/// **and** totalling fewer than 32 characters is called vestigial.
785pub fn text_layer_is_vestigial(pages: &[crate::pdfium_backend::PdfPage]) -> bool {
786    let lines: usize = pages.iter().map(|p| p.cells.len()).sum();
787    if lines == 0 {
788        return true;
789    }
790    let chars: usize = pages
791        .iter()
792        .flat_map(|p| &p.cells)
793        .map(|c| c.text.chars().count())
794        .sum();
795    lines <= pages.len() && chars < 32
796}
797
798/// Why the cross-reference repair did or did not fire, for the `text_layer`
799/// diagnostic. A PDF that will not load is indistinguishable from a scan in
800/// production (both convert to nothing), so the reason has to be askable.
801pub fn xref_repair_status(bytes: &[u8]) -> String {
802    if Document::load_mem(bytes).is_ok() {
803        return "loads unaided; no repair needed".into();
804    }
805    match pad_short_xref_entries(bytes) {
806        Ok(fixed) => match Document::load_mem(&fixed) {
807            Ok(_) => "repaired: cross-reference entries padded to 20 bytes".into(),
808            Err(e) => format!("padded the entries, but it still will not load: {e}"),
809        },
810        Err(why) => format!("repair declined — {why}"),
811    }
812}
813
814/// Load a PDF, repairing the one malformation that otherwise costs us the whole
815/// document: **19-byte cross-reference entries**.
816///
817/// The spec fixes an xref entry at 20 bytes — `nnnnnnnnnn ggggg n` plus a
818/// *two*-byte EOL. Some generators (an Austrian telecom's invoices, for one)
819/// emit a bare LF instead, making each entry 19 bytes. lopdf rejects the file
820/// outright (`invalid file trailer`) where pdfium reads it happily, so a
821/// perfectly good text layer looked to the browser exactly like a scan and cost
822/// ten seconds of OCR.
823///
824/// Padding is only attempted when it cannot move anything the xref points at:
825/// a single `xref` section that begins after the last object. The repair then
826/// has to prove itself — the padded bytes are used only if they load — so a
827/// mis-repair degrades to today's behaviour rather than to silent garbage.
828pub(crate) fn load_document(bytes: &[u8]) -> Option<Document> {
829    open_document(bytes, None).ok()
830}
831
832/// Why [`open_document`] could not hand back a readable document.
833#[derive(Debug, Clone, Copy, PartialEq, Eq)]
834pub(crate) enum OpenError {
835    /// lopdf cannot read the file even after the repairs.
836    Unreadable,
837    /// The file is encrypted and the password given (or none) does not open it.
838    Password,
839}
840
841/// Load `bytes` with the document's password. lopdf decrypts while reading —
842/// with `password`, or the empty user password most "protected" PDFs carry
843/// (what every viewer opens silently) — so every reader downstream sees plain
844/// streams; a file the password does not open loads with its streams still
845/// encrypted, which is reported as [`OpenError::Password`] rather than handed
846/// on as a document whose every stream decodes to nothing.
847pub(crate) fn open_document(bytes: &[u8], password: Option<&str>) -> Result<Document, OpenError> {
848    let Some(doc) = load_document_raw(bytes, password) else {
849        // lopdf refuses to load at all under a *wrong* password (a missing
850        // one loads the file with its streams still encrypted); tell the two
851        // apart by loading without it.
852        if password.is_some() && load_document_raw(bytes, None).is_some_and(|d| d.is_encrypted()) {
853            return Err(OpenError::Password);
854        }
855        return Err(OpenError::Unreadable);
856    };
857    if doc.is_encrypted() {
858        return Err(OpenError::Password);
859    }
860    Ok(doc)
861}
862
863fn load_options(password: Option<&str>) -> lopdf::LoadOptions {
864    lopdf::LoadOptions {
865        password: password.map(str::to_string),
866        ..lopdf::LoadOptions::default()
867    }
868}
869
870fn load_document_raw(bytes: &[u8], password: Option<&str>) -> Option<Document> {
871    // Try progressively more repair, and accept a candidate only once the pages
872    // actually carry content — a document whose streams were dropped still
873    // "loads", so loading alone is not evidence the repair helped. A
874    // well-formed file returns on the first attempt and pays for nothing.
875    let mut fallback = None;
876    if let Some(doc) = best_effort_load(bytes, password, &mut fallback) {
877        return Some(doc);
878    }
879    let xref_fixed = pad_short_xref_entries(bytes).ok();
880    if let Some(fixed) = &xref_fixed {
881        if let Some(doc) = best_effort_load(fixed, password, &mut fallback) {
882            return Some(doc);
883        }
884    }
885    // Both defects can coexist, and the second only becomes visible once the
886    // first is repaired, so build on whatever the previous step produced.
887    let lengths_fixed = fix_stream_lengths(xref_fixed.as_deref().unwrap_or(bytes));
888    if let Some(doc) = best_effort_load(&lengths_fixed, password, &mut fallback) {
889        return Some(doc);
890    }
891    fallback
892}
893
894/// Load `data`, returning it only when its pages carry content; a document that
895/// merely parses is remembered as the fallback for when nothing does better.
896fn best_effort_load(
897    data: &[u8],
898    password: Option<&str>,
899    fallback: &mut Option<Document>,
900) -> Option<Document> {
901    match Document::load_mem_with_options(data, load_options(password)) {
902        Ok(doc) if has_page_content(&doc) => Some(doc),
903        Ok(doc) => {
904            fallback.get_or_insert(doc);
905            None
906        }
907        Err(_) => None,
908    }
909}
910
911/// Does any page actually hand us a content stream? A document whose streams
912/// were dropped still parses — it simply has nothing to read — so this is what
913/// tells a successful repair from a pointless one.
914fn has_page_content(doc: &Document) -> bool {
915    doc.get_pages()
916        .into_values()
917        .take(4)
918        .any(|pid| !doc.get_page_content(pid).is_empty())
919}
920
921/// Correct `/Length` values that disagree with where `endstream` actually is.
922///
923/// The same generator that writes short xref entries also overstates its
924/// content-stream lengths by a byte or two. lopdf trusts `/Length`, reads past
925/// the data, fails to find `endstream` there and drops the stream — the object
926/// comes back as a bare dictionary, so the page has no content at all and the
927/// document looks like a scan. pdfium instead trusts `endstream`, which is what
928/// this does.
929///
930/// The rewrite is length-preserving: the corrected number is written over the
931/// old digits and padded with spaces, so every byte offset in the file — and
932/// therefore the whole cross-reference table — stays valid.
933fn fix_stream_lengths(bytes: &[u8]) -> Vec<u8> {
934    let mut out = bytes.to_vec();
935    let mut i = 0;
936    while let Some(rel) = find(&out[i..], b"stream") {
937        let kw = i + rel;
938        i = kw + 6;
939        // Skip `endstream` (the keyword we are measuring *to*).
940        if kw >= 3 && &out[kw - 3..kw] == b"end" {
941            continue;
942        }
943        // The stream data starts after the EOL that follows the keyword.
944        let mut data = kw + 6;
945        if out.get(data..data + 2) == Some(b"\r\n".as_slice()) {
946            data += 2;
947        } else if matches!(out.get(data), Some(b'\n' | b'\r')) {
948            data += 1;
949        }
950        let Some(end) = find(&out[data..], b"endstream").map(|r| data + r) else {
951            continue;
952        };
953        // `/Length <digits>` in the dictionary just before the keyword.
954        let dict_start = out[..kw].iter().rposition(|&c| c == b'<').unwrap_or(0);
955        let Some(lrel) = find(&out[dict_start..kw], b"/Length") else {
956            continue;
957        };
958        let mut d = dict_start + lrel + 7;
959        while matches!(out.get(d), Some(b' ')) {
960            d += 1;
961        }
962        let digits = out[d..].iter().take_while(|c| c.is_ascii_digit()).count();
963        if digits == 0 {
964            continue;
965        }
966        let declared: usize = match std::str::from_utf8(&out[d..d + digits])
967            .ok()
968            .and_then(|s| s.parse().ok())
969        {
970            Some(v) => v,
971            None => continue,
972        };
973        let actual = end - data;
974        // Only shrink, and only when the new value fits the space the old one
975        // occupied — growing the number would move every following byte.
976        let replacement = actual.to_string();
977        if actual == declared || replacement.len() > digits {
978            continue;
979        }
980        out[d..d + digits].fill(b' ');
981        out[d..d + replacement.len()].copy_from_slice(replacement.as_bytes());
982    }
983    out
984}
985
986fn find(haystack: &[u8], needle: &[u8]) -> Option<usize> {
987    haystack.windows(needle.len()).position(|w| w == needle)
988}
989
990/// Rewrite a classic cross-reference table's entries to the spec's 20 bytes,
991/// or `None` when the file's shape makes that unsafe (see [`load_document`]).
992fn pad_short_xref_entries(bytes: &[u8]) -> Result<Vec<u8>, &'static str> {
993    // Exactly one xref section, and it must start after every object, so that
994    // growing it shifts nothing the table's offsets refer to.
995    let is_boundary = |i: usize| i == 0 || matches!(bytes[i - 1], b'\n' | b'\r');
996    let mut starts = (0..bytes.len().saturating_sub(4))
997        .filter(|&i| &bytes[i..i + 4] == b"xref" && is_boundary(i));
998    let xref_at = starts
999        .next()
1000        .ok_or("no classic `xref` section (an xref stream?)")?;
1001    if starts.next().is_some() {
1002        return Err("more than one xref section (incremental update)");
1003    }
1004    let last_obj = bytes
1005        .windows(3)
1006        .rposition(|w| w == b"obj")
1007        .ok_or("no objects found")?;
1008    if last_obj > xref_at {
1009        return Err("an object follows the xref — padding would move it");
1010    }
1011
1012    let mut out = bytes[..xref_at].to_vec();
1013    out.extend_from_slice(b"xref\n");
1014    let mut i = xref_at + 4;
1015    let skip_ws = |i: &mut usize| {
1016        while matches!(bytes.get(*i), Some(b'\r' | b'\n' | b' ')) {
1017            *i += 1;
1018        }
1019    };
1020    loop {
1021        skip_ws(&mut i);
1022        // Either the next subsection header ("first count") or the trailer.
1023        if bytes[i..].starts_with(b"trailer") {
1024            out.extend_from_slice(&bytes[i..]);
1025            return Ok(out);
1026        }
1027        let header_end = i + bytes[i..]
1028            .iter()
1029            .position(|c| matches!(c, b'\n' | b'\r'))
1030            .ok_or("subsection header runs off the end")?;
1031        let header = std::str::from_utf8(&bytes[i..header_end])
1032            .map_err(|_| "subsection header is not text")?
1033            .trim();
1034        let mut parts = header.split_whitespace();
1035        let count: usize = parts
1036            .nth(1)
1037            .and_then(|c| c.parse().ok())
1038            .ok_or("unparseable subsection header")?;
1039        if parts.next().is_some() || count == 0 {
1040            return Err("unexpected subsection header shape");
1041        }
1042        out.extend_from_slice(header.as_bytes());
1043        out.push(b'\n');
1044        i = header_end;
1045        for _ in 0..count {
1046            skip_ws(&mut i);
1047            // `nnnnnnnnnn ggggg n` — the 18 bytes before whatever EOL follows.
1048            let entry = bytes.get(i..i + 18).ok_or("xref entry runs off the end")?;
1049            let well_formed = entry[..10].iter().all(u8::is_ascii_digit)
1050                && entry[10] == b' '
1051                && entry[11..16].iter().all(u8::is_ascii_digit)
1052                && entry[16] == b' '
1053                && matches!(entry[17], b'n' | b'f');
1054            if !well_formed {
1055                return Err("xref entry is not `nnnnnnnnnn ggggg n`");
1056            }
1057            out.extend_from_slice(entry);
1058            out.extend_from_slice(b" \n"); // the spec's 2-byte EOL -> 20 bytes
1059            i += 18;
1060        }
1061    }
1062}
1063
1064/// Debug: raw glyph stream `(ch, ll, lr, lb, lt)` (native coords) for page
1065/// `index`, before the sanitizer. For comparing char cells to docling-parse.
1066pub fn debug_glyphs(bytes: &[u8], index: usize) -> Vec<(char, f32, f32, f32, f32)> {
1067    let Some(doc) = load_document(bytes) else {
1068        return Vec::new();
1069    };
1070    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1071    pages.sort_by_key(|(n, _)| *n);
1072    let Some((_, pid)) = pages.get(index) else {
1073        return Vec::new();
1074    };
1075    page_glyphs(&doc, *pid)
1076        .into_iter()
1077        .map(|g| (g.ch, g.ll, g.lr, g.lb, g.lt))
1078        .collect()
1079}
1080
1081/// Public entry: per-page (width, height, line cells) for a PDF, via the Rust
1082/// text parser + the docling-parse line sanitizer. Used by the pipeline and the
1083/// `textparse_dump` example.
1084pub fn pdf_textlines(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
1085    let Some(doc) = load_document(bytes) else {
1086        return Vec::new();
1087    };
1088    let mut caches = DocCaches::default();
1089    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1090    pages.sort_by_key(|(n, _)| *n);
1091    pages
1092        .into_iter()
1093        .map(|(_, pid)| {
1094            let (w, h) = page_size(&doc, pid);
1095            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1096            let cells = crate::dp_lines::line_cells(&glyphs, h, true);
1097            (w, h, cells)
1098        })
1099        .collect()
1100}
1101
1102/// Debug/diagnostic entry: per-page (width, height, word cells) for a PDF, via
1103/// the Rust parser glyphs run through the docling-parse word grouping. Used to
1104/// compare parser word cells against docling-parse's `word_cells` oracle (roadmap
1105/// item 6).
1106pub fn pdf_words(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
1107    let Some(doc) = load_document(bytes) else {
1108        return Vec::new();
1109    };
1110    let mut caches = DocCaches::default();
1111    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1112    pages.sort_by_key(|(n, _)| *n);
1113    pages
1114        .into_iter()
1115        .map(|(_, pid)| {
1116            let (w, h) = page_size(&doc, pid);
1117            let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1118            let cells = crate::dp_lines::word_cells(&glyphs, h, true);
1119            (w, h, cells)
1120        })
1121        .collect()
1122}
1123
1124/// One page's text cells from the pure-Rust parser: prose line cells, per-word
1125/// cells, and code line cells — all from a single glyph parse. Replaces the
1126/// pdfium text path (roadmap item 6) when the parser drop is enabled.
1127#[derive(Default)]
1128pub struct PageParserCells {
1129    pub prose: Vec<crate::pdfium_backend::TextCell>,
1130    pub words: Vec<crate::pdfium_backend::TextCell>,
1131    pub code: Vec<crate::pdfium_backend::TextCell>,
1132    /// Drawn checkbox squares (#609), in the cells' frame.
1133    pub checkboxes: Vec<crate::checkbox::CheckBox>,
1134}
1135
1136/// The parser text layer, driven one page at a time: the document is loaded
1137/// (and repaired, see [`load_document`]) once, the font/form caches persist
1138/// across pages, and each page's glyphs are parsed only when asked for.
1139///
1140/// The eager whole-document walk this replaces ran *before* the first page
1141/// was rendered, so on a long PDF it was a serial prefix the page-worker pool
1142/// sat idle through — 6.2 s on the 1913-page .NET reference, in front of a
1143/// pipeline that otherwise overlaps parsing with inference — and a `--pages`
1144/// window still paid for every page in the file. Pulling pages on demand
1145/// keeps the parse on the producer thread but interleaved with rendering,
1146/// and skips unselected pages entirely. Output per page is unchanged: same
1147/// glyph walk, same shared caches, same contraction.
1148pub struct PageTextParser {
1149    doc: Document,
1150    caches: DocCaches,
1151    /// Page object ids in document order (page 1 first).
1152    pages: Vec<lopdf::ObjectId>,
1153}
1154
1155impl PageTextParser {
1156    /// Load the document; `None` when lopdf cannot read it at all.
1157    pub fn open(bytes: &[u8]) -> Option<Self> {
1158        Self::open_with_password(bytes, None)
1159    }
1160
1161    /// [`open`](Self::open) with the document's password (an encrypted file
1162    /// the password does not open reads as `None` here; the pipeline reports
1163    /// it through [`crate::pdf_meta::PdfMeta::open_with_password`] first).
1164    pub fn open_with_password(bytes: &[u8], password: Option<&str>) -> Option<Self> {
1165        let doc = open_document(bytes, password).ok()?;
1166        let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1167        pages.sort_by_key(|(n, _)| *n);
1168        Some(Self {
1169            doc,
1170            caches: DocCaches::default(),
1171            pages: pages.into_iter().map(|(_, pid)| pid).collect(),
1172        })
1173    }
1174
1175    /// The glyph boxes and font styles of the 0-based page `index` — the
1176    /// heading-hierarchy stage's style signal (#302): every non-space glyph's
1177    /// box (font ascent + descent at its size — the font-size proxy pdfium's
1178    /// loose char box also gave) in top-left coordinates, with the weight
1179    /// class and slant its `/BaseFont` name declares. Empty for a page
1180    /// without a text layer (a scan), and the stage falls back to its other
1181    /// signals.
1182    pub(crate) fn glyph_styles(
1183        &mut self,
1184        index: usize,
1185    ) -> Vec<crate::heading_hierarchy::GlyphStyle> {
1186        let Some(&pid) = self.pages.get(index) else {
1187            return Vec::new();
1188        };
1189        let (_w, h) = page_size(&self.doc, pid);
1190        let glyphs = page_glyphs_cached(&self.doc, pid, &mut self.caches);
1191        // Font hash → style, over the fonts the walk just parsed (an inline,
1192        // uncached font dictionary reads as unstyled).
1193        let styles: HashMap<u64, crate::font_style::FontStyle> = self
1194            .caches
1195            .fonts
1196            .values()
1197            .map(|f| (f.hash, f.style))
1198            .collect();
1199        glyphs
1200            .iter()
1201            .filter(|g| !g.ch.is_whitespace() && g.ll.is_finite())
1202            .map(|g| {
1203                let st = styles.get(&g.font).copied().unwrap_or_default();
1204                crate::heading_hierarchy::GlyphStyle {
1205                    l: g.ll,
1206                    t: h - g.lt,
1207                    r: g.lr,
1208                    b: h - g.lb,
1209                    height: g.height(),
1210                    weight_cls: crate::font_style::weight_class(st.weight),
1211                    italic: st.italic,
1212                    styled: st.known,
1213                }
1214            })
1215            .collect()
1216    }
1217
1218    /// Prose, word and code cells of the 0-based page `index` — empty for an
1219    /// index the parser's page tree doesn't have (a damaged file whose page
1220    /// tree disagrees with the object model's count).
1221    pub fn cells(&mut self, index: usize) -> PageParserCells {
1222        let Some(&pid) = self.pages.get(index) else {
1223            return PageParserCells::default();
1224        };
1225        let (_w, h) = page_size(&self.doc, pid);
1226        let (glyphs, inks) = page_glyphs_and_inks(&self.doc, pid, &mut self.caches);
1227        let (prose, words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1228        PageParserCells {
1229            prose,
1230            words,
1231            code: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1232            checkboxes: crate::checkbox::find(&inks, h),
1233        }
1234    }
1235}
1236
1237/// Full parser text layer: prose + word + code cells per page, glyphs parsed once.
1238/// `prose`/`words` come from the docling-parse contraction ([`crate::dp_lines`]);
1239/// `code` splits only at the parser's own space glyphs (monospace keeps its
1240/// source spacing). The eager form of [`PageTextParser`].
1241pub fn pdf_all_cells(bytes: &[u8]) -> Vec<PageParserCells> {
1242    let Some(mut parser) = PageTextParser::open(bytes) else {
1243        return Vec::new();
1244    };
1245    (0..parser.pages.len()).map(|i| parser.cells(i)).collect()
1246}
1247
1248/// Whole pages for the text-layer-only conversion ([`crate::convert_text_layer`]):
1249/// the parser's prose/word/code cells plus page geometry, assembled into
1250/// [`PdfPage`]s with no rendered image and no link annotations. Everything here
1251/// is pure Rust (lopdf), so it compiles without the `ml` feature — including on
1252/// wasm32. A page the parser can't read (no text layer) comes back with empty
1253/// cells; there is no pdfium fallback on this path.
1254pub fn pdf_text_pages(bytes: &[u8]) -> Vec<crate::pdfium_backend::PdfPage> {
1255    let Some(doc) = load_document(bytes) else {
1256        return Vec::new();
1257    };
1258    let mut caches = DocCaches::default();
1259    let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1260    pages.sort_by_key(|(n, _)| *n);
1261    pages
1262        .into_iter()
1263        .map(|(_, pid)| {
1264            let (w, h) = page_size(&doc, pid);
1265            let (glyphs, inks) = page_glyphs_and_inks(&doc, pid, &mut caches);
1266            let (mut prose, mut words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1267            drop_overpainted_cells(&mut prose);
1268            drop_overpainted_cells(&mut words);
1269            crate::pdfium_backend::PdfPage {
1270                #[cfg(feature = "ocr-prep")]
1271                image_layout: None,
1272                width: w,
1273                height: h,
1274                // Cells are native PDF points; there is no rendered bitmap.
1275                scale: 1.0,
1276                cells: prose,
1277                code_cells: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1278                word_cells: words,
1279                checkboxes: crate::checkbox::find(&inks, h),
1280                #[cfg(feature = "ocr-prep")]
1281                image: image::RgbImage::new(1, 1),
1282                links: Vec::new(),
1283                rotation: 0,
1284            }
1285        })
1286        .collect()
1287}
1288
1289/// Drop line cells that are *painted over each other* — glyphs used as artwork.
1290///
1291/// Some generators draw their logo with a symbol font: on the reporting
1292/// invoice, a `TeleLogo` Type1 paints the T-Mobile mark by stacking the glyphs
1293/// encoded as `"` and `==` on top of one another, and the flat text-layer
1294/// output opened with that garbage. Nothing in the font metadata gives it away
1295/// (the *text* fonts in the same file are also flagged symbolic, and the logo
1296/// font names its glyphs `quotedbl` &c.), but the geometry does: two cells with
1297/// different text where one lies inside the other on the same line is
1298/// physically impossible for prose — ink from two words never occupies the
1299/// same box. Both cells of such a pair are paint, not text.
1300///
1301/// Containment (not mere overlap) keeps this narrow: adjacent words touch but
1302/// never contain each other, and a same-text near-duplicate (double-draw faux
1303/// bold) is left alone for the sanitizer's usual handling. Applied on the
1304/// flat/browser path only — the ML pipeline's text layer is byte-pinned by the
1305/// PDF corpus, and there the layout model already sinks logo marks into
1306/// `picture` regions.
1307fn drop_overpainted_cells(cells: &mut Vec<crate::pdfium_backend::TextCell>) {
1308    let mut paint = vec![false; cells.len()];
1309    for i in 0..cells.len() {
1310        for j in 0..cells.len() {
1311            if i == j || cells[i].text == cells[j].text {
1312                continue;
1313            }
1314            let (a, b) = (&cells[i], &cells[j]);
1315            // Same line band: the vertical overlap covers most of the shorter.
1316            let vo = (a.b.min(b.b) - a.t.max(b.t)).max(0.0);
1317            if vo < 0.6 * (a.b - a.t).min(b.b - b.t) {
1318                continue;
1319            }
1320            // `a` horizontally inside `b` (with a small tolerance).
1321            let ho = (a.r.min(b.r) - a.l.max(b.l)).max(0.0);
1322            if ho >= 0.8 * (a.r - a.l) && (a.r - a.l) <= (b.r - b.l) {
1323                paint[i] = true;
1324                paint[j] = true;
1325            }
1326        }
1327    }
1328    let mut keep = paint.iter().map(|p| !p);
1329    cells.retain(|_| keep.next().unwrap());
1330}
1331
1332/// The text-state scalars inherited by a Form XObject when it is invoked via
1333/// `Do` (the PDF graphics state includes the text parameters, but not the text
1334/// matrices, which a form re-establishes inside its own `BT`/`ET`).
1335#[derive(Clone, Copy)]
1336struct TextState {
1337    tc: f64,
1338    tw: f64,
1339    th: f64,
1340    tl: f64,
1341    trise: f64,
1342    fsize: f64,
1343}
1344
1345impl TextState {
1346    const INIT: TextState = TextState {
1347        tc: 0.0,
1348        tw: 0.0,
1349        th: 1.0,
1350        tl: 0.0,
1351        trise: 0.0,
1352        fsize: 0.0,
1353    };
1354}
1355
1356/// The effective `/Resources` dictionary for a page (inline or via reference,
1357/// falling back to an inherited one from a `/Parent`).
1358fn page_res(doc: &Document, page_id: lopdf::ObjectId) -> Option<&Dictionary> {
1359    let (inline, ids) = doc.get_page_resources(page_id).ok()?;
1360    if let Some(d) = inline {
1361        return Some(d);
1362    }
1363    ids.into_iter().find_map(|id| doc.get_dictionary(id).ok())
1364}
1365
1366/// Build the code→[`Font`] map for a resources dictionary's `/Font` sub-dict,
1367/// reusing the per-document cache for fonts referenced indirectly (the common
1368/// case — the same font objects recur on every page).
1369fn fonts_from_res(
1370    doc: &Document,
1371    res: &Dictionary,
1372    caches: &mut DocCaches,
1373) -> HashMap<Vec<u8>, Arc<Font>> {
1374    let mut map = HashMap::new();
1375    let font_dict = res
1376        .get(b"Font")
1377        .ok()
1378        .and_then(|o| deref(doc, o))
1379        .and_then(|o| o.as_dict().ok());
1380    if let Some(fd) = font_dict {
1381        for (name, value) in fd.iter() {
1382            let font = match value {
1383                Object::Reference(id) => {
1384                    let key = (*id, name.clone());
1385                    if let Some(f) = caches.fonts.get(&key) {
1386                        Arc::clone(f)
1387                    } else if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1388                        let f = Arc::new(parse_font(doc, name, fdict));
1389                        caches.fonts.insert(key, Arc::clone(&f));
1390                        f
1391                    } else {
1392                        continue;
1393                    }
1394                }
1395                _ => {
1396                    if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1397                        Arc::new(parse_font(doc, name, fdict))
1398                    } else {
1399                        continue;
1400                    }
1401                }
1402            };
1403            map.insert(name.clone(), font);
1404        }
1405    }
1406    map
1407}
1408
1409/// Extract every glyph on a page as a native-coordinate [`Glyph`].
1410/// [`PageTextParser::glyph_styles`] for the given **1-based** pages of a
1411/// document, keyed by page number — a separate, on-demand pass (no
1412/// rendering), so the extraction pipeline stays byte-identical whether or not
1413/// the heading-hierarchy stage runs. Empty when lopdf cannot open the file.
1414pub(crate) fn glyph_styles(
1415    bytes: &[u8],
1416    pages: &[usize],
1417) -> HashMap<usize, Vec<crate::heading_hierarchy::GlyphStyle>> {
1418    let mut out = HashMap::new();
1419    let Some(mut parser) = PageTextParser::open(bytes) else {
1420        return out;
1421    };
1422    for &page_no in pages {
1423        if page_no == 0 {
1424            continue;
1425        }
1426        let styles = parser.glyph_styles(page_no - 1);
1427        if !styles.is_empty() {
1428            out.insert(page_no, styles);
1429        }
1430    }
1431    out
1432}
1433
1434pub(crate) fn page_glyphs(doc: &Document, page_id: lopdf::ObjectId) -> Vec<Glyph> {
1435    page_glyphs_cached(doc, page_id, &mut DocCaches::default())
1436}
1437
1438/// [`page_glyphs`] with an explicit per-document cache, so a multi-page walk
1439/// parses each font / decodes each form once instead of once per page.
1440fn page_glyphs_cached(
1441    doc: &Document,
1442    page_id: lopdf::ObjectId,
1443    caches: &mut DocCaches,
1444) -> Vec<Glyph> {
1445    page_glyphs_and_inks(doc, page_id, caches).0
1446}
1447
1448/// [`page_glyphs_cached`] plus the small painted path pieces the same walk
1449/// passed — what [`crate::checkbox::find`] looks for checkbox squares in
1450/// (#609) — in the glyphs' frame.
1451fn page_glyphs_and_inks(
1452    doc: &Document,
1453    page_id: lopdf::ObjectId,
1454    caches: &mut DocCaches,
1455) -> (Vec<Glyph>, Vec<crate::checkbox::Ink>) {
1456    let mut out = Vec::new();
1457    let mut inks = Vec::new();
1458    // lopdf 0.44: get_page_content returns the assembled content-stream bytes
1459    // directly (an empty Vec when the page has none).
1460    let content_bytes = doc.get_page_content(page_id);
1461    let Ok(content) = lopdf::content::Content::decode(&content_bytes) else {
1462        return (out, inks);
1463    };
1464    if let Some(res) = page_res(doc, page_id) {
1465        // pdfium's page matrix: user space translated so the display box's
1466        // lower-left corner is the origin (see [`PageBox`]).
1467        let pb = page_box(doc, page_id);
1468        let base = Mat {
1469            e: -(pb.l as f64),
1470            f: -(pb.b as f64),
1471            ..Mat::ID
1472        };
1473        run_content(
1474            doc,
1475            res,
1476            &content,
1477            base,
1478            TextState::INIT,
1479            0,
1480            caches,
1481            &mut out,
1482            &mut inks,
1483        );
1484        out.retain(|g| on_page(g, pb.w, pb.h));
1485    }
1486    (out, inks)
1487}
1488
1489/// Whether a glyph is on the page (#529): docling-parse keeps a character
1490/// only when its whole box lies inside the display box — the CropBox, or
1491/// the MediaBox without one — edges included, so a FrameMaker print slug
1492/// drawn beside the CropBox or a tiled page's neighbouring text beyond the
1493/// MediaBox never becomes a cell, and a line crossing the edge is cut at the
1494/// last glyph that fits (`Crossing the rig`). The box is docling-parse's
1495/// char box — the advance by the font's ascent/descent, the loose box here
1496/// (its axis-aligned extent for rotated text) — in the frame `page_glyphs`
1497/// already moved to the display box's corner, so the page is `[0, w] × [0,
1498/// h]`. A glyph without a finite box is kept, as before. (A standard-14 font
1499/// without a FontDescriptor gets the 750 / −250 default ascent / descent
1500/// here where docling-parse reads the AFM's — Helvetica's 718 / −207 — so
1501/// for those a glyph within ~0.04 em of the top or bottom edge can fall the
1502/// other way; horizontally the advance is the same.)
1503fn on_page(g: &Glyph, w: f32, h: f32) -> bool {
1504    // f32 noise only: docling-parse already drops a glyph 0.01 pt over.
1505    const EPS: f32 = 1e-3;
1506    let (l, b, r, t) = if [g.ll, g.lb, g.lr, g.lt].iter().all(|v| v.is_finite()) {
1507        (g.ll, g.lb, g.lr, g.lt)
1508    } else {
1509        (g.l, g.b, g.r, g.t)
1510    };
1511    if ![l, b, r, t].iter().all(|v| v.is_finite()) {
1512        return true;
1513    }
1514    l >= -EPS && b >= -EPS && r <= w + EPS && t <= h + EPS
1515}
1516
1517/// Run a content stream's operators, emitting glyphs into `out`. Recurses into
1518/// Form XObjects on `Do` (bulk body text in heavy PDFs lives inside a form, not
1519/// the page content stream). `res` is the resources dict in scope (the page's,
1520/// or the form's own); `base_ctm` is the CTM at the point of invocation.
1521#[allow(clippy::too_many_arguments)]
1522fn run_content(
1523    doc: &Document,
1524    res: &Dictionary,
1525    content: &lopdf::content::Content,
1526    base_ctm: Mat,
1527    init: TextState,
1528    depth: u32,
1529    caches: &mut DocCaches,
1530    out: &mut Vec<Glyph>,
1531    inks: &mut Vec<crate::checkbox::Ink>,
1532) {
1533    let fonts = fonts_from_res(doc, res, caches);
1534    let xobjects = res
1535        .get(b"XObject")
1536        .ok()
1537        .and_then(|o| deref(doc, o))
1538        .and_then(|o| o.as_dict().ok());
1539
1540    // Graphics + text state. `q`/`Q` save and restore the whole graphics state,
1541    // which includes the text parameters (Tc, Tw, Tz, TL, Tfs, Trise, font) —
1542    // *not* the text matrix (that is reset by BT). Saving only the CTM let a Tc
1543    // set inside a `q…Q` block leak out and drift every later glyph.
1544    #[allow(clippy::type_complexity)]
1545    let mut gstate_stack: Vec<(Mat, f64, f64, f64, f64, f64, f64, Option<&Arc<Font>>)> = Vec::new();
1546    let mut ctm = base_ctm;
1547    let mut tm = Mat::ID;
1548    let mut tlm = Mat::ID;
1549    let mut font: Option<&Arc<Font>> = None;
1550    let mut fsize = init.fsize;
1551    let mut tc = init.tc; // char spacing
1552    let mut tw = init.tw; // word spacing
1553    let mut th = init.th; // horizontal scale (Tz/100)
1554    let mut tl = init.tl; // leading
1555    let mut trise = init.trise;
1556
1557    let op_f = |operands: &[Object], i: usize| operands.get(i).and_then(num).unwrap_or(0.0);
1558    // The path under construction, already in page space (a path is built
1559    // under one CTM — `cm` is not allowed inside it), for the checkbox
1560    // squares (#609). Text extraction never reads it.
1561    let mut path: Vec<PathPiece> = Vec::new();
1562    let (mut cur, mut start) = ((0.0, 0.0), (0.0, 0.0));
1563    // Whether the fill colour is (near) white, saved/restored with `q`/`Q`:
1564    // a white-filled stroked box is a checkbox outline, a coloured one a
1565    // chart legend's swatch. The initial fill colour is black.
1566    let mut fill_light = false;
1567    let mut fill_stack: Vec<bool> = Vec::new();
1568
1569    for op in &content.operations {
1570        let operands = &op.operands;
1571        match op.operator.as_str() {
1572            "q" => {
1573                gstate_stack.push((ctm, tc, tw, th, tl, trise, fsize, font));
1574                fill_stack.push(fill_light);
1575            }
1576            "Q" => {
1577                fill_light = fill_stack.pop().unwrap_or(false);
1578                if let Some((c, a, b, h, l, r, fs, f)) = gstate_stack.pop() {
1579                    ctm = c;
1580                    tc = a;
1581                    tw = b;
1582                    th = h;
1583                    tl = l;
1584                    trise = r;
1585                    fsize = fs;
1586                    font = f;
1587                }
1588            }
1589            "cm" => {
1590                let m = Mat {
1591                    a: op_f(operands, 0),
1592                    b: op_f(operands, 1),
1593                    c: op_f(operands, 2),
1594                    d: op_f(operands, 3),
1595                    e: op_f(operands, 4),
1596                    f: op_f(operands, 5),
1597                };
1598                ctm = m.then(ctm);
1599            }
1600            "BT" => {
1601                tm = Mat::ID;
1602                tlm = Mat::ID;
1603            }
1604            "ET" => {}
1605            "Tf" => {
1606                if let Some(Object::Name(n)) = operands.first() {
1607                    font = fonts.get(n.as_slice());
1608                }
1609                fsize = op_f(operands, 1);
1610            }
1611            "Td" => {
1612                tlm = Mat {
1613                    a: 1.0,
1614                    b: 0.0,
1615                    c: 0.0,
1616                    d: 1.0,
1617                    e: op_f(operands, 0),
1618                    f: op_f(operands, 1),
1619                }
1620                .then(tlm);
1621                tm = tlm;
1622            }
1623            "TD" => {
1624                tl = -op_f(operands, 1);
1625                tlm = Mat {
1626                    a: 1.0,
1627                    b: 0.0,
1628                    c: 0.0,
1629                    d: 1.0,
1630                    e: op_f(operands, 0),
1631                    f: op_f(operands, 1),
1632                }
1633                .then(tlm);
1634                tm = tlm;
1635            }
1636            "Tm" => {
1637                tlm = Mat {
1638                    a: op_f(operands, 0),
1639                    b: op_f(operands, 1),
1640                    c: op_f(operands, 2),
1641                    d: op_f(operands, 3),
1642                    e: op_f(operands, 4),
1643                    f: op_f(operands, 5),
1644                };
1645                tm = tlm;
1646            }
1647            "T*" => {
1648                tlm = Mat {
1649                    a: 1.0,
1650                    b: 0.0,
1651                    c: 0.0,
1652                    d: 1.0,
1653                    e: 0.0,
1654                    f: -tl,
1655                }
1656                .then(tlm);
1657                tm = tlm;
1658            }
1659            "Tc" => tc = op_f(operands, 0),
1660            "Tw" => tw = op_f(operands, 0),
1661            "Tz" => th = op_f(operands, 0) / 100.0,
1662            "TL" => tl = op_f(operands, 0),
1663            "Ts" => trise = op_f(operands, 0),
1664            "Tj" | "'" | "\"" => {
1665                if op.operator == "'" || op.operator == "\"" {
1666                    // move to next line first
1667                    tlm = Mat {
1668                        a: 1.0,
1669                        b: 0.0,
1670                        c: 0.0,
1671                        d: 1.0,
1672                        e: 0.0,
1673                        f: -tl,
1674                    }
1675                    .then(tlm);
1676                    tm = tlm;
1677                }
1678                if op.operator == "\"" {
1679                    // `aw ac string "` sets word- and char-spacing before
1680                    // showing the string (PDF 32000-1 §9.4.3), persisting after.
1681                    tw = op_f(operands, 0);
1682                    tc = op_f(operands, 1);
1683                }
1684                if let (Some(f), Some(Object::String(s, _))) = (font, operands.last()) {
1685                    show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out);
1686                }
1687            }
1688            "TJ" => {
1689                if let (Some(f), Some(Object::Array(arr))) = (font, operands.first()) {
1690                    for el in arr {
1691                        match el {
1692                            Object::String(s, _) => {
1693                                show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out)
1694                            }
1695                            other => {
1696                                if let Some(adj) = num(other) {
1697                                    // negative number moves text right (PDF: subtract)
1698                                    let tx = -adj / 1000.0 * fsize * th;
1699                                    tm = Mat {
1700                                        a: 1.0,
1701                                        b: 0.0,
1702                                        c: 0.0,
1703                                        d: 1.0,
1704                                        e: tx,
1705                                        f: 0.0,
1706                                    }
1707                                    .then(tm);
1708                                }
1709                            }
1710                        }
1711                    }
1712                }
1713            }
1714            "m" => {
1715                cur = ctm.apply(op_f(operands, 0), op_f(operands, 1));
1716                start = cur;
1717            }
1718            "l" => {
1719                let p = ctm.apply(op_f(operands, 0), op_f(operands, 1));
1720                path.push(PathPiece::Line(cur, p));
1721                cur = p;
1722            }
1723            "c" | "v" | "y" => {
1724                let pts: Vec<(f64, f64)> = (0..operands.len() / 2)
1725                    .map(|k| ctm.apply(op_f(operands, 2 * k), op_f(operands, 2 * k + 1)))
1726                    .collect();
1727                if let Some(&end) = pts.last() {
1728                    path.push(PathPiece::Curve(
1729                        std::iter::once(cur).chain(pts.iter().copied()).collect(),
1730                    ));
1731                    cur = end;
1732                }
1733            }
1734            "h" => {
1735                if cur != start {
1736                    path.push(PathPiece::Line(cur, start));
1737                }
1738                cur = start;
1739            }
1740            "re" => {
1741                let (x, y, w, h) = (
1742                    op_f(operands, 0),
1743                    op_f(operands, 1),
1744                    op_f(operands, 2),
1745                    op_f(operands, 3),
1746                );
1747                path.push(PathPiece::Rect([
1748                    ctm.apply(x, y),
1749                    ctm.apply(x + w, y),
1750                    ctm.apply(x + w, y + h),
1751                    ctm.apply(x, y + h),
1752                ]));
1753                cur = ctm.apply(x, y);
1754                start = cur;
1755            }
1756            "S" | "s" | "f" | "F" | "f*" | "B" | "B*" | "b" | "b*" => {
1757                let op = op.operator.as_str();
1758                if matches!(op, "s" | "b" | "b*") && cur != start {
1759                    path.push(PathPiece::Line(cur, start));
1760                }
1761                let stroke = matches!(op, "S" | "s" | "B" | "B*" | "b" | "b*");
1762                let fill = !matches!(op, "S" | "s");
1763                paint_path(&path, stroke, fill, fill_light, inks);
1764                path.clear();
1765            }
1766            "n" => path.clear(),
1767            // The fill colour, by its components: gray `g`, `rg`, CMYK `k`,
1768            // and `sc`/`scn` by operand count (a pattern name is not light).
1769            "g" | "rg" | "k" | "sc" | "scn" => {
1770                let v: Vec<f64> = operands
1771                    .iter()
1772                    .map(num)
1773                    .collect::<Option<_>>()
1774                    .unwrap_or_default();
1775                fill_light = match v.as_slice() {
1776                    [g] => *g >= 0.9,
1777                    [r, g, b] => r.min(*g).min(*b) >= 0.9,
1778                    [c, m, y, k] => c.max(*m).max(*y).max(*k) <= 0.1,
1779                    _ => false,
1780                };
1781            }
1782            "cs" => fill_light = false,
1783            "Do" => {
1784                // Invoke a Form XObject: bulk body text in many PDFs lives inside
1785                // a form, reached only here. Image XObjects are skipped (no text).
1786                if depth >= 8 {
1787                    continue;
1788                }
1789                let Some(Object::Name(n)) = operands.first() else {
1790                    continue;
1791                };
1792                let obj = xobjects.and_then(|d| d.get(n.as_slice()).ok());
1793                let form_id = match obj {
1794                    Some(Object::Reference(id)) => Some(*id),
1795                    _ => None,
1796                };
1797                let stream = obj
1798                    .and_then(|o| deref(doc, o))
1799                    .and_then(|o| o.as_stream().ok());
1800                let Some(stream) = stream else { continue };
1801                let is_form = stream
1802                    .dict
1803                    .get(b"Subtype")
1804                    .ok()
1805                    .and_then(|o| o.as_name().ok())
1806                    == Some(b"Form".as_slice());
1807                if !is_form {
1808                    continue;
1809                }
1810                // Decode the form's content once per document (headers/footers
1811                // and bulk body text invoke the same form on every page).
1812                let cached = form_id.and_then(|id| caches.forms.get(&id).cloned());
1813                let form_content = match cached {
1814                    Some(c) => c,
1815                    None => {
1816                        let Ok(data) = stream.decompressed_content() else {
1817                            continue;
1818                        };
1819                        let Ok(c) = lopdf::content::Content::decode(&data) else {
1820                            continue;
1821                        };
1822                        let c = Arc::new(c);
1823                        if let Some(id) = form_id {
1824                            caches.forms.insert(id, Arc::clone(&c));
1825                        }
1826                        c
1827                    }
1828                };
1829                // The form's /Matrix maps form space into the CTM at invocation.
1830                let form_mat = match stream.dict.get(b"Matrix").ok() {
1831                    Some(Object::Array(a)) if a.len() == 6 => {
1832                        let v: Vec<f64> = a.iter().filter_map(num).collect();
1833                        if v.len() == 6 {
1834                            Mat {
1835                                a: v[0],
1836                                b: v[1],
1837                                c: v[2],
1838                                d: v[3],
1839                                e: v[4],
1840                                f: v[5],
1841                            }
1842                        } else {
1843                            Mat::ID
1844                        }
1845                    }
1846                    _ => Mat::ID,
1847                };
1848                // The form's own /Resources, falling back to the inherited ones.
1849                let form_res = stream
1850                    .dict
1851                    .get(b"Resources")
1852                    .ok()
1853                    .and_then(|o| deref(doc, o))
1854                    .and_then(|o| o.as_dict().ok())
1855                    .unwrap_or(res);
1856                let state = TextState {
1857                    tc,
1858                    tw,
1859                    th,
1860                    tl,
1861                    trise,
1862                    fsize,
1863                };
1864                run_content(
1865                    doc,
1866                    form_res,
1867                    &form_content,
1868                    form_mat.then(ctm),
1869                    state,
1870                    depth + 1,
1871                    caches,
1872                    out,
1873                    inks,
1874                );
1875            }
1876            _ => {}
1877        }
1878    }
1879}
1880
1881/// One piece of a path under construction, page space (see `run_content`).
1882enum PathPiece {
1883    Line((f64, f64), (f64, f64)),
1884    /// An `re`: its four corners, counter-clockwise from `(x, y)`.
1885    Rect([(f64, f64); 4]),
1886    /// A Bézier segment: its start and control/end points (their hull).
1887    Curve(Vec<(f64, f64)>),
1888}
1889
1890/// Record a painted path's pieces for the checkbox finder (#609): a stroked
1891/// line is a [`Seg`](crate::checkbox::Ink::Seg), a stroked rectangle that
1892/// stays axis-aligned a [`Rect`](crate::checkbox::Ink::Rect) (one filled
1893/// white too — still an outline), and a coloured fill or a curve only its
1894/// bounding box, a [`Blot`](crate::checkbox::Ink::Blot): a possible tick
1895/// mark, or a swatch. White fills draw nothing and are dropped. Only pieces small
1896/// enough to be a checkbox edge or a mark inside one are kept.
1897fn paint_path(
1898    path: &[PathPiece],
1899    stroke: bool,
1900    fill: bool,
1901    fill_light: bool,
1902    inks: &mut Vec<crate::checkbox::Ink>,
1903) {
1904    use crate::checkbox::Ink;
1905    let bbox = |pts: &mut dyn Iterator<Item = (f64, f64)>| {
1906        pts.fold(
1907            (
1908                f64::INFINITY,
1909                f64::INFINITY,
1910                f64::NEG_INFINITY,
1911                f64::NEG_INFINITY,
1912            ),
1913            |(l, b, r, t), (x, y)| (l.min(x), b.min(y), r.max(x), t.max(y)),
1914        )
1915    };
1916    let mut keep = |ink: Ink| {
1917        if ink.worth_keeping() {
1918            inks.push(ink);
1919        }
1920    };
1921    if fill && !fill_light {
1922        // A coloured fill is a blot — a tick mark, or (the size of a square)
1923        // a legend swatch, which `checkbox::find` then rules out. A white
1924        // fill is no ink on paper: it only leaves the stroke, if any, below.
1925        let (l, b, r, t) = bbox(&mut path.iter().flat_map(|p| match p {
1926            PathPiece::Line(a, z) => vec![*a, *z],
1927            PathPiece::Rect(c) => c.to_vec(),
1928            PathPiece::Curve(c) => c.clone(),
1929        }));
1930        if l.is_finite() {
1931            keep(Ink::Blot { l, b, r, t });
1932        }
1933        return;
1934    }
1935    if !stroke {
1936        return;
1937    }
1938    for piece in path {
1939        match piece {
1940            PathPiece::Line(a, z) => keep(Ink::Seg {
1941                x0: a.0,
1942                y0: a.1,
1943                x1: z.0,
1944                y1: z.1,
1945            }),
1946            PathPiece::Rect(c) => {
1947                let axis = ((c[0].1 - c[1].1).abs() < 1e-6 && (c[1].0 - c[2].0).abs() < 1e-6)
1948                    || ((c[0].0 - c[1].0).abs() < 1e-6 && (c[1].1 - c[2].1).abs() < 1e-6);
1949                if axis {
1950                    let (l, b, r, t) = bbox(&mut c.iter().copied());
1951                    keep(Ink::Rect { l, b, r, t });
1952                } else {
1953                    for k in 0..4 {
1954                        let (a, z) = (c[k], c[(k + 1) % 4]);
1955                        keep(Ink::Seg {
1956                            x0: a.0,
1957                            y0: a.1,
1958                            x1: z.0,
1959                            y1: z.1,
1960                        });
1961                    }
1962                }
1963            }
1964            PathPiece::Curve(c) => {
1965                let (l, b, r, t) = bbox(&mut c.iter().copied());
1966                keep(Ink::Blot { l, b, r, t });
1967            }
1968        }
1969    }
1970}
1971
1972#[allow(clippy::too_many_arguments)]
1973fn show_text(
1974    font: &Font,
1975    bytes: &[u8],
1976    fsize: f64,
1977    tc: f64,
1978    tw: f64,
1979    th: f64,
1980    trise: f64,
1981    tm: &mut Mat,
1982    ctm: Mat,
1983    out: &mut Vec<Glyph>,
1984) {
1985    for code in codes(font, bytes) {
1986        let (text, w) = font.decode_code(code);
1987        let w0 = w / 1000.0; // advance in text-space (em) units
1988                             // The glyph→user transform: scale glyph space by font size, then Tm, CTM.
1989        let scale = Mat {
1990            a: fsize * th,
1991            b: 0.0,
1992            c: 0.0,
1993            d: fsize,
1994            e: 0.0,
1995            f: trise,
1996        };
1997        let trm = scale.then(*tm).then(ctm);
1998        // Box in glyph space (1000-unit em): x 0..w, y descent..ascent.
1999        let (desc, asc) = (font.descent / 1000.0, font.ascent / 1000.0);
2000        let (x0, y0) = trm.apply(0.0, desc);
2001        let (x1, y1) = trm.apply(w0, desc);
2002        let (x2, y2) = trm.apply(w0, asc);
2003        let (x3, y3) = trm.apply(0.0, asc);
2004        // Upright = the baseline runs left-to-right along +x. Only then is the
2005        // loose box the rectangle spanned by the baseline's x and the em's y
2006        // (a synthetic oblique's skew is ignored, as it always was). A rotated
2007        // matrix — the `0 s -s 0 tx ty Tm` landscape pages are built with
2008        // (#528) — collapsed that rectangle to zero width, and every such run
2009        // vanished; those glyphs keep their real quad (docling-parse's char
2010        // rect) and its extent instead.
2011        let upright = trm.a > 0.0 && trm.b.abs() <= 1e-6 * trm.a;
2012        let (left, bot, right, top, quad) = if upright {
2013            (x0.min(x1), y0.min(y3), x0.max(x1), y0.max(y3), None)
2014        } else {
2015            let (xs, ys) = ([x0, x1, x2, x3], [y0, y1, y2, y3]);
2016            let fold = |v: [f64; 4], f: fn(f64, f64) -> f64| v.into_iter().reduce(f).unwrap();
2017            let q = [x0, y0, x1, y1, x2, y2, x3, y3].map(|v| v as f32);
2018            (
2019                fold(xs, f64::min),
2020                fold(ys, f64::min),
2021                fold(xs, f64::max),
2022                fold(ys, f64::max),
2023                Some(q),
2024            )
2025        };
2026        if let Some(s) = text {
2027            // A run may map one code to multiple chars (ligature/fraction); share box.
2028            for ch in s.chars() {
2029                if ch != '\u{0}' {
2030                    out.push(Glyph {
2031                        ch,
2032                        l: left as f32,
2033                        b: bot as f32,
2034                        r: right as f32,
2035                        t: top as f32,
2036                        ll: left as f32,
2037                        lb: bot as f32,
2038                        lr: right as f32,
2039                        lt: top as f32,
2040                        font: font.hash,
2041                        quad,
2042                    });
2043                }
2044            }
2045        }
2046        // Advance the text matrix. Word spacing applies to single-byte code 32.
2047        let is_space = !font.two_byte && code == 32;
2048        let tx = (w0 * fsize + tc + if is_space { tw } else { 0.0 }) * th;
2049        *tm = Mat {
2050            a: 1.0,
2051            b: 0.0,
2052            c: 0.0,
2053            d: 1.0,
2054            e: tx,
2055            f: 0.0,
2056        }
2057        .then(*tm);
2058    }
2059}
2060
2061/// Build a simple font's code→char table from its `/Encoding`: the base
2062/// encoding (WinAnsi / MacRoman) plus any `/Differences` overrides (glyph names
2063/// resolved through a small Adobe-glyph-name subset).
2064fn simple_encoding_table(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
2065    let enc = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o));
2066    let base_name = match enc {
2067        Some(Object::Name(n)) => n.clone(),
2068        Some(Object::Dictionary(d)) => d
2069            .get(b"BaseEncoding")
2070            .ok()
2071            .and_then(|o| o.as_name().ok())
2072            .map(|n| n.to_vec())
2073            .unwrap_or_default(),
2074        _ => Vec::new(),
2075    };
2076    let mut m = if base_name == b"MacRomanEncoding" {
2077        macroman_table()
2078    } else if base_name.is_empty() {
2079        // No PDF /Encoding at all: the font's *built-in* encoding applies. For
2080        // the standard TeX math fonts that is their fixed TeX layout — falling
2081        // back to StandardEncoding read CMSY's braces as `f`/`g`, `→` as `!`,
2082        // `∈` as `2` (2203's `{ahn,…}` author line). The font program (often
2083        // CFF, which this parser does not read) carries the same mapping;
2084        // docling-parse decodes it from there.
2085        tex_math_builtin(fdict).unwrap_or_else(winansi_table)
2086    } else {
2087        winansi_table()
2088    };
2089    // Apply /Differences: `code /glyphname /glyphname ... code ...`.
2090    if let Some(Object::Dictionary(d)) = enc {
2091        if let Some(Object::Array(diffs)) = d.get(b"Differences").ok().and_then(|o| deref(doc, o)) {
2092            let mut code = 0u8;
2093            for el in diffs {
2094                match el {
2095                    Object::Integer(i) => code = *i as u8,
2096                    Object::Name(name) => {
2097                        if let Some(ch) = glyph_name_to_char(name) {
2098                            m.insert(code, ch);
2099                        }
2100                        code = code.wrapping_add(1);
2101                    }
2102                    _ => {}
2103                }
2104            }
2105        }
2106    }
2107    m
2108}
2109
2110/// The fixed built-in encodings of the standard TeX math fonts (TeXbook
2111/// Appendix F), keyed off the base font name: `CMSY*` (symbols; `CMBSY` is its
2112/// bold) and `CMMI*` (math italic). These fonts ship no PDF `/Encoding` and no
2113/// ToUnicode, and their program is usually CFF — without this table the codes
2114/// fell through to StandardEncoding and rendered as the wrong ASCII.
2115fn tex_math_builtin(fdict: &Dictionary) -> Option<HashMap<u8, char>> {
2116    const CMSY: [char; 128] = [
2117        '−', '·', '×', '∗', '÷', '⋄', '±', '∓', '⊕', '⊖', '⊗', '⊘', '⊙', '◯', '∘', '•', '≍', '≡',
2118        '⊆', '⊇', '≤', '≥', '≼', '≽', '∼', '≈', '⊂', '⊃', '≪', '≫', '≺', '≻', '←', '→', '↑', '↓',
2119        '↔', '↗', '↘', '≃', '⇐', '⇒', '⇑', '⇓', '⇔', '↖', '↙', '∝', '′', '∞', '∈', '∋', '△', '▽',
2120        '\u{338}', '↦', '∀', '∃', '¬', '∅', 'ℜ', 'ℑ', '⊤', '⊥', 'ℵ', 'A', 'B', 'C', 'D', 'E', 'F',
2121        'G', 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X',
2122        'Y', 'Z', '∪', '∩', '⊎', '∧', '∨', '⊢', '⊣', '⌊', '⌋', '⌈', '⌉', '{', '}', '⟨', '⟩', '|',
2123        '∥', '↕', '⇕', '\\', '≀', '√', '∐', '∇', '∫', '⊔', '⊓', '⊑', '⊒', '§', '†', '‡', '¶', '♣',
2124        '♢', '♡', '♠',
2125    ];
2126    const CMMI: [char; 128] = [
2127        'Γ', 'Δ', 'Θ', 'Λ', 'Ξ', 'Π', 'Σ', 'Υ', 'Φ', 'Ψ', 'Ω', 'α', 'β', 'γ', 'δ', 'ε', 'ζ', 'η',
2128        'θ', 'ι', 'κ', 'λ', 'μ', 'ν', 'ξ', 'π', 'ρ', 'σ', 'τ', 'υ', 'φ', 'χ', 'ψ', 'ω', 'ϵ', 'ϑ',
2129        'ϖ', 'ϱ', 'ς', 'ϕ', '↼', '↽', '⇀', '⇁', '↩', '↪', '▷', '◁', '0', '1', '2', '3', '4', '5',
2130        '6', '7', '8', '9', '.', ',', '<', '/', '>', '⋆', '∂', 'A', 'B', 'C', 'D', 'E', 'F', 'G',
2131        'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y',
2132        'Z', '♭', '♮', '♯', '⌣', '⌢', 'ℓ', 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k',
2133        'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', 'ı', 'ȷ', '℘',
2134        '\u{20d7}', '⁀',
2135    ];
2136    let name = base_font_name(fdict)?;
2137    let up = name.to_ascii_uppercase();
2138    let table: &[char; 128] = if up.starts_with(b"CMSY") || up.starts_with(b"CMBSY") {
2139        &CMSY
2140    } else if up.starts_with(b"CMMI") {
2141        &CMMI
2142    } else {
2143        return None;
2144    };
2145    Some(
2146        table
2147            .iter()
2148            .enumerate()
2149            .map(|(i, &c)| (i as u8, c))
2150            .collect(),
2151    )
2152}
2153
2154/// Resolve an Adobe glyph name to Unicode: `uniXXXX`, single ASCII letters, the
2155/// digit/punctuation names from the Adobe Glyph List, and common typographic
2156/// names. A `.suffix` (`one.taboldstyle`, `a.sc`) is stripped and the base name
2157/// retried — docling renders these as the base character.
2158pub(crate) fn glyph_name_to_char(name: &[u8]) -> Option<char> {
2159    let s = std::str::from_utf8(name).ok()?;
2160    if let Some(hex) = s.strip_prefix("uni") {
2161        if let Ok(cp) = u32::from_str_radix(hex.get(0..4)?, 16) {
2162            return char::from_u32(cp);
2163        }
2164    }
2165    // Single ASCII letter names (`A`, `m`) map to themselves.
2166    if s.len() == 1 {
2167        let b = s.as_bytes()[0];
2168        if b.is_ascii_alphabetic() {
2169            return Some(b as char);
2170        }
2171    }
2172    let resolved = match s {
2173        "space" => ' ',
2174        "exclam" => '!',
2175        "quotedbl" => '"',
2176        "numbersign" => '#',
2177        "dollar" => '$',
2178        "percent" => '%',
2179        "ampersand" => '&',
2180        "quotesingle" => '\'',
2181        "parenleft" => '(',
2182        "parenright" => ')',
2183        "asterisk" => '*',
2184        "plus" => '+',
2185        "comma" => ',',
2186        "hyphen" => '-',
2187        "period" => '.',
2188        "slash" => '/',
2189        "zero" => '0',
2190        "one" => '1',
2191        "two" => '2',
2192        "three" => '3',
2193        "four" => '4',
2194        "five" => '5',
2195        "six" => '6',
2196        "seven" => '7',
2197        "eight" => '8',
2198        "nine" => '9',
2199        "colon" => ':',
2200        "semicolon" => ';',
2201        "less" => '<',
2202        "equal" => '=',
2203        "greater" => '>',
2204        "question" => '?',
2205        "at" => '@',
2206        "bracketleft" => '[',
2207        "backslash" => '\\',
2208        "bracketright" => ']',
2209        "asciicircum" => '^',
2210        "underscore" => '_',
2211        "grave" => '`',
2212        "braceleft" => '{',
2213        "bar" => '|',
2214        "braceright" => '}',
2215        "asciitilde" => '~',
2216        "bullet" => '\u{2022}',
2217        "periodcentered" => '\u{00B7}',
2218        "endash" => '\u{2013}',
2219        "emdash" => '\u{2014}',
2220        "quoteright" => '\u{2019}',
2221        "quoteleft" => '\u{2018}',
2222        "quotedblleft" => '\u{201C}',
2223        "quotedblright" => '\u{201D}',
2224        "quotedblbase" => '\u{201E}',
2225        "quotesinglbase" => '\u{201A}',
2226        // Latin f-ligatures named in `/Differences` (e.g. 2305's body font). These
2227        // map to the presentation-form code points, which `decompose_ligatures`
2228        // then spells back out (`ff`→"ff") — without them the glyph decodes to
2229        // nothing and the sanitizer fills the gap with a space (`di erences`).
2230        "ff" => '\u{FB00}',
2231        "fi" => '\u{FB01}',
2232        "fl" => '\u{FB02}',
2233        "ffi" => '\u{FB03}',
2234        "ffl" => '\u{FB04}',
2235        "ft" => '\u{FB05}',
2236        "st" => '\u{FB06}',
2237        "degree" => '\u{00B0}',
2238        "trademark" => '\u{2122}',
2239        "registered" => '\u{00AE}',
2240        "copyright" => '\u{00A9}',
2241        "ellipsis" => '\u{2026}',
2242        "minus" => '\u{2212}',
2243        "fraction" => '\u{2044}',
2244        "nbspace" => '\u{00A0}',
2245        // Greek + math glyph names (standard Adobe Glyph List). Standard TeX math
2246        // fonts (CMMI/CMSY/…) name their glyphs this way in the embedded font
2247        // program's `/Encoding`; without these a `λ`/`≤` decodes to nothing and is
2248        // dropped from body text (`and λ set to 0.5` → `and set to 0.5`).
2249        "alpha" => '\u{03B1}',
2250        "beta" => '\u{03B2}',
2251        "gamma" => '\u{03B3}',
2252        "delta" => '\u{03B4}',
2253        "epsilon" | "epsilon1" => '\u{03B5}',
2254        "zeta" => '\u{03B6}',
2255        "eta" => '\u{03B7}',
2256        "theta" | "theta1" => '\u{03B8}',
2257        "iota" => '\u{03B9}',
2258        "kappa" => '\u{03BA}',
2259        "lambda" => '\u{03BB}',
2260        "mu" => '\u{03BC}',
2261        "nu" => '\u{03BD}',
2262        "xi" => '\u{03BE}',
2263        "omicron" => '\u{03BF}',
2264        "pi" | "pi1" => '\u{03C0}',
2265        "rho" | "rho1" => '\u{03C1}',
2266        "sigma" => '\u{03C3}',
2267        "sigma1" => '\u{03C2}',
2268        "tau" => '\u{03C4}',
2269        "upsilon" => '\u{03C5}',
2270        "phi" | "phi1" => '\u{03C6}',
2271        "chi" => '\u{03C7}',
2272        "psi" => '\u{03C8}',
2273        "omega" | "omega1" => '\u{03C9}',
2274        "Gamma" => '\u{0393}',
2275        "Delta" => '\u{0394}',
2276        "Theta" => '\u{0398}',
2277        "Lambda" => '\u{039B}',
2278        "Xi" => '\u{039E}',
2279        "Pi" => '\u{03A0}',
2280        "Sigma" => '\u{03A3}',
2281        "Upsilon" => '\u{03A5}',
2282        "Phi" => '\u{03A6}',
2283        "Psi" => '\u{03A8}',
2284        "Omega" => '\u{03A9}',
2285        "lessequal" => '\u{2264}',
2286        "greaterequal" => '\u{2265}',
2287        "notequal" => '\u{2260}',
2288        "approxequal" => '\u{2248}',
2289        "equivalence" => '\u{2261}',
2290        "element" => '\u{2208}',
2291        "plusminus" => '\u{00B1}',
2292        "multiply" => '\u{00D7}',
2293        "divide" => '\u{00F7}',
2294        "infinity" => '\u{221E}',
2295        "partialdiff" => '\u{2202}',
2296        "gradient" => '\u{2207}',
2297        "summation" => '\u{2211}',
2298        "product" => '\u{220F}',
2299        "integral" => '\u{222B}',
2300        "radical" => '\u{221A}',
2301        "proportional" => '\u{221D}',
2302        "arrowright" => '\u{2192}',
2303        "arrowleft" => '\u{2190}',
2304        "arrowup" => '\u{2191}',
2305        "arrowdown" => '\u{2193}',
2306        "arrowboth" => '\u{2194}',
2307        "arrowdblright" => '\u{21D2}',
2308        "logicaland" => '\u{2227}',
2309        "logicalor" => '\u{2228}',
2310        "intersection" => '\u{2229}',
2311        "union" => '\u{222A}',
2312        "similar" => '\u{223C}',
2313        "congruent" => '\u{2245}',
2314        "dotmath" => '\u{22C5}',
2315        "asteriskmath" => '\u{2217}',
2316        _ => {
2317            // Strip an AGL `.suffix` (oldstyle/small-cap variant) and retry.
2318            if let Some((base, _)) = s.split_once('.') {
2319                if !base.is_empty() {
2320                    return glyph_name_to_char(base.as_bytes());
2321                }
2322            }
2323            return None;
2324        }
2325    };
2326    Some(resolved)
2327}
2328
2329/// Minimal WinAnsiEncoding (Latin-1-ish) for simple fonts lacking ToUnicode.
2330fn winansi_table() -> HashMap<u8, char> {
2331    let mut m = HashMap::new();
2332    for b in 0x20u8..=0x7e {
2333        m.insert(b, b as char);
2334    }
2335    // High range: Windows-1252 printable points that differ from Latin-1.
2336    let extra: &[(u8, char)] = &[
2337        (0x91, '\u{2018}'),
2338        (0x92, '\u{2019}'),
2339        (0x93, '\u{201C}'),
2340        (0x94, '\u{201D}'),
2341        (0x95, '\u{2022}'),
2342        (0x96, '\u{2013}'),
2343        (0x97, '\u{2014}'),
2344        (0x85, '\u{2026}'),
2345        (0xA0, '\u{00A0}'),
2346    ];
2347    for &(b, c) in extra {
2348        m.insert(b, c);
2349    }
2350    for b in 0xA1u8..=0xFF {
2351        m.entry(b).or_insert(b as char);
2352    }
2353    m
2354}
2355
2356/// Minimal MacRomanEncoding: ASCII plus the high-range points our corpus hits
2357/// (notably 0xA5 = bullet, used as a list marker).
2358fn macroman_table() -> HashMap<u8, char> {
2359    let mut m = HashMap::new();
2360    for b in 0x20u8..=0x7e {
2361        m.insert(b, b as char);
2362    }
2363    let high: &[(u8, char)] = &[
2364        (0xA5, '\u{2022}'), // bullet
2365        (0xD0, '\u{2013}'), // endash
2366        (0xD1, '\u{2014}'), // emdash
2367        (0xD2, '\u{201C}'),
2368        (0xD3, '\u{201D}'),
2369        (0xD4, '\u{2018}'),
2370        (0xD5, '\u{2019}'),
2371        (0xCA, '\u{00A0}'),
2372        (0xC9, '\u{2026}'),
2373        (0xDE, '\u{FB01}'),
2374        (0xDF, '\u{FB02}'),
2375    ];
2376    for &(b, c) in high {
2377        m.insert(b, c);
2378    }
2379    m
2380}
2381
2382#[cfg(test)]
2383mod page_box_frame {
2384    use super::*;
2385
2386    /// One page, `boxes` spliced into the page dictionary verbatim, one text
2387    /// run at user-space `(x, y)`.
2388    fn pdf(boxes: &str, x: f32, y: f32) -> Vec<u8> {
2389        pdf_content(
2390            boxes,
2391            &format!("BT /F1 12 Tf {x} {y} Td (First printing) Tj ET\n"),
2392        )
2393    }
2394
2395    /// One page, `boxes` spliced into the page dictionary, `content` as its
2396    /// content stream (font `/F1` = Helvetica).
2397    fn pdf_content(boxes: &str, content: &str) -> Vec<u8> {
2398        let objs: Vec<String> = vec![
2399            "<</Type/Catalog/Pages 2 0 R>>".into(),
2400            format!("<</Type/Pages/Kids[3 0 R]/Count 1{boxes}>>"),
2401            "<</Type/Page/Parent 2 0 R/Contents 4 0 R/Resources<</Font<</F1 5 0 R>>>>>>".into(),
2402            format!("<</Length {}>>stream\n{content}endstream", content.len()),
2403            "<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".into(),
2404        ];
2405        let mut out = b"%PDF-1.4\n".to_vec();
2406        let mut offsets = Vec::new();
2407        for (i, body) in objs.iter().enumerate() {
2408            offsets.push(out.len());
2409            out.extend_from_slice(format!("{} 0 obj{body}endobj\n", i + 1).as_bytes());
2410        }
2411        let xref_at = out.len();
2412        out.extend_from_slice(
2413            format!("xref\n0 {}\n0000000000 65535 f \n", objs.len() + 1).as_bytes(),
2414        );
2415        for off in &offsets {
2416            out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2417        }
2418        out.extend_from_slice(
2419            format!(
2420                "trailer<</Size {}/Root 1 0 R>>\nstartxref\n{xref_at}\n%%EOF\n",
2421                objs.len() + 1
2422            )
2423            .as_bytes(),
2424        );
2425        out
2426    }
2427
2428    fn only_page(bytes: &[u8]) -> (PageBox, Vec<Glyph>) {
2429        let doc = load_document(bytes).expect("loads");
2430        let pid = *doc.get_pages().values().next().expect("one page");
2431        (page_box(&doc, pid), page_glyphs(&doc, pid))
2432    }
2433
2434    /// The trimmed-book-page shape: MediaBox and CropBox both away from the
2435    /// origin (inherited from `/Pages`). The display box is the CropBox, and a
2436    /// glyph's coordinates count from its lower-left corner — the same numbers
2437    /// the same text gets on a page whose CropBox *is* `[0 0 w h]`.
2438    #[test]
2439    fn glyphs_count_from_the_cropbox_corner_like_pdfium() {
2440        let (pb, shifted) = only_page(&pdf(
2441            "/MediaBox[-56.505 -58.25 576.303 723.31]/CropBox[1.095 -0.65 518.703 665.71]",
2442            37.0 + 1.095,
2443            58.0 - 0.65,
2444        ));
2445        assert!((pb.l - 1.095).abs() < 1e-3 && (pb.b + 0.65).abs() < 1e-3);
2446        assert!(
2447            (pb.w - 517.608).abs() < 1e-3 && (pb.h - 666.36).abs() < 1e-3,
2448            "{pb:?}"
2449        );
2450        let (pb0, plain) = only_page(&pdf("/MediaBox[0 0 517.608 666.36]", 37.0, 58.0));
2451        assert!((pb0.w - pb.w).abs() < 1e-3 && (pb0.h - pb.h).abs() < 1e-3);
2452        assert_eq!(shifted.len(), plain.len());
2453        assert!(!plain.is_empty());
2454        for (a, b) in shifted.iter().zip(&plain) {
2455            assert!(
2456                (a.l - b.l).abs() < 1e-3 && (a.b - b.b).abs() < 1e-3,
2457                "{:?} vs {:?}",
2458                (a.l, a.b),
2459                (b.l, b.b)
2460            );
2461        }
2462        assert!((plain[0].l - 37.0).abs() < 1e-3, "{}", plain[0].l);
2463    }
2464
2465    fn text(glyphs: &[Glyph]) -> String {
2466        glyphs.iter().map(|g| g.ch).collect()
2467    }
2468
2469    /// #529: text drawn outside the display box — a print slug beside the
2470    /// CropBox, a tiled page's neighbour beyond the MediaBox — is dropped,
2471    /// glyph by glyph like docling-parse, instead of being clamped onto the
2472    /// page edge; a line crossing the edge keeps the glyphs that fit.
2473    #[test]
2474    fn glyphs_outside_the_display_box_are_dropped() {
2475        let (_, g) = only_page(&pdf_content(
2476            "/MediaBox[0 0 400 400]/CropBox[100 100 400 400]",
2477            "BT /F1 14 Tf 120 300 Td (Visible.) Tj ET\nBT /F1 8 Tf 5 40 Td (Slug) Tj ET\n",
2478        ));
2479        assert_eq!(text(&g), "Visible.");
2480        let (_, g) = only_page(&pdf_content(
2481            "/MediaBox[0 0 300 400]",
2482            "BT /F1 14 Tf 40 300 Td (On page) Tj ET\nBT /F1 14 Tf 400 250 Td (Beyond) Tj ET\n",
2483        ));
2484        assert_eq!(text(&g), "On page");
2485        // docling-parse: "Crossing the right edge" on a 300 pt page → "Crossing the rig".
2486        let (_, g) = only_page(&pdf_content(
2487            "/MediaBox[0 0 300 400]",
2488            "BT /F1 14 Tf 200 250 Td (Crossing the right edge) Tj ET\n",
2489        ));
2490        assert_eq!(text(&g), "Crossing the rig");
2491    }
2492
2493    /// The containment test is docling-parse's: the whole char box (advance
2494    /// × the font's ascent / descent) inside the display box, edges
2495    /// included — a Helvetica `W` at 50 pt (47.2 wide) ending exactly on the
2496    /// right edge stays, 0.01 pt further it goes; likewise on the left.
2497    #[test]
2498    fn a_glyph_on_the_edge_stays_one_over_it_goes() {
2499        let w_at = |x: f32, y: f32| {
2500            let (_, g) = only_page(&pdf_content(
2501                "/MediaBox[0 0 300 400]",
2502                &format!("BT /F1 50 Tf {x} {y} Td (W) Tj ET\n"),
2503            ));
2504            text(&g)
2505        };
2506        assert_eq!(w_at(252.8, 200.0), "W");
2507        assert_eq!(w_at(252.81, 200.0), "");
2508        assert_eq!(w_at(0.0, 200.0), "W");
2509        assert_eq!(w_at(-0.01, 200.0), "");
2510        // Vertically the box is the font's ascent / descent: a baseline far
2511        // enough up keeps the descender on the page, one at 0 does not.
2512        assert_eq!(w_at(100.0, 20.0), "W");
2513        assert_eq!(w_at(100.0, 0.0), "");
2514        assert_eq!(w_at(100.0, 380.0), "");
2515    }
2516
2517    /// pdfium's fallbacks: no MediaBox → Letter; a CropBox is clipped to the
2518    /// MediaBox, and one that misses it entirely is ignored.
2519    #[test]
2520    fn page_box_follows_pdfium_fallbacks() {
2521        let (pb, _) = only_page(&pdf("", 10.0, 10.0));
2522        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 612.0, 792.0));
2523        let (pb, _) = only_page(&pdf(
2524            "/MediaBox[0 0 500 700]/CropBox[-100 100 600 900]",
2525            10.0,
2526            10.0,
2527        ));
2528        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 100.0, 500.0, 600.0));
2529        let (pb, _) = only_page(&pdf(
2530            "/MediaBox[0 0 500 700]/CropBox[800 800 900 900]",
2531            10.0,
2532            10.0,
2533        ));
2534        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 500.0, 700.0));
2535        // Reversed corners normalize.
2536        let (pb, _) = only_page(&pdf("/MediaBox[500 700 0 0]", 10.0, 10.0));
2537        assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 500.0, 700.0));
2538    }
2539}
2540
2541#[cfg(test)]
2542mod xref_repair {
2543    /// Build a tiny one-page PDF whose cross-reference entries are either the
2544    /// spec's 20 bytes (`two_byte_eol`) or the 19-byte form some generators
2545    /// emit — everything else about the two files is identical.
2546    fn pdf_with_xref(two_byte_eol: bool) -> Vec<u8> {
2547        let content = b"BT /F1 12 Tf 72 700 Td (Invoice 922769430725) Tj ET\n";
2548        let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2549        let objs: Vec<Vec<u8>> = vec![
2550            b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2551            b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2552            b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2553               /Resources<</Font<</F1 5 0 R>>>>>>"
2554                .to_vec(),
2555            [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
2556            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".to_vec(),
2557        ];
2558
2559        let mut out = b"%PDF-1.4\n".to_vec();
2560        let mut offsets = Vec::new();
2561        for (i, body) in objs.iter().enumerate() {
2562            offsets.push(out.len());
2563            out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2564            out.extend_from_slice(body);
2565            out.extend_from_slice(b"endobj\n");
2566        }
2567        let xref_at = out.len();
2568        let eol: &[u8] = if two_byte_eol { b" \n" } else { b"\n" };
2569        out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2570        out.extend_from_slice(b"0000000000 65535 f");
2571        out.extend_from_slice(eol);
2572        for off in &offsets {
2573            out.extend_from_slice(format!("{off:010} 00000 n").as_bytes());
2574            out.extend_from_slice(eol);
2575        }
2576        out.extend_from_slice(
2577            format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2578        );
2579        out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2580        out
2581    }
2582
2583    /// A 19-byte cross-reference entry (a bare LF where the spec wants a
2584    /// two-byte EOL) makes lopdf reject the whole file, so a readable text
2585    /// layer used to look exactly like a scan — in the browser that meant ten
2586    /// seconds of OCR for nothing. The repair must recover *the same* parse the
2587    /// well-formed file gives.
2588    #[test]
2589    fn short_xref_entries_still_parse() {
2590        let good = pdf_with_xref(true);
2591        let broken = pdf_with_xref(false);
2592        assert!(
2593            broken.len() < good.len(),
2594            "the broken file is the shorter one"
2595        );
2596        assert!(
2597            lopdf::Document::load_mem(&good).is_ok(),
2598            "the control file must load unaided"
2599        );
2600        assert!(
2601            lopdf::Document::load_mem(&broken).is_err(),
2602            "lopdf rejects 19-byte entries — if this ever passes, drop the repair"
2603        );
2604
2605        let cells = |b: &[u8]| -> Vec<String> {
2606            super::pdf_textlines(b)
2607                .into_iter()
2608                .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2609                .collect()
2610        };
2611        let from_good = cells(&good);
2612        assert!(
2613            from_good.iter().any(|t| t.contains("922769430725")),
2614            "control text: {from_good:?}"
2615        );
2616        assert_eq!(
2617            cells(&broken),
2618            from_good,
2619            "repair must match the good parse"
2620        );
2621    }
2622
2623    /// The same generator overstates `/Length`, so lopdf reads past the data,
2624    /// misses `endstream` and drops the stream — the object comes back as a
2625    /// bare dictionary and the page has no content at all. Trust `endstream`
2626    /// instead, and do it without moving a single byte.
2627    #[test]
2628    fn overstated_stream_length_still_yields_content() {
2629        let good = pdf_with_xref(true);
2630        // Inflate the content stream's /Length by one, exactly as the invoice
2631        // that prompted this does.
2632        let broken = {
2633            let at = good
2634                .windows(8)
2635                .position(|w| w == b"/Length ")
2636                .expect("a /Length")
2637                + 8;
2638            let digits = good[at..].iter().take_while(|c| c.is_ascii_digit()).count();
2639            let n: usize = std::str::from_utf8(&good[at..at + digits])
2640                .unwrap()
2641                .parse()
2642                .unwrap();
2643            let inflated = (n + 1).to_string();
2644            assert_eq!(inflated.len(), digits, "keep the digit count");
2645            let mut b = good.clone();
2646            b[at..at + digits].copy_from_slice(inflated.as_bytes());
2647            b
2648        };
2649        assert_eq!(broken.len(), good.len(), "the defect must not move bytes");
2650        // lopdf alone loses the stream: the page parses but carries no content.
2651        let raw = lopdf::Document::load_mem(&broken).expect("still loads");
2652        assert!(
2653            raw.get_pages()
2654                .into_values()
2655                .all(|p| raw.get_page_content(p).is_empty()),
2656            "lopdf should drop the stream — if it stops, drop this repair"
2657        );
2658        // Ours recovers the same text the well-formed file gives.
2659        let text = |b: &[u8]| -> Vec<String> {
2660            super::pdf_textlines(b)
2661                .into_iter()
2662                .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2663                .collect()
2664        };
2665        let expected = text(&good);
2666        assert!(!expected.is_empty(), "control must produce text");
2667        assert_eq!(text(&broken), expected);
2668    }
2669
2670    /// The repair only fires where padding cannot move an object: it declines a
2671    /// file whose xref precedes an object (an incremental update), rather than
2672    /// shifting every offset the table records.
2673    #[test]
2674    fn repair_declines_when_padding_would_move_objects() {
2675        let mut incremental = pdf_with_xref(false);
2676        incremental.extend_from_slice(b"6 0 obj<</Type/Whatever>>endobj\n");
2677        let declined = super::pad_short_xref_entries(&incremental).unwrap_err();
2678        assert!(
2679            declined.contains("object follows the xref"),
2680            "reason: {declined}"
2681        );
2682    }
2683}
2684
2685/// #187: standard-14 fonts referenced without an embedded program (and thus
2686/// usually without `/Widths` or a `/FontDescriptor`) must decode with the
2687/// built-in Adobe Core 14 metrics instead of collapsing every cell to zero
2688/// width — the failure mode where a valid text layer was silently dropped
2689/// while pdfium read the same file fine.
2690#[cfg(test)]
2691mod base14_fonts {
2692    /// A one-page PDF whose single `Tj` uses `fontdict` (no embedded program).
2693    fn pdf_with_font(fontdict: &[u8], text: &[u8]) -> Vec<u8> {
2694        let content = [b"BT /F1 12 Tf 72 700 Td (".as_slice(), text, b") Tj ET\n"].concat();
2695        pdf_with_content(fontdict, &content)
2696    }
2697
2698    /// A one-page A4 PDF drawing `content` with `fontdict` as `/F1`.
2699    fn pdf_with_content(fontdict: &[u8], content: &[u8]) -> Vec<u8> {
2700        let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2701        let objs: Vec<Vec<u8>> = vec![
2702            b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2703            b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2704            b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2705               /Resources<</Font<</F1 5 0 R>>>>>>"
2706                .to_vec(),
2707            [stream.as_slice(), content, b"endstream"].concat(),
2708            fontdict.to_vec(),
2709        ];
2710        let mut out = b"%PDF-1.4\n".to_vec();
2711        let mut offsets = Vec::new();
2712        for (i, body) in objs.iter().enumerate() {
2713            offsets.push(out.len());
2714            out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2715            out.extend_from_slice(body);
2716            out.extend_from_slice(b"endobj\n");
2717        }
2718        let xref_at = out.len();
2719        out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2720        out.extend_from_slice(b"0000000000 65535 f \n");
2721        for off in &offsets {
2722            out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2723        }
2724        out.extend_from_slice(
2725            format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2726        );
2727        out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2728        out
2729    }
2730
2731    /// The parsed cells of the only page.
2732    fn cells(pdf: &[u8]) -> Vec<crate::pdfium_backend::TextCell> {
2733        super::pdf_textlines(pdf)
2734            .into_iter()
2735            .flat_map(|(_, _, c)| c)
2736            .collect()
2737    }
2738
2739    /// Text drawn with a rotated text matrix reads as one line in its own
2740    /// reading order, boxed by the run's axis-aligned extent (#528). The
2741    /// landscape case is `0 s -s 0 tx ty Tm` on a `/Rotate 90` page; every
2742    /// such run used to collapse to zero width and vanish (90°/270°/tilted)
2743    /// or read backwards (180°). Expected boxes are docling-parse 7.22's line
2744    /// cells for the same runs (y-up, PDF points).
2745    #[test]
2746    fn rotated_text_matrix_reads_in_order() {
2747        const HELV: &[u8] = b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>";
2748        const TEXT: &str = "Upright text on a rotated page.";
2749        for (tm, [l, b, r, t]) in [
2750            ("0 14 -14 0 200 100", [189.9, 100.0, 202.9, 289.1]), // 90°
2751            ("-14 0 0 -14 350 300", [160.9, 289.9, 350.0, 302.9]), // 180°
2752            ("0 -14 14 0 200 500", [197.1, 310.9, 210.1, 500.0]), // 270°
2753            (
2754                "9.8995 9.8995 -9.8995 9.8995 100 100",
2755                [92.9, 98.0, 235.8, 240.8],
2756            ), // 45°
2757        ] {
2758            let content = format!("BT /F1 1 Tf {tm} Tm ({TEXT}) Tj ET\n");
2759            let pdf = pdf_with_content(HELV, content.as_bytes());
2760            let cs = cells(&pdf);
2761            assert_eq!(cs.len(), 1, "{tm}: {cs:?}");
2762            let c = &cs[0];
2763            assert_eq!(c.text, TEXT, "{tm}");
2764            // `cells` is top-left-origin on the 842 pt page. The descriptor-
2765            // less base-14 face gets the parser's 1-em box where docling-parse
2766            // reads Helvetica's AFM (718/−207) — the same ≤ 0.6 pt
2767            // across-the-baseline offset upright text has — so allow 1 pt.
2768            let got = [c.l, 842.0 - c.b, c.r, 842.0 - c.t];
2769            for (g, w) in got.iter().zip([l, b, r, t]) {
2770                assert!(
2771                    (g - w).abs() < 1.0,
2772                    "{tm}: box {got:?} vs docling-parse {:?}",
2773                    [l, b, r, t]
2774                );
2775            }
2776            let words: Vec<String> = super::pdf_words(&pdf)
2777                .into_iter()
2778                .flat_map(|(_, _, c)| c)
2779                .map(|c| c.text)
2780                .collect();
2781            assert_eq!(words, TEXT.split(' ').collect::<Vec<_>>(), "{tm}");
2782        }
2783    }
2784
2785    /// A rotated run past the page edge (a cover's spine title) is dropped
2786    /// by the on-page test (#529) — measured on the quad's extent, not the
2787    /// zero-width box rotated glyphs used to get — instead of clamping onto
2788    /// the edge as a zero-width cell. One inside the page is kept.
2789    #[test]
2790    fn rotated_text_off_the_page_is_dropped() {
2791        const HELV: &[u8] = b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>";
2792        let off = pdf_with_content(
2793            HELV,
2794            b"BT /F1 1 Tf 0 14 -14 0 -4 200 Tm (Spine title) Tj ET\n",
2795        );
2796        assert!(cells(&off).is_empty(), "{:?}", cells(&off));
2797        let on = pdf_with_content(
2798            HELV,
2799            b"BT /F1 1 Tf 0 14 -14 0 30 200 Tm (Margin note) Tj ET\n",
2800        );
2801        assert_eq!(cells(&on).len(), 1);
2802    }
2803
2804    /// Upright text keeps the plain loose rectangle — no quad, and the
2805    /// sanitizer path is the one every pinned PDF baseline was made with.
2806    #[test]
2807    fn upright_glyphs_carry_no_quad() {
2808        let only_page = |pdf: &[u8]| {
2809            let doc = super::load_document(pdf).expect("loads");
2810            let pid = *doc.get_pages().values().next().expect("one page");
2811            super::page_glyphs(&doc, pid)
2812        };
2813        let pdf = pdf_with_font(b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>", b"Plain");
2814        let glyphs = only_page(&pdf);
2815        assert!(!glyphs.is_empty());
2816        assert!(glyphs.iter().all(|g| g.quad.is_none() && g.lr > g.ll));
2817        // A 90° glyph's style height is its font size, not its advance.
2818        let pdf = pdf_with_content(
2819            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>",
2820            b"BT /F1 1 Tf 0 14 -14 0 200 100 Tm (W) Tj ET\n",
2821        );
2822        let glyphs = only_page(&pdf);
2823        let g = &glyphs[0];
2824        assert!(g.quad.is_some());
2825        // 14 pt × the face's 1-em box (no descriptor: ascent − descent = 1).
2826        assert!((g.height() - 14.0).abs() < 0.05, "{}", g.height());
2827    }
2828
2829    /// #609: ReportLab's chart axis sets each tick label with the *same* `Tm`
2830    /// inside its own `q … cm … Q`, so only the CTM stacks `3`…`8` one under
2831    /// another at one x. Each tick is its own line cell with its own box (as
2832    /// docling-parse reads it), not one `345678` cell boxed like the first.
2833    #[test]
2834    fn ticks_stacked_by_cm_are_separate_line_cells() {
2835        let mut content = String::new();
2836        for (k, tick) in "345678".chars().enumerate() {
2837            let y = 300.0 + 25.92 * k as f64;
2838            content.push_str(&format!(
2839                "q 1 0 0 1 125 {y} cm BT /F1 10 Tf 1 0 0 1 -5 -4 Tm ({tick}) Tj ET Q\n"
2840            ));
2841        }
2842        let pdf = pdf_with_content(
2843            b"<</Type/Font/Subtype/Type1/BaseFont/Times-Roman>>",
2844            content.as_bytes(),
2845        );
2846        let cells = cells(&pdf);
2847        let texts: Vec<&str> = cells.iter().map(|c| c.text.as_str()).collect();
2848        assert_eq!(texts, ["3", "4", "5", "6", "7", "8"]);
2849        for (k, c) in cells.iter().enumerate() {
2850            // Top-left y of a tick whose baseline is 4 pt under its cm origin.
2851            let base = 842.0 - (300.0 + 25.92 * k as f32 - 4.0);
2852            assert!((c.l - 120.0).abs() < 0.1, "{k}: l {}", c.l);
2853            assert!(
2854                c.t < base && base - c.t < 10.0,
2855                "{k}: t {} base {base}",
2856                c.t
2857            );
2858        }
2859    }
2860
2861    /// #609: the path walk finds drawn checkbox squares — ReportLab's four
2862    /// `m … l S` edges under a `cm`, and a stroked `re` with a tick inside
2863    /// (checked) — in the text cells' top-left frame, and a stroked box
2864    /// filled white; a filled bar, a big frame and a colour-filled legend
2865    /// swatch are not checkboxes.
2866    #[test]
2867    fn drawn_checkbox_squares_are_found() {
2868        let content = b"q 1 0 0 1 119.52 449.04 cm \
2869            n 0 12.96 m 12.96 12.96 l S n 0 0 m 12.96 0 l S \
2870            n 0 0 m 0 12.96 l S n 12.96 0 m 12.96 12.96 l S Q\n\
2871            200 400 10 10 re S 202 405 m 204.5 402 l 208.5 408.5 l S\n\
2872            300 400 12 12 re f 50 50 100 100 re S\n\
2873            q 0.2 0.4 0.8 rg 400 400 12 12 re B Q q 1 g 500 400 12 12 re B Q\n";
2874        let pdf = pdf_with_content(b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>", content);
2875        let mut parser = super::PageTextParser::open(&pdf).expect("parses");
2876        let mut boxes = parser.cells(0).checkboxes;
2877        boxes.sort_by(|a, b| a.l.total_cmp(&b.l));
2878        let got: Vec<([i32; 4], bool)> = boxes
2879            .iter()
2880            .map(|c| {
2881                let r = |v: f32| (v * 100.0).round() as i32;
2882                ([r(c.l), r(c.t), r(c.r), r(c.b)], c.checked)
2883            })
2884            .collect();
2885        // Page height 842: y-up 449.04..462.0 → top-left 380.0..392.96.
2886        assert_eq!(
2887            got,
2888            [
2889                ([11952, 38000, 13248, 39296], false),
2890                ([20000, 43200, 21000, 44200], true),
2891                ([50000, 43000, 51200, 44200], false),
2892            ]
2893        );
2894    }
2895
2896    /// Every standard-14 alias/style decodes with real (positive-width) boxes.
2897    #[test]
2898    fn standard14_faces_get_builtin_widths() {
2899        for fontdict in [
2900            // ReportLab's default: base-14 Helvetica, WinAnsi, nothing else.
2901            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>".as_slice(),
2902            // No /Encoding at all (StandardEncoding-ish default).
2903            b"<</Type/Font/Subtype/Type1/BaseFont/Times-BoldItalic>>",
2904            // Substitution aliases + a subset prefix.
2905            b"<</Type/Font/Subtype/TrueType/BaseFont/Arial,Bold>>",
2906            b"<</Type/Font/Subtype/Type1/BaseFont/ABCDEF+Courier-Oblique>>",
2907        ] {
2908            let pdf = pdf_with_font(fontdict, b"Words have width now");
2909            let cs = cells(&pdf);
2910            let text: String = cs
2911                .iter()
2912                .map(|c| c.text.as_str())
2913                .collect::<Vec<_>>()
2914                .join(" ");
2915            assert!(
2916                text.contains("Words have width now"),
2917                "{}: text lost: {text:?}",
2918                String::from_utf8_lossy(fontdict)
2919            );
2920            assert!(
2921                cs.iter().all(|c| c.r > c.l),
2922                "{}: zero-width cells: {cs:?}",
2923                String::from_utf8_lossy(fontdict)
2924            );
2925        }
2926    }
2927
2928    /// An explicit `/Widths` array always wins over the built-in metrics, and a
2929    /// non-standard face without `/Widths` stays as before (no invented boxes).
2930    #[test]
2931    fn explicit_widths_win_and_unknown_faces_are_untouched() {
2932        // Helvetica with explicit 100/1000-em widths: the word's box must be
2933        // ~4×100 units at 12pt = 4.8pt wide — far narrower than the ~2.7×
2934        // wider built-in Helvetica advances would make it.
2935        let explicit = pdf_with_font(
2936            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/FirstChar 65\
2937               /Widths[100 100 100 100]/Encoding/WinAnsiEncoding>>",
2938            b"ABBA",
2939        );
2940        let builtin = pdf_with_font(
2941            b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>",
2942            b"ABBA",
2943        );
2944        let w = |pdf: &[u8]| {
2945            let cs = cells(pdf);
2946            assert_eq!(cs.len(), 1, "one word cell: {cs:?}");
2947            cs[0].r - cs[0].l
2948        };
2949        let (we, wb) = (w(&explicit), w(&builtin));
2950        assert!(
2951            (we - 4.8).abs() < 0.1,
2952            "explicit widths must win: got {we}, want 4×100×12/1000"
2953        );
2954        assert!(
2955            wb > 2.0 * we,
2956            "built-in Helvetica is much wider: {wb} vs {we}"
2957        );
2958
2959        // An unknown face with no /Widths: still parses (text kept), but no
2960        // built-in table applies — the old zero-width behavior is preserved
2961        // rather than inventing Helvetica metrics for an arbitrary font.
2962        let unknown = pdf_with_font(
2963            b"<</Type/Font/Subtype/Type1/BaseFont/FancyCorp-Display>>",
2964            b"Mystery",
2965        );
2966        let cs = cells(&unknown);
2967        let text: String = cs.iter().map(|c| c.text.as_str()).collect();
2968        assert!(text.contains("Mystery"), "text still decodes: {cs:?}");
2969    }
2970}
2971
2972#[cfg(test)]
2973mod overpainted {
2974    use crate::pdfium_backend::TextCell;
2975
2976    fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
2977        TextCell {
2978            text: text.into(),
2979            l,
2980            t,
2981            r,
2982            b,
2983        }
2984    }
2985
2986    /// The reporting invoice's logo: a `"` painted inside a `==` on one band —
2987    /// artwork drawn with glyphs. Both cells go; the real text on the next
2988    /// band stays.
2989    #[test]
2990    fn stacked_logo_glyphs_are_dropped() {
2991        let mut cells = vec![
2992            cell("\"", 72.7, 21.5, 86.4, 31.5),
2993            cell("==", 59.4, 21.5, 99.6, 31.5),
2994            cell("Herr", 65.2, 151.3, 81.7, 161.3),
2995        ];
2996        super::drop_overpainted_cells(&mut cells);
2997        assert_eq!(cells.len(), 1, "cells: {cells:?}");
2998        assert_eq!(cells[0].text, "Herr");
2999    }
3000
3001    /// Adjacent words on a line touch but never contain each other — prose is
3002    /// untouched, and so is a same-text near-duplicate (double-drawn faux
3003    /// bold), which is not evidence of artwork.
3004    #[test]
3005    fn prose_and_double_draw_are_kept() {
3006        let mut cells = vec![
3007            cell("Telefon", 354.3, 133.2, 381.5, 143.2),
3008            cell("0676/2000", 387.3, 133.2, 428.7, 143.2),
3009            cell("Bold", 100.0, 50.0, 130.0, 60.0),
3010            cell("Bold", 100.3, 50.0, 130.3, 60.0),
3011        ];
3012        super::drop_overpainted_cells(&mut cells);
3013        assert_eq!(cells.len(), 4);
3014    }
3015}
3016
3017#[cfg(test)]
3018mod vestigial_layer {
3019    use crate::pdfium_backend::{PdfPage, TextCell};
3020
3021    fn page_with(texts: &[&str]) -> PdfPage {
3022        let cells = texts
3023            .iter()
3024            .enumerate()
3025            .map(|(i, t)| TextCell {
3026                text: t.to_string(),
3027                l: 10.0,
3028                t: 10.0 + 12.0 * i as f32,
3029                r: 90.0,
3030                b: 20.0 + 12.0 * i as f32,
3031            })
3032            .collect();
3033        PdfPage::from_cells(595.0, 842.0, 1.0, cells)
3034    }
3035
3036    /// The reported scanned form: three typed-in field values ("03", "05",
3037    /// "2025") over three image pages. That must read as *no usable layer*,
3038    /// so the browser routes the document to OCR instead of extracting
3039    /// thirteen characters and skipping the letter entirely.
3040    #[test]
3041    fn typed_in_form_fields_are_not_a_text_layer() {
3042        let pages = vec![
3043            page_with(&["03", "05", "2025"]),
3044            page_with(&[]),
3045            page_with(&[]),
3046        ];
3047        assert!(super::text_layer_is_vestigial(&pages));
3048        assert!(super::text_layer_is_vestigial(&[page_with(&[])]));
3049    }
3050
3051    /// A short but genuine digital document — one page, a few real lines —
3052    /// keeps the fast text path.
3053    #[test]
3054    fn sparse_but_real_documents_pass() {
3055        let one_pager = vec![page_with(&[
3056            "Confidential briefing",
3057            "Prepared for the board meeting",
3058            "Do not distribute",
3059        ])];
3060        assert!(!super::text_layer_is_vestigial(&one_pager));
3061    }
3062}