Skip to main content

docling_pdf/render/
prepass.rs

1//! A pre-pass over a content stream for the two constructs lopdf's content
2//! lexer (0.44) does not hand over intact:
3//!
4//! * **Inline images** (`BI … ID … EI`): lopdf drops the whole image when
5//!   the colour space is an abbreviation it does not know (`/G`, `/I`), an
6//!   `Indexed` array or a resource name, and *every* filtered one
7//!   ("filters for inline images are not yet implemented"). pdfTeX rules,
8//!   scanner-driver strips and dvips bitmaps are all of those. The pre-pass
9//!   cuts each inline image out itself — the header parsed as a dictionary,
10//!   the data by its computed length (unfiltered) or the `EI` delimiter
11//!   (filtered) — and leaves `/I<n> BIX` in its place, which the interpreter
12//!   resolves against [`Prepared::inline`].
13//! * **Type 3 glyph operators** `d0` / `d1`: the alphabetic operator lexer
14//!   splits them into `d` and a stray `0`/`1` operand that then prefixes the
15//!   next operator's operands (a `re` reads the wrong rectangle). They are
16//!   rewritten to `dZero` / `dOne`.
17//!
18//! The pass is one byte scan honouring strings, hex strings, dictionaries
19//! and comments; a stream without either construct is returned borrowed.
20
21use std::borrow::Cow;
22
23use lopdf::{Dictionary, Document, Object, Stream};
24
25use super::color::ColorSpace;
26use super::objects::{get2, get_bool2, get_int2, name, resource};
27
28pub struct Prepared<'a> {
29    pub content: Cow<'a, [u8]>,
30    /// The inline images in order of appearance; `/I<n> BIX` names index `n`.
31    pub inline: Vec<Stream>,
32}
33
34fn is_white(c: u8) -> bool {
35    matches!(c, b'\0' | b'\t' | b'\n' | b'\x0c' | b'\r' | b' ')
36}
37
38fn is_delim(c: u8) -> bool {
39    matches!(
40        c,
41        b'(' | b')' | b'<' | b'>' | b'[' | b']' | b'{' | b'}' | b'/' | b'%'
42    )
43}
44
45fn is_regular(c: u8) -> bool {
46    !is_white(c) && !is_delim(c)
47}
48
49/// Skip one lexical item starting at `i` (a string, hex string, comment,
50/// delimiter or regular token); returns the index after it, and the token's
51/// bytes when it was a regular token.
52fn skip_item(content: &[u8], i: usize) -> (usize, Option<&[u8]>) {
53    let c = content[i];
54    match c {
55        b'%' => {
56            let mut j = i;
57            while j < content.len() && content[j] != b'\n' && content[j] != b'\r' {
58                j += 1;
59            }
60            (j, None)
61        }
62        b'(' => {
63            let mut depth = 0usize;
64            let mut j = i;
65            while j < content.len() {
66                match content[j] {
67                    b'\\' => j += 1,
68                    b'(' => depth += 1,
69                    b')' => {
70                        depth -= 1;
71                        if depth == 0 {
72                            return (j + 1, None);
73                        }
74                    }
75                    _ => {}
76                }
77                j += 1;
78            }
79            (content.len(), None)
80        }
81        b'<' => {
82            if content.get(i + 1) == Some(&b'<') {
83                (i + 2, None)
84            } else {
85                let mut j = i + 1;
86                while j < content.len() && content[j] != b'>' {
87                    j += 1;
88                }
89                ((j + 1).min(content.len()), None)
90            }
91        }
92        b'/' => {
93            let mut j = i + 1;
94            while j < content.len() && is_regular(content[j]) {
95                j += 1;
96            }
97            (j, None)
98        }
99        c if is_delim(c) => (i + 1, None),
100        _ => {
101            let mut j = i;
102            while j < content.len() && is_regular(content[j]) {
103                j += 1;
104            }
105            (j, Some(&content[i..j]))
106        }
107    }
108}
109
110/// Does the content at `i` (after the data) read as whitespace, `EI`, then
111/// a delimiter, whitespace or the end?
112fn ei_at(content: &[u8], mut i: usize) -> Option<usize> {
113    while i < content.len() && is_white(content[i]) {
114        i += 1;
115    }
116    if content.get(i..i + 2) != Some(b"EI") {
117        return None;
118    }
119    let end = i + 2;
120    if end == content.len() || is_white(content[end]) || is_delim(content[end]) {
121        Some(end)
122    } else {
123        None
124    }
125}
126
127/// The first `EI` after `start` that stands alone between whitespace (or a
128/// delimiter) and is followed by plausible content-stream text, as pdfium's
129/// and lopdf's own heuristics have it.
130fn find_ei(content: &[u8], start: usize) -> Option<(usize, usize)> {
131    let mut i = start;
132    while i + 1 < content.len() {
133        if content[i] == b'E'
134            && content[i + 1] == b'I'
135            && (i == start || is_white(content[i - 1]))
136            && (i + 2 == content.len() || is_white(content[i + 2]) || is_delim(content[i + 2]))
137        {
138            let tail = &content[i + 2..(i + 34).min(content.len())];
139            if tail
140                .iter()
141                .all(|&c| is_white(c) || (0x20..0x7f).contains(&c))
142            {
143                let mut data_end = i;
144                while data_end > start && is_white(content[data_end - 1]) {
145                    data_end -= 1;
146                }
147                return Some((data_end, i + 2));
148            }
149        }
150        i += 1;
151    }
152    None
153}
154
155/// The header of an inline image as a dictionary: lopdf's operand parser
156/// on `<< … >> X`.
157fn parse_header(header: &[u8]) -> Option<Dictionary> {
158    let mut src = Vec::with_capacity(header.len() + 8);
159    src.extend_from_slice(b"<<");
160    src.extend_from_slice(header);
161    src.extend_from_slice(b">> X");
162    let ops = lopdf::content::Content::decode(&src).ok()?;
163    let op = ops.operations.into_iter().next()?;
164    match op.operands.into_iter().next()? {
165        Object::Dictionary(d) => Some(d),
166        _ => None,
167    }
168}
169
170/// Components per sample of an inline image's colour space (1 for masks and
171/// Indexed), `None` when it cannot be told without decoding.
172fn components(doc: &Document, d: &Dictionary, res: Option<&Dictionary>) -> Option<usize> {
173    if get_bool2(doc, d, b"ImageMask", b"IM") == Some(true) {
174        return Some(1);
175    }
176    let cs = get2(doc, d, b"ColorSpace", b"CS")?;
177    let cs = match cs {
178        Object::Name(n) => match n.as_slice() {
179            b"G" | b"DeviceGray" | b"CalGray" | b"I" | b"Indexed" => return Some(1),
180            b"RGB" | b"DeviceRGB" | b"CalRGB" => return Some(3),
181            b"CMYK" | b"DeviceCMYK" => return Some(4),
182            other => resource(doc, res, b"ColorSpace", other)?,
183        },
184        other => other,
185    };
186    if let Object::Array(a) = cs {
187        if let Some(Object::Name(n)) = a.first() {
188            if n == b"I" || n == b"Indexed" {
189                return Some(1);
190            }
191        }
192    }
193    ColorSpace::parse(doc, cs, res).map(|c| c.components())
194}
195
196/// Rewrite `content` as the module docs describe.
197pub fn prepare<'a>(content: &'a [u8], doc: &Document, res: Option<&Dictionary>) -> Prepared<'a> {
198    let mut out: Option<Vec<u8>> = None;
199    let mut inline = Vec::new();
200    let mut copied = 0usize; // content[..copied] is already in `out`
201    let mut i = 0usize;
202    while i < content.len() {
203        if is_white(content[i]) {
204            i += 1;
205            continue;
206        }
207        let (next, tok) = skip_item(content, i);
208        let Some(tok) = tok else {
209            i = next;
210            continue;
211        };
212        match tok {
213            b"d0" | b"d1" => {
214                let o = out.get_or_insert_with(|| Vec::with_capacity(content.len() + 64));
215                o.extend_from_slice(&content[copied..i]);
216                o.extend_from_slice(if tok == b"d0" { b"dZero" } else { b"dOne" });
217                copied = next;
218                i = next;
219            }
220            b"BI" => {
221                // The header runs to the `ID` token; the data starts one
222                // byte after it.
223                let header_start = next;
224                let mut j = next;
225                let mut id_end = None;
226                while j < content.len() {
227                    if is_white(content[j]) {
228                        j += 1;
229                        continue;
230                    }
231                    let (n2, t2) = skip_item(content, j);
232                    if t2 == Some(b"ID".as_slice()) {
233                        id_end = Some((j, n2));
234                        break;
235                    }
236                    j = n2;
237                }
238                let Some((id_start, id_end)) = id_end else {
239                    i = next;
240                    continue;
241                };
242                let header = &content[header_start..id_start];
243                let dict = parse_header(header);
244                let data_start = (id_end + 1).min(content.len());
245                // Unfiltered data has a computed length; otherwise (or when
246                // the computed end is not followed by `EI`) the delimiter
247                // search decides.
248                let mut span = None;
249                if let Some(d) = &dict {
250                    let filtered = get2(doc, d, b"Filter", b"F").is_some();
251                    if let Some(len) = get_int2(doc, d, b"Length", b"L") {
252                        let end = data_start.saturating_add(len.max(0) as usize);
253                        if end <= content.len() {
254                            if let Some(ei) = ei_at(content, end) {
255                                span = Some((end, ei));
256                            }
257                        }
258                    }
259                    if span.is_none() && !filtered {
260                        let w = get_int2(doc, d, b"Width", b"W").unwrap_or(0).max(0) as usize;
261                        let h = get_int2(doc, d, b"Height", b"H").unwrap_or(0).max(0) as usize;
262                        let bpc = if get_bool2(doc, d, b"ImageMask", b"IM") == Some(true) {
263                            1
264                        } else {
265                            get_int2(doc, d, b"BitsPerComponent", b"BPC")
266                                .unwrap_or(8)
267                                .max(1) as usize
268                        };
269                        if let Some(nc) = components(doc, d, res) {
270                            let len = (w * nc * bpc).div_ceil(8) * h;
271                            let end = data_start.saturating_add(len);
272                            if end <= content.len() {
273                                if let Some(ei) = ei_at(content, end) {
274                                    span = Some((end, ei));
275                                }
276                            }
277                        }
278                    }
279                }
280                if span.is_none() {
281                    span = find_ei(content, data_start);
282                }
283                let Some((data_end, ei_end)) = span else {
284                    // No terminator: the rest of the stream is image data.
285                    i = content.len();
286                    continue;
287                };
288                let o = out.get_or_insert_with(|| Vec::with_capacity(content.len() + 64));
289                o.extend_from_slice(&content[copied..i]);
290                if let Some(d) = dict {
291                    o.extend_from_slice(format!(" /I{} BIX ", inline.len()).as_bytes());
292                    inline.push(Stream::new(d, content[data_start..data_end].to_vec()));
293                } else {
294                    o.push(b' ');
295                }
296                copied = ei_end;
297                i = ei_end;
298            }
299            _ => i = next,
300        }
301    }
302    let content = match out {
303        Some(mut o) => {
304            o.extend_from_slice(&content[copied..]);
305            Cow::Owned(o)
306        }
307        None => Cow::Borrowed(content),
308    };
309    Prepared { content, inline }
310}
311
312/// The inline image `/I<n> BIX` names.
313pub fn inline_index(operand: &Object) -> Option<usize> {
314    let n = name(operand)?;
315    n.strip_prefix(b"I")
316        .and_then(|s| std::str::from_utf8(s).ok())
317        .and_then(|s| s.parse().ok())
318}
319
320#[cfg(test)]
321mod tests {
322    use super::*;
323
324    fn doc() -> Document {
325        Document::with_version("1.5")
326    }
327
328    #[test]
329    fn passes_plain_content_through_borrowed() {
330        let c = b"q 1 0 0 1 0 0 cm (BI ID EI d1) Tj Q";
331        let p = prepare(c, &doc(), None);
332        assert!(matches!(p.content, Cow::Borrowed(_)));
333        assert!(p.inline.is_empty());
334    }
335
336    #[test]
337    fn rewrites_type3_operators() {
338        let p = prepare(b"1000 0 0 0 750 750 d1 0 0 750 750 re f", &doc(), None);
339        assert_eq!(
340            &*p.content,
341            &b"1000 0 0 0 750 750 dOne 0 0 750 750 re f"[..]
342        );
343        let p = prepare(b"1000 0 d0\n", &doc(), None);
344        assert_eq!(&*p.content, &b"1000 0 dZero\n"[..]);
345        // `d0`/`d1` inside a string or as part of a longer token are left.
346        let p = prepare(b"(d1) Tj /d1 gs xd1 cm", &doc(), None);
347        assert!(matches!(p.content, Cow::Borrowed(_)));
348    }
349
350    #[test]
351    fn cuts_inline_images_by_length() {
352        // 2 × 1 gray, 8 bpc: two data bytes — one of them `E`, the other `I`.
353        let mut c = b"q BI /W 2 /H 1 /CS /G /BPC 8 ID ".to_vec();
354        c.extend_from_slice(b"EI");
355        c.extend_from_slice(b" EI Q");
356        let p = prepare(&c, &doc(), None);
357        assert_eq!(&*p.content, &b"q  /I0 BIX  Q"[..]);
358        assert_eq!(p.inline.len(), 1);
359        assert_eq!(p.inline[0].content, b"EI");
360        assert_eq!(p.inline[0].dict.get(b"W").unwrap().as_i64().unwrap(), 2);
361        let ops = lopdf::content::Content::decode(&p.content).unwrap();
362        let names: Vec<_> = ops.operations.iter().map(|o| o.operator.as_str()).collect();
363        assert_eq!(names, ["q", "BIX", "Q"]);
364        assert_eq!(inline_index(&ops.operations[1].operands[0]), Some(0));
365    }
366
367    #[test]
368    fn cuts_filtered_inline_images_at_the_delimiter() {
369        let c = b"BI /W 4 /H 4 /CS /RGB /BPC 8 /F /AHx ID\n00ff00 ff0000 >\nEI\n0 g 0 0 1 1 re f";
370        let p = prepare(c, &doc(), None);
371        assert_eq!(p.inline.len(), 1);
372        assert_eq!(p.inline[0].content, b"00ff00 ff0000 >");
373        let ops = lopdf::content::Content::decode(&p.content).unwrap();
374        let names: Vec<_> = ops.operations.iter().map(|o| o.operator.as_str()).collect();
375        assert_eq!(names, ["BIX", "g", "re", "f"]);
376    }
377
378    #[test]
379    fn indexed_and_mask_headers_count_one_component() {
380        let d = doc();
381        let mut c = b"BI /W 8 /H 1 /IM true /D [1 0] ID ".to_vec();
382        c.push(0xAA);
383        c.extend_from_slice(b" EI");
384        let p = prepare(&c, &d, None);
385        assert_eq!(p.inline.len(), 1);
386        assert_eq!(p.inline[0].content, [0xAA]);
387        let mut c = b"BI /W 2 /H 1 /BPC 8 /CS [/I /RGB 1 <000000ffffff>] ID ".to_vec();
388        c.extend_from_slice(&[0, 1]);
389        c.extend_from_slice(b" EI ");
390        let p = prepare(&c, &d, None);
391        assert_eq!(p.inline.len(), 1);
392        assert_eq!(p.inline[0].content, [0, 1]);
393    }
394}