Skip to main content

docling_pdf/
enrich.rs

1//! Optional enrichment models (docling's `do_picture_classification` /
2//! `do_code_enrichment` / `do_formula_enrichment`, issue #76).
3//!
4//! * [`PictureClassifier`] — docling-project/DocumentFigureClassifier-v2.5, an
5//!   EfficientNet image classifier over 26 figure classes (bar_chart, logo,
6//!   signature, …). The HF repo ships the ONNX graph as-is; docling's ViT
7//!   preprocessing is 224×224 bilinear + rescale + normalize, and the raw
8//!   logits are softmaxed and sorted descending — the full distribution lands
9//!   on the picture item like docling's `PictureClassificationData`.
10//!
11//! * [`CodeFormula`] — docling-project/CodeFormulaV2, an Idefics3/SmolVLM-class
12//!   VLM that rewrites a code crop as clean source text prefixed with
13//!   `<_language_>`, or a formula crop as LaTeX. Exported to three graphs by
14//!   `scripts/install/export_code_formula.py` (vision tower+connector, token
15//!   embeddings, and a KV-cached Llama decoder step verified argmax-identical
16//!   to `transformers.generate`); this module ports the Idefics3 preprocessing
17//!   (longest-edge 2048 resize → multiple-of-512 resize → 512×512 tiling + a
18//!   squashed global tile), the tiled `<image>` prompt, and the greedy decode.
19
20use image::RgbImage;
21use ort::session::Session;
22use ort::value::Tensor;
23use tokenizers::Tokenizer;
24
25use docling_core::PictureClass;
26
27/// docling crops enrichment inputs at `images_scale` pixels per point:
28/// 2.0 for the picture classifier, 1.67 (≈120 dpi) for CodeFormula.
29pub const CLASSIFIER_SCALE: f32 = 2.0;
30pub const CODE_FORMULA_SCALE: f32 = 1.67;
31/// CodeFormula expands the region box by 18% of its size on every side
32/// (docling's `expansion_factor`) before cropping.
33pub const CODE_FORMULA_EXPANSION: f32 = 0.18;
34
35// ---------------------------------------------------------------------------
36// Picture classifier
37// ---------------------------------------------------------------------------
38
39/// The 26 classes of DocumentFigureClassifier-v2.5, indexed by model class id
40/// (`config.json` `id2label`).
41const PICTURE_CLASSES: [&str; 26] = [
42    "logo",
43    "photograph",
44    "icon",
45    "engineering_drawing",
46    "line_chart",
47    "bar_chart",
48    "other",
49    "table",
50    "flow_chart",
51    "screenshot_from_computer",
52    "signature",
53    "screenshot_from_manual",
54    "geographical_map",
55    "pie_chart",
56    "page_thumbnail",
57    "stamp",
58    "music",
59    "calendar",
60    "qr_code",
61    "bar_code",
62    "full_page_image",
63    "scatter_plot",
64    "chemistry_structure",
65    "topographical_map",
66    "crossword_puzzle",
67    "box_plot",
68];
69
70const CLASSIFIER_SIDE: u32 = 224;
71/// ViT preprocessing constants from the model's `preprocessor_config.json`.
72const CLASSIFIER_MEAN: [f32; 3] = [0.485, 0.456, 0.406];
73const CLASSIFIER_STD: [f32; 3] = [0.478_539_44, 0.473_286_4, 0.474_341_63];
74
75pub struct PictureClassifier {
76    session: Session,
77}
78
79impl PictureClassifier {
80    /// Load from `DOCLING_PICTURE_CLASSIFIER_ONNX` /
81    /// `.models/picture_classifier(_int8).onnx`. `None` when the graph is
82    /// absent — the caller warns once and skips classification.
83    pub fn load_with(intra: usize) -> Option<Self> {
84        let path = crate::model_path(
85            "DOCLING_PICTURE_CLASSIFIER_ONNX",
86            ".models/picture_classifier.onnx",
87            ".models/picture_classifier_int8.onnx",
88        );
89        if !std::path::Path::new(&path).exists() {
90            eprintln!(
91                "docling-pdf: picture classifier model not found ({path}); \
92                 picture classification skipped. Run scripts/install/download_dependencies.sh."
93            );
94            return None;
95        }
96        let builder = docling_onnx::session_builder()
97            .map_err(|e| eprintln!("docling-pdf: picture classifier: {e}"))
98            .ok()?
99            .with_intra_threads(intra)
100            .ok()?;
101        let builder = docling_onnx::apply(builder)
102            .map_err(|e| eprintln!("docling-pdf: picture classifier: {e}"))
103            .ok()?;
104        let session = docling_onnx::commit_uncached(builder, &path)
105            .map_err(|e| eprintln!("docling-pdf: picture classifier load {path}: {e}"))
106            .ok()?;
107        Some(Self { session })
108    }
109
110    /// Classify one picture crop: the full 26-class distribution, descending
111    /// confidence (docling attaches all predicted classes, not just the top).
112    pub fn classify(&mut self, crop: &RgbImage) -> Result<Vec<PictureClass>, String> {
113        let resized = image::imageops::resize(
114            crop,
115            CLASSIFIER_SIDE,
116            CLASSIFIER_SIDE,
117            image::imageops::FilterType::Triangle,
118        );
119        let n = (CLASSIFIER_SIDE * CLASSIFIER_SIDE) as usize;
120        let mut data = vec![0f32; 3 * n];
121        for (i, px) in resized.pixels().enumerate() {
122            for c in 0..3 {
123                data[c * n + i] = (px[c] as f32 / 255.0 - CLASSIFIER_MEAN[c]) / CLASSIFIER_STD[c];
124            }
125        }
126        let input = Tensor::from_array((
127            [
128                1usize,
129                3,
130                CLASSIFIER_SIDE as usize,
131                CLASSIFIER_SIDE as usize,
132            ],
133            data,
134        ))
135        .map_err(|e| format!("picture classifier: input: {e}"))?;
136        let outputs = self
137            .session
138            .run(ort::inputs!["input" => input])
139            .map_err(|e| format!("picture classifier: inference: {e}"))?;
140        let (_, logits) = outputs[0]
141            .try_extract_tensor::<f32>()
142            .map_err(|e| format!("picture classifier: output: {e}"))?;
143        // softmax → (class, prob) sorted descending, like docling's engine.
144        let max = logits.iter().copied().fold(f32::NEG_INFINITY, f32::max);
145        let exp: Vec<f32> = logits.iter().map(|&v| (v - max).exp()).collect();
146        let sum: f32 = exp.iter().sum();
147        let mut preds: Vec<PictureClass> = exp
148            .iter()
149            .enumerate()
150            .map(|(i, &e)| PictureClass {
151                class_name: PICTURE_CLASSES.get(i).copied().unwrap_or("other").into(),
152                confidence: e / sum,
153            })
154            .collect();
155        preds.sort_by(|a, b| b.confidence.total_cmp(&a.confidence));
156        Ok(preds)
157    }
158}
159
160// ---------------------------------------------------------------------------
161// CodeFormula (Idefics3 VLM)
162// ---------------------------------------------------------------------------
163
164/// Which prompt the model gets for a region.
165#[derive(Debug, Clone, Copy, PartialEq, Eq)]
166pub enum CodeFormulaKind {
167    Code,
168    Formula,
169}
170
171/// Idefics3 processor constants (CodeFormulaV2's `preprocessor_config.json` /
172/// tokenizer). The token ids are fixed by the checkpoint's tokenizer.json.
173const TILE: u32 = 512; // max_image_size.longest_edge
174const LONGEST_EDGE: u32 = 2048; // size.longest_edge
175const MAX_IMAGE_SIZE: u32 = 4096; // transformers' hard upper bound
176const IMAGE_SEQ_LEN: usize = 64; // visual tokens per tile
177const IMAGE_TOKEN_ID: i64 = 100270; // <image>
178const EOS_ID: i64 = 100338; // <end_of_utterance>
179const MODEL_MAX_LEN: usize = 8192;
180const HIDDEN: usize = 576;
181const N_LAYERS: usize = 30;
182const N_KV: usize = 3;
183const HEAD_DIM: usize = 64;
184
185pub struct CodeFormula {
186    vision: Session,
187    embed: Session,
188    decoder: Session,
189    tokenizer: Tokenizer,
190}
191
192impl CodeFormula {
193    /// Load the three graphs + tokenizer from `DOCLING_CODE_FORMULA_DIR`
194    /// (default `.models/code_formula/`). `None` when absent, with a one-time
195    /// warning from the caller's slot.
196    pub fn load_with(intra: usize) -> Option<Self> {
197        let dir = docling_core::env::nonempty("DOCLING_CODE_FORMULA_DIR")
198            .unwrap_or_else(|| crate::resolve_asset(".models/code_formula"));
199        let file = |name: &str| format!("{dir}/{name}");
200        // INT8 variants take priority when present, like the other models.
201        let graph = |base: &str| {
202            let int8 = file(&format!("{base}_int8.onnx"));
203            if !crate::prefer_fp32() && std::path::Path::new(&int8).exists() {
204                int8
205            } else {
206                file(&format!("{base}.onnx"))
207            }
208        };
209        for f in [&graph("vision"), &graph("embed"), &graph("decoder_kv")] {
210            if !std::path::Path::new(f.as_str()).exists() {
211                eprintln!(
212                    "docling-pdf: CodeFormula model not found ({f}); code/formula \
213                     enrichment skipped. Run scripts/install/download_dependencies.sh."
214                );
215                return None;
216            }
217        }
218        let load = |p: String| {
219            let builder = docling_onnx::session_builder()
220                .map_err(|e| eprintln!("docling-pdf: CodeFormula: {e}"))
221                .ok()?
222                .with_intra_threads(intra)
223                .ok()?;
224            let builder = docling_onnx::apply(builder)
225                .map_err(|e| eprintln!("docling-pdf: CodeFormula: {e}"))
226                .ok()?;
227            docling_onnx::commit_uncached(builder, &p)
228                .map_err(|e| eprintln!("docling-pdf: CodeFormula load {p}: {e}"))
229                .ok()
230        };
231        let tokenizer = Tokenizer::from_file(file("tokenizer.json"))
232            .map_err(|e| eprintln!("docling-pdf: CodeFormula tokenizer: {e}"))
233            .ok()?;
234        Some(Self {
235            vision: load(graph("vision"))?,
236            embed: load(graph("embed"))?,
237            decoder: load(graph("decoder_kv"))?,
238            tokenizer,
239        })
240    }
241
242    /// Run the VLM on a code/formula crop and return the post-processed text
243    /// (still carrying the `<_language_>` prefix for code — see
244    /// [`extract_code_language`]).
245    pub fn predict(&mut self, crop: &RgbImage, kind: CodeFormulaKind) -> Result<String, String> {
246        // Debug aid: dump each crop the VLM sees (compare against docling's
247        // `prepare_element` output when chasing a generation divergence).
248        if let Some(dir) = docling_core::env::nonempty("DOCLING_RS_ENRICH_DEBUG") {
249            use std::sync::atomic::{AtomicUsize, Ordering};
250            static N: AtomicUsize = AtomicUsize::new(0);
251            let n = N.fetch_add(1, Ordering::Relaxed);
252            let _ = crop.save(format!("{dir}/rs_crop_{n}.png"));
253        }
254        let (tiles, rows, cols) = preprocess_idefics3(crop);
255        let n_tiles = tiles.len() / (3 * (TILE * TILE) as usize);
256
257        // Vision tower over the tile batch → [T, 64, 576]. (Scoped: the
258        // session outputs hold a mutable borrow of the session until dropped.)
259        let feats: Vec<f32> = {
260            let input = Tensor::from_array(([n_tiles, 3, TILE as usize, TILE as usize], tiles))
261                .map_err(|e| format!("code-formula: vision input: {e}"))?;
262            let outputs = self
263                .vision
264                .run(ort::inputs!["pixel_values" => input])
265                .map_err(|e| format!("code-formula: vision: {e}"))?;
266            let (_, feats) = outputs["image_features"]
267                .try_extract_tensor::<f32>()
268                .map_err(|e| format!("code-formula: vision output: {e}"))?;
269            feats.to_vec()
270        };
271
272        // Prompt: the chat template with the single <image> expanded into the
273        // per-tile token grid (transformers' Idefics3Processor layout).
274        let query = match kind {
275            CodeFormulaKind::Code => "<code>",
276            CodeFormulaKind::Formula => "<formula>",
277        };
278        let prompt = format!(
279            "<|start_of_role|>user:{}{query}<end_of_utterance>\nassistant:",
280            image_prompt(rows, cols)
281        );
282        let enc = self
283            .tokenizer
284            .encode(prompt, false)
285            .map_err(|e| format!("code-formula: tokenize: {e}"))?;
286        let ids: Vec<i64> = enc.get_ids().iter().map(|&v| v as i64).collect();
287        let seq = ids.len();
288
289        // Token embeddings, then scatter the visual tokens into the <image>
290        // positions (Idefics3's inputs_merger).
291        let mut embeds = self.embed_ids(&ids)?;
292        let image_positions: Vec<usize> = ids
293            .iter()
294            .enumerate()
295            .filter(|(_, &t)| t == IMAGE_TOKEN_ID)
296            .map(|(i, _)| i)
297            .collect();
298        if image_positions.len() != n_tiles * IMAGE_SEQ_LEN {
299            return Err(format!(
300                "code-formula: {} image tokens for {} tiles",
301                image_positions.len(),
302                n_tiles
303            ));
304        }
305        for (v, &pos) in image_positions.iter().enumerate() {
306            embeds[pos * HIDDEN..(pos + 1) * HIDDEN]
307                .copy_from_slice(&feats[v * HIDDEN..(v + 1) * HIDDEN]);
308        }
309
310        // Greedy KV-cache decode until <end_of_utterance> (docling caps
311        // generation at the model's 8192 context). The K/V caches stay owned
312        // `ort` values fed straight back into the next step — never extracted
313        // or copied (they grow every step, so per-step copies would be
314        // O(steps²) float traffic; same pattern as the TableFormer decoder).
315        // ort's array constructors reject a 0-length dim, so the zero-`past`
316        // first-step tensors go through the session allocator.
317        let mut cache: Option<(ort::value::DynValue, ort::value::DynValue)> = None;
318        let empty = {
319            let mk = || {
320                Tensor::<f32>::new(
321                    self.decoder.allocator(),
322                    [N_LAYERS, 1, N_KV, 0usize, HEAD_DIM],
323                )
324                .map_err(|e| format!("code-formula: empty kv cache: {e}"))
325            };
326            (mk()?, mk()?)
327        };
328        let mut past_len = 0usize;
329        let mut positions: Vec<i64> = (0..seq as i64).collect();
330        let mut x = embeds;
331        let mut x_seq = seq;
332        let mut out_ids: Vec<u32> = Vec::new();
333        let max_new = MODEL_MAX_LEN.saturating_sub(seq);
334        for _ in 0..max_new {
335            let embeds_t = Tensor::from_array(([1usize, x_seq, HIDDEN], x))
336                .map_err(|e| format!("code-formula: embeds: {e}"))?;
337            let pos_t = Tensor::from_array(([1usize, positions.len()], positions.clone()))
338                .map_err(|e| format!("code-formula: positions: {e}"))?;
339            let next = {
340                let mut out = match cache.as_ref() {
341                    Some((k, v)) => self.decoder.run(ort::inputs![
342                        "inputs_embeds" => embeds_t, "position_ids" => pos_t,
343                        "past_k" => k, "past_v" => v]),
344                    None => self.decoder.run(ort::inputs![
345                        "inputs_embeds" => embeds_t, "position_ids" => pos_t,
346                        "past_k" => &empty.0, "past_v" => &empty.1]),
347                }
348                .map_err(|e| format!("code-formula: decoder: {e}"))?;
349                let (_, logits) = out["logits"]
350                    .try_extract_tensor::<f32>()
351                    .map_err(|e| format!("code-formula: logits: {e}"))?;
352                let next = logits
353                    .iter()
354                    .enumerate()
355                    .max_by(|a, b| a.1.total_cmp(b.1))
356                    .map(|(i, _)| i as i64)
357                    .unwrap_or(EOS_ID);
358                cache = Some((
359                    out.remove("new_k")
360                        .ok_or_else(|| "code-formula: new_k missing".to_string())?,
361                    out.remove("new_v")
362                        .ok_or_else(|| "code-formula: new_v missing".to_string())?,
363                ));
364                next
365            };
366            past_len += x_seq;
367            if next == EOS_ID {
368                break;
369            }
370            out_ids.push(next as u32);
371            x = self.embed_ids(&[next])?;
372            x_seq = 1;
373            positions = vec![past_len as i64];
374        }
375
376        let text = self
377            .tokenizer
378            .decode(&out_ids, false)
379            .map_err(|e| format!("code-formula: decode: {e}"))?;
380        Ok(post_process(&text))
381    }
382
383    fn embed_ids(&mut self, ids: &[i64]) -> Result<Vec<f32>, String> {
384        let input = Tensor::from_array(([1usize, ids.len()], ids.to_vec()))
385            .map_err(|e| format!("code-formula: ids: {e}"))?;
386        let out = self
387            .embed
388            .run(ort::inputs!["input_ids" => input])
389            .map_err(|e| format!("code-formula: embed: {e}"))?;
390        let (_, embeds) = out["inputs_embeds"]
391            .try_extract_tensor::<f32>()
392            .map_err(|e| format!("code-formula: embed output: {e}"))?;
393        Ok(embeds.to_vec())
394    }
395}
396
397/// The `<image>` expansion for a rows×cols tile grid + global tile
398/// (transformers' `Idefics3Processor.replace_image_token`).
399fn image_prompt(rows: u32, cols: u32) -> String {
400    let img = "<image>".repeat(IMAGE_SEQ_LEN);
401    let mut s = String::new();
402    for r in 1..=rows {
403        for c in 1..=cols {
404            s.push_str(&format!("<fake_token_around_image><row_{r}_col_{c}>{img}"));
405        }
406        s.push('\n');
407    }
408    s.push_str(&format!(
409        "\n<fake_token_around_image><global-img>{img}<fake_token_around_image>"
410    ));
411    s
412}
413
414/// Idefics3 image preprocessing: longest-edge 2048 resize (LANCZOS), resize to
415/// multiples of 512, split into 512×512 tiles (row-major) plus the whole image
416/// squashed to 512×512 as the trailing global tile, then `(x/255 - 0.5)/0.5`.
417/// Returns the flattened `[T,3,512,512]` tensor data and the grid shape.
418fn preprocess_idefics3(crop: &RgbImage) -> (Vec<f32>, u32, u32) {
419    use image::imageops::FilterType;
420    // 1. longest edge → 2048 exactly (up- or down-scale), short side rounded
421    //    to even, both clamped below 4096 (rescale_to_max_len).
422    let (w0, h0) = crop.dimensions();
423    let (mut w, mut h) = rescale_to_max_len(w0, h0, LONGEST_EDGE);
424    (h, w) = scale_below_upper_bound(h, w, MAX_IMAGE_SIZE);
425    let img = image::imageops::resize(crop, w, h, FilterType::Lanczos3);
426
427    // 2. ceil each side to a multiple of the 512 tile (resize_for_vision_encoder).
428    let (tw, th) = if w >= h {
429        let tw = w.div_ceil(TILE) * TILE;
430        let th0 = (tw as f64 / (w as f64 / h as f64)) as u32;
431        (tw, th0.div_ceil(TILE) * TILE)
432    } else {
433        let th = h.div_ceil(TILE) * TILE;
434        let tw0 = (th as f64 * (w as f64 / h as f64)) as u32;
435        (tw0.div_ceil(TILE) * TILE, th)
436    };
437    let img = image::imageops::resize(&img, tw, th, FilterType::Lanczos3);
438
439    // 3. tiles (row-major) + the global squash.
440    let (rows, cols) = (th / TILE, tw / TILE);
441    let mut tensor = Vec::with_capacity(((rows * cols + 1) * 3 * TILE * TILE) as usize);
442    for r in 0..rows {
443        for c in 0..cols {
444            let tile = image::imageops::crop_imm(&img, c * TILE, r * TILE, TILE, TILE).to_image();
445            push_normalized(&mut tensor, &tile);
446        }
447    }
448    let global = image::imageops::resize(&img, TILE, TILE, FilterType::Lanczos3);
449    push_normalized(&mut tensor, &global);
450    (tensor, rows, cols)
451}
452
453/// transformers' `_resize_output_size_rescale_to_max_len`: longest edge to
454/// `max_len` exactly, the short side `int(long/aspect)` bumped to even.
455fn rescale_to_max_len(w0: u32, h0: u32, max_len: u32) -> (u32, u32) {
456    let aspect = w0 as f64 / h0 as f64;
457    let (w, h) = if w0 >= h0 {
458        let w = max_len;
459        let mut h = (w as f64 / aspect) as u32;
460        if !h.is_multiple_of(2) {
461            h += 1;
462        }
463        (w, h)
464    } else {
465        let h = max_len;
466        let mut w = (h as f64 * aspect) as u32;
467        if !w.is_multiple_of(2) {
468            w += 1;
469        }
470        (w, h)
471    };
472    (w.max(1), h.max(1))
473}
474
475/// transformers' `_resize_output_size_scale_below_upper_bound` (a no-op unless
476/// the even-bump pushed a side past the hard 4096 cap).
477fn scale_below_upper_bound(h0: u32, w0: u32, max_len: u32) -> (u32, u32) {
478    let aspect = w0 as f64 / h0 as f64;
479    let (h, w) = if w0 >= h0 && w0 > max_len {
480        let w = max_len;
481        (((w as f64 / aspect) as u32).max(1), w)
482    } else if h0 > w0 && h0 > max_len {
483        let h = max_len;
484        (h, ((h as f64 * aspect) as u32).max(1))
485    } else {
486        (h0, w0)
487    };
488    (h.max(1), w.max(1))
489}
490
491/// Append one 512×512 tile as CHW `(x/255 - 0.5)/0.5`.
492fn push_normalized(tensor: &mut Vec<f32>, tile: &RgbImage) {
493    let n = (TILE * TILE) as usize;
494    let base = tensor.len();
495    tensor.resize(base + 3 * n, 0.0);
496    for (i, px) in tile.pixels().enumerate() {
497        for c in 0..3 {
498            tensor[base + c * n + i] = px[c] as f32 / 255.0 * 2.0 - 1.0;
499        }
500    }
501}
502
503/// docling's CodeFormulaModel post-processing: truncate at
504/// `<end_of_utterance>`, remove the closing/query artifacts, strip leading
505/// whitespace.
506fn post_process(text: &str) -> String {
507    let mut t = match text.find("<end_of_utterance>") {
508        Some(i) => &text[..i],
509        None => text,
510    }
511    .to_string();
512    for tok in ["</code>", "</formula>", "<loc_0><loc_0><loc_500><loc_500>"] {
513        t = t.replace(tok, "");
514    }
515    t.trim_start().to_string()
516}
517
518/// docling's `_extract_code_language`: an output beginning with
519/// `<_language_>` yields `(remainder, Some(language))`.
520pub fn extract_code_language(s: &str) -> (String, Option<String>) {
521    let rest = match s.strip_prefix("<_") {
522        Some(r) => r,
523        None => return (s.to_string(), None),
524    };
525    // The language is everything up to the closing `_>` that contains neither
526    // `_` nor `>` (docling's `[^_>]+`).
527    match rest.find("_>") {
528        Some(end) if !rest[..end].is_empty() && !rest[..end].contains(['_', '>']) => {
529            let lang = rest[..end].to_string();
530            let remainder = rest[end + 2..].trim_start().to_string();
531            (remainder, Some(lang))
532        }
533        _ => (s.to_string(), None),
534    }
535}
536
537#[cfg(test)]
538mod tests {
539    use super::*;
540
541    #[test]
542    fn language_prefix_extraction() {
543        assert_eq!(
544            extract_code_language("<_JavaScript_> function f() {}"),
545            (
546                "function f() {}".to_string(),
547                Some("JavaScript".to_string())
548            )
549        );
550        assert_eq!(
551            extract_code_language("plain text"),
552            ("plain text".to_string(), None)
553        );
554        assert_eq!(
555            extract_code_language("<_x_y_> t"),
556            ("<_x_y_> t".to_string(), None)
557        );
558    }
559
560    #[test]
561    fn idefics3_grid_matches_processor() {
562        // An 800×300 crop resizes to 2048×768, tiles to 2048×1024 → 4×2 grid
563        // (+ global) — the shape verified against transformers' processor.
564        let img = RgbImage::new(800, 300);
565        let (tensor, rows, cols) = preprocess_idefics3(&img);
566        assert_eq!((rows, cols), (2, 4));
567        assert_eq!(tensor.len(), 9 * 3 * 512 * 512);
568    }
569
570    #[test]
571    fn image_prompt_layout() {
572        let p = image_prompt(1, 2);
573        assert!(p.starts_with("<fake_token_around_image><row_1_col_1><image>"));
574        assert!(p.contains("<row_1_col_2>"));
575        let tail = "<fake_token_around_image><global-img>".to_owned()
576            + &"<image>".repeat(64)
577            + "<fake_token_around_image>";
578        assert!(p.ends_with(&tail));
579        assert_eq!(p.matches("<image>").count(), 3 * 64);
580    }
581}