Skip to main content

docling_pdf/
assemble.rs

1//! Layout-driven assembly: map detected [`Region`]s + text cells to a
2//! [`DoclingDocument`], mirroring docling's page-assembly + reading-order.
3//!
4//! Overlapping detections are resolved greedily by score, each text cell is
5//! assigned to its best-containing region, regions are ordered in reading order
6//! (two-column aware), and each becomes a typed node by its layout label.
7
8use docling_core::{CaptionParent, Node, PictureClass, PictureImage, Table};
9#[cfg(feature = "ml")]
10use image::RgbImage;
11
12use crate::layout::Region;
13use crate::pdfium_backend::{PdfPage, TextCell};
14
15fn area(l: f32, t: f32, r: f32, b: f32) -> f32 {
16    ((r - l).max(0.0)) * ((b - t).max(0.0))
17}
18
19/// Intersection area of two boxes.
20fn inter(a: &Region, l: f32, t: f32, r: f32, b: f32) -> f32 {
21    let il = a.l.max(l);
22    let it = a.t.max(t);
23    let ir = a.r.min(r);
24    let ib = a.b.min(b);
25    area(il, it, ir, ib)
26}
27
28/// Wrapper (structured-region) labels, ported from docling
29/// `LayoutPostprocessor.WRAPPER_TYPES`: a region that *contains* other regions
30/// and renders as a structured block (a table / table-of-contents index), not as
31/// its own flat text.
32fn is_wrapper(label: &str) -> bool {
33    matches!(
34        label,
35        "table" | "document_index" | "form" | "key_value_region"
36    )
37}
38
39/// Labels docling's table-structure (TableFormer) model runs on and that render
40/// as a Markdown table: a plain `table` and a `document_index` (a table of
41/// contents), which docling assembles as a `TableItem` too.
42pub fn is_table_like(label: &str) -> bool {
43    matches!(label, "table" | "document_index")
44}
45
46/// Greedily keep regions by descending score, dropping a region that is mostly
47/// covered by an already-kept one (RT-DETR emits overlapping duplicates).
48fn greedy(mut regions: Vec<Region>) -> Vec<Region> {
49    regions.sort_by(|a, b| b.score.total_cmp(&a.score));
50    let mut kept: Vec<Region> = Vec::new();
51    for r in regions {
52        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
53        let covered = kept.iter().any(|k| {
54            let i = inter(&r, k.l, k.t, k.r, k.b);
55            let ka = area(k.l, k.t, k.r, k.b).max(1.0);
56            // drop if most of r is inside k, or they strongly mutually overlap
57            i / ra > 0.7 || i / (ra + ka - i) > 0.5
58        });
59        if !covered {
60            kept.push(r);
61        }
62    }
63    kept
64}
65
66/// Resolve overlapping RT-DETR detections, ported from the bucket structure of
67/// docling's `LayoutPostprocessor`: regular, picture and wrapper clusters live in
68/// **separate** spatial indexes and are de-overlapped independently, so a
69/// high-score picture never suppresses a lower-score table or table-of-contents
70/// index (the redp5110 TOC that was otherwise replaced by a picture box). A
71/// cross-type pass first drops a picture that nearly coincides with a table
72/// (`_handle_cross_type_overlaps`), keeping the structured table.
73/// docling's `_remove_overlapping_clusters("picture")`: same-label picture
74/// detections whose boxes heavily overlap (IoU > 0.8, or either box > 80 %
75/// contained in the other) form one group, and a single survivor is kept per
76/// group. Survivor selection ports `_should_prefer_cluster` /
77/// `_select_best_cluster_from_group` with the picture params
78/// (`area_threshold` 2.0, `conf_threshold` 0.3): a candidate is rejected only
79/// when a rival is both comparable in size (candidate ≤ 2× its area) and
80/// clearly more confident (> 0.3); among the survivors the *larger* box wins
81/// unless it is > 0.3 less confident. Net effect on the corpus: a figure the
82/// detector proposes both whole and as its sub-panels (2206's four-thumbnail
83/// Figure 1) collapses to the whole-figure box, exactly like docling.
84pub(crate) fn dedup_pictures(regions: &mut Vec<Region>) {
85    let idx: Vec<usize> = (0..regions.len())
86        .filter(|&i| regions[i].label == "picture")
87        .collect();
88    if idx.len() < 2 {
89        return;
90    }
91    // Union-find over the picture subset.
92    let mut parent: Vec<usize> = (0..idx.len()).collect();
93    fn find(parent: &mut [usize], i: usize) -> usize {
94        let mut root = i;
95        while parent[root] != root {
96            root = parent[root];
97        }
98        let mut cur = i;
99        while parent[cur] != root {
100            let next = parent[cur];
101            parent[cur] = root;
102            cur = next;
103        }
104        root
105    }
106    let boxed = |r: &Region| (r.l, r.t, r.r, r.b);
107    for a in 0..idx.len() {
108        for b in (a + 1)..idx.len() {
109            let (ra, rb) = (&regions[idx[a]], &regions[idx[b]]);
110            let (al, at, ar, ab_) = boxed(ra);
111            let (bl, bt, br, bb) = boxed(rb);
112            let ix = (ar.min(br) - al.max(bl)).max(0.0);
113            let iy = (ab_.min(bb) - at.max(bt)).max(0.0);
114            let inter = ix * iy;
115            let aa = area(al, at, ar, ab_).max(f32::EPSILON);
116            let ba = area(bl, bt, br, bb).max(f32::EPSILON);
117            let iou = inter / (aa + ba - inter).max(f32::EPSILON);
118            if iou > 0.8 || inter / aa > 0.8 || inter / ba > 0.8 {
119                let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
120                if pa != pb {
121                    parent[pa] = pb;
122                }
123            }
124        }
125    }
126    // Per group, run docling's pairwise preference + larger-wins selection.
127    let mut groups: std::collections::HashMap<usize, Vec<usize>> = std::collections::HashMap::new();
128    for i in 0..idx.len() {
129        let root = find(&mut parent, i);
130        groups.entry(root).or_default().push(i);
131    }
132    let mut drop = vec![false; regions.len()];
133    for group in groups.values() {
134        if group.len() < 2 {
135            continue;
136        }
137        const AREA_THRESHOLD: f32 = 2.0;
138        const CONF_THRESHOLD: f32 = 0.3;
139        let area_of = |i: usize| {
140            let r = &regions[idx[i]];
141            area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
142        };
143        let mut best: Option<usize> = None;
144        for &cand in group {
145            let passes = group.iter().all(|&other| {
146                if other == cand {
147                    return true;
148                }
149                let area_ratio = area_of(cand) / area_of(other);
150                let conf_diff = regions[idx[other]].score - regions[idx[cand]].score;
151                !(area_ratio <= AREA_THRESHOLD && conf_diff > CONF_THRESHOLD)
152            });
153            if passes {
154                best = Some(match best {
155                    None => cand,
156                    Some(cur) => {
157                        if area_of(cand) > area_of(cur)
158                            && regions[idx[cur]].score - regions[idx[cand]].score <= CONF_THRESHOLD
159                        {
160                            cand
161                        } else {
162                            cur
163                        }
164                    }
165                });
166            }
167        }
168        // Every candidate rejected can't happen with docling's rule (rejection
169        // needs a strictly better rival); guard with highest score anyway.
170        let keep = best.unwrap_or_else(|| {
171            *group
172                .iter()
173                .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(&regions[idx[b]].score))
174                .expect("non-empty group")
175        });
176        for &i in group {
177            if i != keep {
178                drop[idx[i]] = true;
179            }
180        }
181    }
182    let mut keep_iter = drop.into_iter();
183    regions.retain(|_| !keep_iter.next().expect("aligned"));
184}
185
186/// `intersection_over_union` of two regions.
187fn iou(a: &Region, b: &Region) -> f32 {
188    let i = inter(a, b.l, b.t, b.r, b.b);
189    let u = area(a.l, a.t, a.r, a.b) + area(b.l, b.t, b.r, b.b) - i;
190    if u > 0.0 {
191        i / u
192    } else {
193        0.0
194    }
195}
196
197/// docling's `_resolve_coincident_pairs` (#4059, 2.122): for every (loser,
198/// winner) pair at a near-identical box (IoU > 0.8) whose confidences are
199/// within 0.1 (`loser.score - winner.score < 0.1`), the loser label is dropped
200/// so the label with the richer downstream semantic survives. Nothing else —
201/// containment, area — is considered; a clearly more confident loser stays.
202fn coincident_losers(regions: &[Region], losers: &[usize], winners: &[usize]) -> Vec<usize> {
203    let mut out = Vec::new();
204    for &li in losers {
205        for &wi in winners {
206            if iou(&regions[li], &regions[wi]) > 0.8 && regions[li].score - regions[wi].score < 0.1
207            {
208                out.push(li);
209                break;
210            }
211        }
212    }
213    out
214}
215
216/// docling's `_handle_cross_type_overlaps` (2.122/2.123 shape): the layout
217/// model can emit one grounded region under several labels, and the picture /
218/// table / container buckets are de-overlapped independently, so such a region
219/// survives twice. Elect a winner for the near-identical pairs:
220///
221/// | pair                                  | loser     | winner              |
222/// |---------------------------------------|-----------|---------------------|
223/// | TABLE vs DOCUMENT_INDEX               | table     | document_index      |
224/// | PICTURE vs TABLE / DOCUMENT_INDEX     | picture   | the table-like      |
225/// | FORM / KEY_VALUE_REGION vs *surviving* TABLE / DOCUMENT_INDEX / PICTURE | container | structured element |
226///
227/// IoU (not containment) so a genuine small figure inside a large table region
228/// is not removed; the confidence tolerance keeps a clearly more confident
229/// loser (an earlier port dropped every coincident picture regardless).
230fn handle_cross_type_overlaps(regions: Vec<Region>) -> Vec<Region> {
231    let by = |pred: &dyn Fn(&str) -> bool| -> Vec<usize> {
232        (0..regions.len())
233            .filter(|&i| pred(regions[i].label))
234            .collect()
235    };
236    let tables = by(&|l| l == "table");
237    let doc_indices = by(&|l| l == "document_index");
238    let pictures = by(&|l| l == "picture");
239    let containers = by(&|l| matches!(l, "form" | "key_value_region"));
240    let mut drop = vec![false; regions.len()];
241    for i in coincident_losers(&regions, &tables, &doc_indices) {
242        drop[i] = true;
243    }
244    let table_like: Vec<usize> = tables.iter().chain(&doc_indices).copied().collect();
245    for i in coincident_losers(&regions, &pictures, &table_like) {
246        drop[i] = true;
247    }
248    let structured: Vec<usize> = table_like
249        .iter()
250        .chain(&pictures)
251        .copied()
252        .filter(|&i| !drop[i])
253        .collect();
254    for i in coincident_losers(&regions, &containers, &structured) {
255        drop[i] = true;
256    }
257    let mut drop = drop.into_iter();
258    let mut regions = regions;
259    regions.retain(|_| !drop.next().expect("aligned"));
260    regions
261}
262
263pub fn resolve(regions: Vec<Region>) -> Vec<Region> {
264    let regions = handle_cross_type_overlaps(regions);
265    // De-overlap each bucket on its own.
266    let pictures = greedy(
267        regions
268            .iter()
269            .filter(|r| r.label == "picture")
270            .cloned()
271            .collect(),
272    );
273    // Tables and containers are separate buckets since docling 2.123
274    // (`TABLE_TYPES` vs `CONTAINER_TYPES`, docling#4064): a form drawn around a
275    // table no longer competes with it for survival — the table nests inside
276    // the container instead (`order_with_containers`).
277    let tables = greedy(
278        regions
279            .iter()
280            .filter(|r| is_table_like(r.label))
281            .cloned()
282            .collect(),
283    );
284    let containers = greedy(
285        regions
286            .iter()
287            .filter(|r| matches!(r.label, "form" | "key_value_region"))
288            .cloned()
289            .collect(),
290    );
291    let mut kept = greedy(
292        regions
293            .iter()
294            .filter(|r| r.label != "picture" && !is_wrapper(r.label))
295            .cloned()
296            .collect(),
297    );
298    dedup_nested_code(&mut kept);
299    kept.extend(pictures);
300    kept.extend(tables);
301    kept.extend(containers);
302    kept
303}
304
305/// Drop a regular region that is >80% contained in a surviving special region we
306/// render **as a single unit** — a table/table-of-contents index — ported from
307/// docling's "Remove regular clusters that are included in wrappers" step: the
308/// special absorbs it as a child (a table cell), so it must not also be emitted
309/// as its own paragraph/list-item. This stops the survey list-items from
310/// appearing both inside the detected table and again as bullets
311/// (`table_mislabeled_as_picture`).
312///
313/// `picture` regions stay in the swallow set even after #165: docling keeps a
314/// picture's contained clusters as the `PictureItem`'s *children* in the
315/// document JSON (`ReadingOrderModel._add_child_elements`), but its
316/// `MarkdownPictureSerializer` prints only the caption and the image — the
317/// children never reach the Markdown (verified against the corpus groundtruth:
318/// `amt_handbook`'s in-figure callout labels are absent). Dropping the
319/// fully-contained regulars here reproduces exactly that. What #165 *does*
320/// change is upstream, in [`add_orphan_regions`]: pictures no longer claim
321/// cells, so a line only partially under a figure box (straddling its border,
322/// ≤80 % contained) now forms an orphan region that survives this drop — those
323/// words were silently erased before, and docling emits them.
324///
325/// `form` / `key_value_region` wrappers are deliberately **excluded**: this
326/// pipeline does not render them as a structured block (they are skipped), so
327/// their textual content comes precisely from the contained regular regions —
328/// dropping those would erase the page (e.g. `right_to_left_03`'s form-heavy
329/// pages). Runs *after* [`drop_false_pictures`] so a phantom picture can't
330/// swallow real text on its way out.
331pub fn drop_contained_regulars(regions: &mut Vec<Region>) {
332    let specials: Vec<(f32, f32, f32, f32)> = regions
333        .iter()
334        .filter(|r| r.label == "picture" || is_table_like(r.label))
335        .map(|r| (r.l, r.t, r.r, r.b))
336        .collect();
337    if specials.is_empty() {
338        return;
339    }
340    regions.retain(|r| {
341        if r.label == "picture" || is_wrapper(r.label) {
342            return true;
343        }
344        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
345        !specials
346            .iter()
347            .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.8)
348    });
349}
350
351/// True for a bare, single-token source-code language label (`XML`, `C#`, `JSON`,
352/// `bash`, …) — the little header the docs render above a code block. Matched
353/// case-insensitively; anything with whitespace or longer than a token is out.
354fn is_code_language(t: &str) -> bool {
355    let t = t.trim();
356    if t.is_empty() || t.chars().any(char::is_whitespace) || t.chars().count() > 12 {
357        return false;
358    }
359    const LANGS: &[&str] = &[
360        "xml",
361        "html",
362        "xhtml",
363        "json",
364        "jsonc",
365        "yaml",
366        "yml",
367        "toml",
368        "ini",
369        "c#",
370        "csharp",
371        "f#",
372        "fsharp",
373        "vb",
374        "c",
375        "c++",
376        "cpp",
377        "java",
378        "kotlin",
379        "scala",
380        "go",
381        "golang",
382        "rust",
383        "swift",
384        "javascript",
385        "js",
386        "typescript",
387        "ts",
388        "jsx",
389        "tsx",
390        "python",
391        "py",
392        "ruby",
393        "rb",
394        "php",
395        "perl",
396        "lua",
397        "r",
398        "dart",
399        "bash",
400        "sh",
401        "shell",
402        "powershell",
403        "zsh",
404        "batch",
405        "cmd",
406        "sql",
407        "tsql",
408        "plsql",
409        "graphql",
410        "dockerfile",
411        "makefile",
412        "css",
413        "scss",
414        "sass",
415        "less",
416        "markdown",
417        "md",
418        "tex",
419        "latex",
420        "diff",
421        "proto",
422        "razor",
423        "cshtml",
424        "xaml",
425        "aspx",
426        "http",
427    ];
428    let lower = t.to_ascii_lowercase();
429    LANGS.contains(&lower.as_str())
430}
431
432/// Mark the region indices that are a code block's **language label** — a bare
433/// `XML`/`C#`/… token sitting directly above a `code` region — so they are consumed
434/// rather than emitted as their own stray paragraph/heading. The label may also be
435/// captured inside a wider code box (rendered as the fence's first line); dropping
436/// the standalone copy just removes the duplicate.
437fn code_language_labels(regions: &[Region], cells: &[TextCell]) -> Vec<bool> {
438    let mut drop = vec![false; regions.len()];
439    for (i, r) in regions.iter().enumerate() {
440        if matches!(r.label, "code" | "picture" | "table") {
441            continue;
442        }
443        if !is_code_language(&region_text(r, cells)) {
444            continue;
445        }
446        // The label sits just above the code (a blank line's gap) or is swallowed
447        // into the top of a wider code box; either way it is that block's label.
448        // The window is generous because the label's own font is small, so a
449        // one-line gap is several times its height.
450        let line_h = (r.b - r.t).abs().max(1.0);
451        let window = (line_h * 4.0).max(28.0);
452        let labels_code = regions.iter().enumerate().any(|(j, c)| {
453            if j == i || c.label != "code" {
454                return false;
455            }
456            let gap = c.t - r.b; // >0 when the code is below the label
457            let h_overlap = (r.r.min(c.r) - r.l.max(c.l)).max(0.0);
458            gap > -line_h * 3.0 && gap < window && h_overlap > 0.0
459        });
460        if labels_code {
461            drop[i] = true;
462        }
463    }
464    drop
465}
466
467/// Collapse `code` regions where one is nested inside another, keeping the larger.
468///
469/// RT-DETR sometimes emits a tight code box *and* a wider near-duplicate that also
470/// captures the block's language label (`XML`, `C#`, …). When the tight box scores
471/// higher it is kept first, and the wider container — not "mostly inside" the tight
472/// box — survives [`resolve`]'s greedy pass, so the block is emitted twice. Keeping
473/// the **larger** box (rather than dropping it) collapses the pair without leaking
474/// the container's extra cells back out as orphan text, since the larger box still
475/// covers every cell. Restricted to `code` so genuinely distinct nested regions of
476/// other kinds are untouched.
477fn dedup_nested_code(kept: &mut Vec<Region>) {
478    let mut drop = vec![false; kept.len()];
479    for i in 0..kept.len() {
480        if kept[i].label != "code" {
481            continue;
482        }
483        let ai = area(kept[i].l, kept[i].t, kept[i].r, kept[i].b).max(1.0);
484        for j in 0..kept.len() {
485            if i == j || drop[j] || kept[j].label != "code" {
486                continue;
487            }
488            let aj = area(kept[j].l, kept[j].t, kept[j].r, kept[j].b).max(1.0);
489            // Drop i when it is mostly inside a strictly larger code box j.
490            let overlap = inter(&kept[i], kept[j].l, kept[j].t, kept[j].r, kept[j].b);
491            if aj > ai && overlap / ai > 0.7 {
492                drop[i] = true;
493                break;
494            }
495        }
496    }
497    let mut keep = drop.iter();
498    kept.retain(|_| !*keep.next().unwrap());
499}
500
501/// Fraction of the page's non-empty text cells that some detected region
502/// claims (>0.2 intersection-over-self, docling's assignment rule). 1.0 for a
503/// page without text cells.
504///
505/// The int8-layout guard keys off this: a dense digital page whose detections
506/// cover almost none of its text is the signature of quantized confidences
507/// flipping under the 0.5 label thresholds on this CPU's kernels — not of a
508/// genuinely empty layout — and is worth re-running on the fp32 graph.
509pub fn layout_cell_coverage(regions: &[Region], cells: &[TextCell]) -> f32 {
510    let mut total = 0usize;
511    let mut covered = 0usize;
512    for c in cells {
513        if c.text.trim().is_empty() {
514            continue;
515        }
516        total += 1;
517        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
518        if regions
519            .iter()
520            .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
521        {
522            covered += 1;
523        }
524    }
525    if total == 0 {
526        1.0
527    } else {
528        covered as f32 / total as f32
529    }
530}
531
532/// Append `text` regions for cells the layout left uncovered ("orphan cells"),
533/// the way docling's `LayoutPostprocessor` does (`create_orphan_clusters`): any
534/// non-empty cell that no kept region covers (>50% of the cell's area) becomes a
535/// text region of its own, so text the detector missed (a stray `.`, a small
536/// label) is still emitted instead of silently dropped. Adjacent orphan cells on a
537/// line are merged so a missed paragraph doesn't shatter into one block per line.
538pub fn add_orphan_regions(regions: &mut Vec<Region>, cells: &[TextCell]) {
539    // docling assigns each cell to its single best-overlapping cluster at
540    // intersection-over-self > 0.2 and serializes exactly the assigned cells —
541    // and since [`region_texts_exclusive`] now emits under that very rule, the
542    // claim test here matches it: any cell over 0.2 will actually render in
543    // its best region, everything else becomes an orphan. Completeness by
544    // construction, with no (0.2, 0.5] hole (the old > 0.5 serializer needed
545    // the claim test raised to > 0.5 to keep right_to_left_03's `20300` from
546    // vanishing; the exclusive port closes that structurally).
547    //
548    // Only *regular* clusters claim cells: docling's `_find_unassigned_cells`
549    // walks `regular_clusters` alone, so a cell under a `picture` or a wrapper
550    // (`table`/`document_index`/`form`/`key_value_region`) that no regular
551    // cluster covers still becomes an orphan text cluster (#165). The orphans
552    // that end up *fully* inside the special are re-dropped by
553    // [`drop_contained_regulars`] (docling's Markdown drops them the same way
554    // — a picture's children never reach its `MarkdownPictureSerializer`
555    // output, a table's text renders through the reconstructed grid). The
556    // observable fix is the border-straddlers: a line only partially under a
557    // figure box used to lose its cells to the picture's 0.2 claim and vanish
558    // — now it forms an orphan region and is emitted, as docling does.
559    let assigned = |c: &TextCell| {
560        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
561        regions
562            .iter()
563            .filter(|r| r.label != "picture" && !is_wrapper(r.label))
564            .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
565    };
566    // Collect orphan cells (non-empty, unassigned), in page order.
567    let mut orphans: Vec<&TextCell> = cells
568        .iter()
569        .filter(|c| !c.text.trim().is_empty() && !assigned(c))
570        .collect();
571    if orphans.is_empty() {
572        return;
573    }
574    orphans.sort_by(|a, b| a.t.total_cmp(&b.t).then(a.l.total_cmp(&b.l)));
575    // Merge cells that sit on the same line and nearly touch into one region, so a
576    // dropped multi-word line stays one block (docling's refinement merges these).
577    let mut merged: Vec<Region> = Vec::new();
578    for c in orphans {
579        let h = (c.b - c.t).abs().max(1.0);
580        if let Some(last) = merged.last_mut() {
581            let same_line = (last.t - c.t).abs() < h * 0.5;
582            let touching = c.l <= last.r + h && c.l >= last.l - h;
583            if same_line && touching {
584                last.l = last.l.min(c.l);
585                last.r = last.r.max(c.r);
586                last.t = last.t.min(c.t);
587                last.b = last.b.max(c.b);
588                continue;
589            }
590        }
591        merged.push(Region {
592            label: "text",
593            score: 0.0,
594            l: c.l,
595            t: c.t,
596            r: c.r,
597            b: c.b,
598        });
599    }
600    regions.extend(merged);
601}
602
603/// Demote a `picture` region that is really a **text panel** — a paragraph block
604/// the layout model boxed as a figure because it is typeset on a colored
605/// background (terms-and-conditions callouts, quote boxes) — into ordinary
606/// `text` regions, one per paragraph, so its words are read instead of shipped
607/// as pixels. docling loses this text the same way (cells assigned to a picture
608/// cluster are never serialized); this is a deliberate improvement, not parity.
609///
610/// The gate is conservative so a genuine figure keeps its crop: the region must
611/// contain at least three text lines whose median width spans most of the panel
612/// (axis labels and chat bubbles are narrow and varied) and whose cells cover a
613/// substantial fraction of its area (a photo or chart with sparse labels does
614/// not). Paragraph boundaries are re-derived from the line pitch: a vertical gap
615/// clearly larger than the panel's own leading starts a new `text` region, so
616/// the panel doesn't collapse into one giant paragraph.
617///
618/// Works on any cell source — the digital text layer or OCR lines recognized
619/// from the picture crop — so the native and browser paths, with or without
620/// force-OCR, demote identically.
621pub fn recover_text_panels(regions: &mut Vec<Region>, cells: &[TextCell]) {
622    // A *captioned* picture is a genuine figure whatever it contains — the
623    // corpus is full of document screenshots ("Figure 3: …" above a page
624    // image) that are exactly as dense and wide as a text panel. Only an
625    // uncaptioned picture is a demotion candidate.
626    let captioned: Vec<bool> = regions
627        .iter()
628        .map(|r| {
629            r.label == "picture"
630                && regions.iter().any(|c| {
631                    c.label == "caption" && c.r.min(r.r) - c.l.max(r.l) > 0.0 && {
632                        let gap = if c.t >= r.b {
633                            c.t - r.b
634                        } else if r.t >= c.b {
635                            r.t - c.b
636                        } else {
637                            f32::MAX // vertically overlapping: not a caption
638                        };
639                        gap <= 25.0
640                    }
641                })
642        })
643        .collect();
644    let mut out: Vec<Region> = Vec::with_capacity(regions.len());
645    // Synthesized paragraphs and the demoted panels' boxes are kept separate
646    // from `out` until the end: the dedup filter below must not confuse a
647    // paragraph we just built with a pre-existing region inside the panel.
648    let mut demoted_paras: Vec<Region> = Vec::new();
649    let mut demoted_boxes: Vec<(f32, f32, f32, f32)> = Vec::new();
650    for (i, r) in regions.drain(..).enumerate() {
651        if r.label != "picture" || captioned[i] {
652            out.push(r);
653            continue;
654        }
655        let inside: Vec<&TextCell> = cells
656            .iter()
657            .filter(|c| {
658                !c.text.trim().is_empty() && {
659                    let ca = area(c.l, c.t, c.r, c.b).max(1.0);
660                    inter(&r, c.l, c.t, c.r, c.b) / ca > 0.5
661                }
662            })
663            .collect();
664        // Group the contained cells into lines by vertical overlap (the same
665        // rule region_text orders by), tracking each line's union box.
666        let mut lines: Vec<(f32, f32, f32, f32)> = Vec::new(); // (t, b, l, r)
667        for c in &inside {
668            let (ct, cb) = (c.t.min(c.b), c.t.max(c.b));
669            match lines.iter_mut().find(|(lt, lb, _, _)| {
670                let ov = cb.min(*lb) - ct.max(*lt);
671                ov > 0.5 * (cb - ct).min(*lb - *lt).max(1.0)
672            }) {
673                Some((lt, lb, ll, lr)) => {
674                    *lt = lt.min(ct);
675                    *lb = lb.max(cb);
676                    *ll = ll.min(c.l);
677                    *lr = lr.max(c.r);
678                }
679                None => lines.push((ct, cb, c.l, c.r)),
680            }
681        }
682        if lines.len() < 3 {
683            out.push(r);
684            continue;
685        }
686        let panel_w = (r.r - r.l).max(1.0);
687        let coverage = inside.iter().map(|c| area(c.l, c.t, c.r, c.b)).sum::<f32>()
688            / area(r.l, r.t, r.r, r.b).max(1.0);
689        let mut widths: Vec<f32> = lines.iter().map(|(_, _, l, rr)| rr - l).collect();
690        widths.sort_by(f32::total_cmp);
691        // A figure's text is ragged: a title line, small axis/tick labels, and
692        // OCR boxes over the plot area come out at wildly different heights,
693        // whereas a real text panel is set in one face with constant leading.
694        // Require near-uniform line heights (median absolute deviation ≤ 35%
695        // of the median) so an uncaptioned chart keeps its crop even when its
696        // labels are dense enough to pass the coverage gate (#173) — garbled
697        // OCR of its bars is not content.
698        let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
699        heights.sort_by(f32::total_cmp);
700        let h_med = heights[heights.len() / 2].max(1.0);
701        let mut devs: Vec<f32> = heights.iter().map(|h| (h - h_med).abs()).collect();
702        devs.sort_by(f32::total_cmp);
703        let uniform = devs[devs.len() / 2] <= 0.35 * h_med;
704        let text_panel = coverage >= 0.2 && widths[widths.len() / 2] >= 0.45 * panel_w && uniform;
705        if !text_panel {
706            out.push(r);
707            continue;
708        }
709        lines.sort_by(|a, b| a.0.total_cmp(&b.0));
710        let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
711        heights.sort_by(f32::total_cmp);
712        let h = heights[heights.len() / 2].max(1.0);
713        let mut gaps: Vec<f32> = lines
714            .windows(2)
715            .map(|w| (w[1].0 - w[0].1).max(0.0))
716            .collect();
717        gaps.sort_by(f32::total_cmp);
718        let leading = if gaps.is_empty() {
719            0.0
720        } else {
721            gaps[gaps.len() / 2]
722        };
723        let brk = (1.8 * leading).max(0.75 * h);
724        let mut para: Option<(f32, f32, f32, f32)> = None; // (l, t, r, b) union
725        for (t, b, l, rr) in &lines {
726            match &mut para {
727                Some((pl, _, pr, pb)) if *t - *pb <= brk => {
728                    *pl = pl.min(*l);
729                    *pr = pr.max(*rr);
730                    *pb = pb.max(*b);
731                }
732                _ => {
733                    if let Some((pl, pt, pr, pb)) = para.take() {
734                        demoted_paras.push(Region {
735                            label: "text",
736                            score: r.score,
737                            l: pl,
738                            t: pt,
739                            r: pr,
740                            b: pb,
741                        });
742                    }
743                    para = Some((*l, *t, *rr, *b));
744                }
745            }
746        }
747        if let Some((pl, pt, pr, pb)) = para {
748            demoted_paras.push(Region {
749                label: "text",
750                score: r.score,
751                l: pl,
752                t: pt,
753                r: pr,
754                b: pb,
755            });
756        }
757        demoted_boxes.push((r.l, r.t, r.r, r.b));
758    }
759    // The paragraphs are rebuilt from *all* of the panel's cells, so any
760    // surviving text region inside a demoted panel (an orphan cluster or a
761    // layout-detected fragment — pictures no longer swallow them, #165) would
762    // say the same words twice. Consume those; wrappers and pictures stay.
763    if !demoted_boxes.is_empty() {
764        out.retain(|r| {
765            r.label == "picture" || is_wrapper(r.label) || {
766                let ra = area(r.l, r.t, r.r, r.b).max(1.0);
767                !demoted_boxes
768                    .iter()
769                    .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.5)
770            }
771        });
772    }
773    out.extend(demoted_paras);
774    *regions = out;
775}
776
777/// Drop a `picture` detection that is a small, empty, low-confidence margin box on
778/// a **text page** — a false positive the RT-DETR layout sometimes emits (e.g.
779/// `right_to_left_02`'s phantom right-column picture, score 0.40); docling does not
780/// emit it. The gate is deliberately narrow so a genuine figure is never dropped:
781/// (1) only on pages with a digital text layer — image/scanned/figure pages have
782/// no `cells` yet at this point (OCR runs later), so their pictures, which *are*
783/// the content, are kept; (2) only a box covering < 25 % of the page (a margin
784/// artifact, not a dominant figure); (3) only when it contains no text and scores
785/// below 0.5 (real empty figures in the corpus all score ≥ 0.86).
786pub fn drop_false_pictures(
787    regions: &mut Vec<Region>,
788    cells: &[TextCell],
789    page_w: f32,
790    page_h: f32,
791) {
792    if cells.iter().all(|c| c.text.trim().is_empty()) {
793        return; // no digital text layer (image/scanned page) — keep all pictures
794    }
795    // A text-document page carries several text-bearing non-picture regions (so a
796    // spurious margin picture is clearly extra). A slide / figure page has at most
797    // one — there the picture is the content, so never drop it.
798    let content_regions = regions
799        .iter()
800        .filter(|r| r.label != "picture" && !region_text(r, cells).trim().is_empty())
801        .count();
802    if content_regions < 2 {
803        return;
804    }
805    let page_area = (page_w * page_h).max(1.0);
806    regions.retain(|r| {
807        if r.label != "picture" || r.score >= 0.5 {
808            return true;
809        }
810        if area(r.l, r.t, r.r, r.b) / page_area >= 0.25 {
811            return true; // a dominant figure, not a margin artifact
812        }
813        // Keep it if any text cell falls mostly inside (a real captioned/labelled
814        // figure); drop only the genuinely empty low-confidence boxes.
815        cells.iter().any(|c| {
816            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
817            !c.text.trim().is_empty() && inter(r, c.l, c.t, c.r, c.b) / ca > 0.5
818        })
819    });
820}
821
822/// A small digit-only region in the top/bottom margin: a page number. docling
823/// emits `right_to_left_02`'s bottom `11` as the page's *first* text item (its
824/// reading-order model floats the page number to the front), whereas our
825/// position-based ordering would place a bottom region last.
826fn is_page_number(region: &Region, cells: &[TextCell], page_h: f32) -> bool {
827    let t = region_text(region, cells);
828    let t = t.trim();
829    !t.is_empty()
830        && t.chars().all(|c| c.is_ascii_digit())
831        && (region.b - region.t).abs() < 30.0
832        && (region.t < page_h * 0.12 || region.b > page_h * 0.88)
833}
834
835/// docling's `form` / `key_value_region` *containers* (2.123, docling#4064):
836/// every region sitting > 0.8 inside one — text, list items, and since #4064
837/// tables and pictures too — is that container's child. Children are
838/// reading-ordered among themselves and emitted as one block where the
839/// container falls in the page's top-level order (a `form_area` /
840/// `key_value_area` group upstream), instead of interleaving with the text
841/// around the form. A child inside several containers belongs to the smallest
842/// (then most confident, then first); a container with children shrinks to
843/// their union for the top-level ordering, like upstream's bbox adjustment.
844///
845/// The containers themselves are still not emitted (`is_skipped`), so the
846/// Markdown is exactly upstream's — a group prints only its children.
847///
848/// `cids` are the items' positions in docling's assembly order
849/// ([`cluster_cids`]) — the reading-order predictor's same-row rule (#424)
850/// pairs consecutive ones, within the top level and within each container.
851fn order_with_containers<T: Clone>(
852    items: &mut Vec<T>,
853    cids: &[usize],
854    page_w: f32,
855    page_h: f32,
856    reg: impl Fn(&T) -> &Region,
857) {
858    let is_container = |r: &Region| matches!(r.label, "form" | "key_value_region");
859    let containers: Vec<usize> = (0..items.len())
860        .filter(|&i| is_container(reg(&items[i])))
861        .collect();
862    if containers.is_empty() {
863        order_regions(items, cids, page_w, page_h, reg);
864        return;
865    }
866    // Parent container per item (containers never nest in each other here —
867    // upstream assigns regulars and tables/pictures only).
868    let mut parent: Vec<Option<usize>> = vec![None; items.len()];
869    for i in 0..items.len() {
870        let r = reg(&items[i]);
871        if is_container(r) {
872            continue;
873        }
874        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
875        let mut best: Option<(usize, f32, f32)> = None; // (idx, area, -score)
876        for &c in &containers {
877            let cr = reg(&items[c]);
878            if inter(r, cr.l, cr.t, cr.r, cr.b) / ra > 0.8 {
879                let key = (area(cr.l, cr.t, cr.r, cr.b), -cr.score);
880                if best.is_none_or(|(_, a, s)| key.0 < a || (key.0 == a && key.1 < s)) {
881                    best = Some((c, key.0, key.1));
882                }
883            }
884        }
885        parent[i] = best.map(|(c, _, _)| c);
886    }
887    // Top-level pass: non-children plus the containers, the latter shrunk to
888    // their children's union.
889    let mut top: Vec<(usize, Region)> = Vec::new();
890    for i in 0..items.len() {
891        if parent[i].is_some() {
892            continue;
893        }
894        let mut r = reg(&items[i]).clone();
895        if is_container(&r) {
896            let kids: Vec<&Region> = (0..items.len())
897                .filter(|&k| parent[k] == Some(i))
898                .map(|k| reg(&items[k]))
899                .collect();
900            if !kids.is_empty() {
901                r.l = kids.iter().map(|k| k.l).fold(f32::INFINITY, f32::min);
902                r.t = kids.iter().map(|k| k.t).fold(f32::INFINITY, f32::min);
903                r.r = kids.iter().map(|k| k.r).fold(f32::NEG_INFINITY, f32::max);
904                r.b = kids.iter().map(|k| k.b).fold(f32::NEG_INFINITY, f32::max);
905            }
906        }
907        top.push((i, r));
908    }
909    let top_cids: Vec<usize> = top.iter().map(|(i, _)| cids[*i]).collect();
910    order_regions(&mut top, &top_cids, page_w, page_h, |it| &it.1);
911    let mut out: Vec<T> = Vec::with_capacity(items.len());
912    for (i, _) in top {
913        if is_container(reg(&items[i])) {
914            let kid_idx: Vec<usize> = (0..items.len()).filter(|&k| parent[k] == Some(i)).collect();
915            let mut kids: Vec<T> = kid_idx.iter().map(|&k| items[k].clone()).collect();
916            let kid_cids: Vec<usize> = kid_idx.iter().map(|&k| cids[k]).collect();
917            order_regions(&mut kids, &kid_cids, page_w, page_h, &reg);
918            out.push(items[i].clone());
919            out.extend(kids);
920        } else {
921            out.push(items[i].clone());
922        }
923    }
924    *items = out;
925}
926
927/// Furniture / not-yet-emitted labels.
928fn is_skipped(label: &str) -> bool {
929    matches!(
930        label,
931        "page_header" | "page_footer" | "form" | "key_value_region"
932    )
933}
934
935/// Reading-order sort of a page's regions, via the ported rule-based
936/// [`reading_order`](crate::reading_order) predictor (docling's
937/// `ReadingOrderPredictor`): an up/down geometry graph with same-row links
938/// between `cids`-consecutive elements (#424), horizontal dilation and a
939/// depth-first traversal, with `page_header`/`page_footer` ordered as their own
940/// groups (first/last) as docling does.
941fn order_regions<T: Clone>(
942    items: &mut Vec<T>,
943    cids: &[usize],
944    page_w: f32,
945    page_h: f32,
946    reg: impl Fn(&T) -> &Region,
947) {
948    let boxes: Vec<(f32, f32, f32, f32)> = items
949        .iter()
950        .map(|it| {
951            let r = reg(it);
952            (r.l, r.t, r.r, r.b)
953        })
954        .collect();
955    let is_header: Vec<bool> = items
956        .iter()
957        .map(|it| reg(it).label == "page_header")
958        .collect();
959    let is_footer: Vec<bool> = items
960        .iter()
961        .map(|it| reg(it).label == "page_footer")
962        .collect();
963    let order =
964        crate::reading_order::order_page(&boxes, cids, &is_header, &is_footer, page_w, page_h);
965    *items = order.iter().map(|&i| items[i].clone()).collect();
966}
967
968/// docling's assembly order of a page's clusters (`LayoutPostprocessor`'s
969/// final `_sort_clusters(mode="id")`, #424): each region's rank when sorted by
970/// its first source cell, then by top edge, then left edge; a region with no
971/// cells sorts after every one that has some. docling numbers its page
972/// elements (`cid`) in this order, and the reading-order predictor's same-row
973/// rule pairs elements with consecutive numbers, so the ranks are what
974/// [`order_with_containers`] hands the predictor.
975///
976/// A regular region's first cell is the smallest index among the cells it
977/// claims. A table, picture or container has no cells of its own upstream
978/// either — its cells are its *children's*: the regular clusters > 0.8 inside
979/// it, and upstream every cell no regular cluster claimed is an orphan cluster
980/// of its own, so a table's interior text (which no regular cluster claims)
981/// reaches the table through those orphans. Here that is the cells > 0.8
982/// inside the region plus the claimed cells of the regular regions > 0.8
983/// inside it. Without the interior cells every table would sort last, and two
984/// side-by-side tables would then be consecutive and row-linked — reading the
985/// right table's caption ahead of the left column's headings (2206 page 8).
986pub fn cluster_cids(regions: &[Region], cells: &[TextCell]) -> Vec<usize> {
987    let owned = assign_cells(regions, cells);
988    let first_cell: Vec<usize> = regions
989        .iter()
990        .enumerate()
991        .map(|(i, r)| {
992            if claims_cells(r) {
993                return owned[i].iter().copied().min().unwrap_or(usize::MAX);
994            }
995            let interior = cells
996                .iter()
997                .enumerate()
998                .filter(|(_, c)| {
999                    !c.text.trim().is_empty()
1000                        && inter(r, c.l, c.t, c.r, c.b) / area(c.l, c.t, c.r, c.b).max(1.0) > 0.8
1001                })
1002                .map(|(ci, _)| ci)
1003                .min();
1004            let children = regions
1005                .iter()
1006                .enumerate()
1007                .filter(|(j, child)| {
1008                    *j != i && claims_cells(child) && {
1009                        let ca = area(child.l, child.t, child.r, child.b).max(1.0);
1010                        inter(r, child.l, child.t, child.r, child.b) / ca > 0.8
1011                    }
1012                })
1013                .filter_map(|(j, _)| owned[j].iter().copied().min())
1014                .min();
1015            interior
1016                .into_iter()
1017                .chain(children)
1018                .min()
1019                .unwrap_or(usize::MAX)
1020        })
1021        .collect();
1022    let mut by_source: Vec<usize> = (0..regions.len()).collect();
1023    // Stable, like Python's `sorted`: full ties keep the layout order.
1024    by_source.sort_by(|&a, &b| {
1025        first_cell[a]
1026            .cmp(&first_cell[b])
1027            .then(regions[a].t.total_cmp(&regions[b].t))
1028            .then(regions[a].l.total_cmp(&regions[b].l))
1029    });
1030    let mut cids = vec![0; regions.len()];
1031    for (rank, &i) in by_source.iter().enumerate() {
1032        cids[i] = rank;
1033    }
1034    cids
1035}
1036
1037/// Clean a region's assembled text: undo soft-hyphen line wraps, map curly
1038/// quotes and the ellipsis to ASCII (matching docling), and collapse runs of
1039/// whitespace. pdfium emits the line-wrap hyphen as U+0002 in this corpus
1040/// (U+00AD elsewhere), so `word\u{2} continuation` is one hyphenated word —
1041/// drop the hyphen + the joining space and merge (`com\u{2} pact` → `compact`,
1042/// `end-to\u{2} end` → `end-toend`), exactly as docling does.
1043///
1044/// Token spacing is otherwise left as the geometric join produced it. We do not
1045/// tighten punctuation spacing: docling preserves the PDF's own spaces (it keeps
1046/// `{ ahn }`, `Name 1 .`, `[ 9 ]`), and a geometric gap heuristic diverges from
1047/// it more than a plain single-space join does.
1048/// An ordered-list enumeration marker at the start of a list item: leading ASCII
1049/// digits followed by `.`, e.g. `1. Undo/Redo` → `(1, "Undo/Redo")`. Returns
1050/// `None` when the text doesn't start with `digits.`.
1051fn parse_ordered_marker(s: &str) -> Option<(u64, String)> {
1052    let digits: String = s.chars().take_while(|c| c.is_ascii_digit()).collect();
1053    if digits.is_empty() {
1054        return None;
1055    }
1056    let rest = s[digits.len()..].strip_prefix('.')?;
1057    let number = digits.parse().ok()?;
1058    Some((number, rest.trim_start().to_string()))
1059}
1060
1061/// Escape markdown special characters the way docling-core's markdown serializer
1062/// does (`markdown.py` post_process): `_` → `\_`, then HTML-escape `&`, `<`, `>`
1063/// (quote=False, so quotes are left). Applied to prose (headings, list items,
1064/// paragraphs); code blocks, the formula placeholder, and table cells are left raw.
1065fn md_escape(text: &str) -> String {
1066    text.replace('_', "\\_")
1067        .replace('&', "&amp;")
1068        .replace('<', "&lt;")
1069        .replace('>', "&gt;")
1070}
1071
1072fn clean_text(text: &str) -> String {
1073    // Typographic-quote normalization follows docling-parse's sanitizer table
1074    // (`pdf_sanitators/constants.h`): every curly quote — single *and double* —
1075    // becomes the ASCII apostrophe `'`, and `‚` a comma. A `"` in docling's
1076    // output only ever comes from a literal `quotedbl` glyph, never from `“ ”`
1077    // (2206's `'text in the wild"` pairs a curly open with a literal-quote
1078    // close). This replaces an earlier Hangul-only special case that patched
1079    // one symptom of mapping `“ ”` to `"`.
1080    let replaced = text
1081        .replace("\u{2} ", "")
1082        .replace("\u{ad} ", "")
1083        .replace(['\u{2}', '\u{ad}'], "") // any stray wrap hyphens not at a join
1084        .replace(
1085            [
1086                '\u{2018}', '\u{2019}', '\u{201b}', '\u{201c}', '\u{201d}', '\u{201e}', '\u{201f}',
1087            ],
1088            "'",
1089        ) // ‘ ’ ‛ “ ” „ ‟ → '
1090        .replace('\u{201a}', ",") // ‚ → ,
1091        .replace(
1092            [
1093                '\u{2010}', '\u{2011}', '\u{2012}', '\u{2013}', '\u{2014}', '\u{2015}', '\u{2212}',
1094            ],
1095            "-",
1096        ) // hyphen/dash family → -
1097        .replace('\u{2044}', "/") // ⁄ fraction slash → /
1098        .replace('\u{2022}', "\u{b7}") // • → · (docling never emits •; inline CCS-concept separators)
1099        .replace('\u{2026}', "..."); // … → ...
1100    let out = if crate::pdfium_backend::use_dp_lines() {
1101        // The docling-parse sanitizer already placed the correct spacing (e.g.
1102        // justified double spaces); preserve internal runs of spaces, only
1103        // normalizing line breaks/tabs and trimming the ends.
1104        replaced.replace(['\n', '\r', '\t'], " ").trim().to_string()
1105    } else {
1106        // Legacy: collapse all whitespace runs to single spaces.
1107        replaced.split_whitespace().collect::<Vec<_>>().join(" ")
1108    };
1109    fix_arabic_lam_alef(&out)
1110}
1111
1112/// pdfium decomposes the Arabic lam-alef ligature (لا / لإ / لأ / لآ) into its
1113/// glyph constituents in *visual* order — `alef-variant, lam` — but docling keeps
1114/// logical order, `lam, alef-variant`. Swap a mid-word `alef-variant + lam` back
1115/// to `lam + alef-variant`. "Mid-word" (the previous char is an Arabic letter)
1116/// distinguishes the ligature from the definite article `ال` (word-initial
1117/// `alef + lam`), which must stay. No-op for non-Arabic text.
1118fn fix_arabic_lam_alef(s: &str) -> String {
1119    let is_arabic_letter = |c: char| ('\u{0620}'..='\u{064A}').contains(&c);
1120    let chars: Vec<char> = s.chars().collect();
1121    if !chars.iter().any(|&c| is_arabic_letter(c)) {
1122        return s.to_string(); // no-op for non-Arabic text
1123    }
1124    // Pass 1: swap mid-word `alef-variant + lam` → `lam + alef-variant`. Only the
1125    // hamza/madda alef variants (إ أ آ) are safe: the definite article is always
1126    // plain `ا + ل`, so plain `alef + lam` is ambiguous (a legitimate `فعالة` vs a
1127    // reversed `لا` ligature look identical) — leaving plain alef alone avoids
1128    // corrupting legitimate words.
1129    let mut a: Vec<char> = Vec::with_capacity(chars.len());
1130    let mut i = 0;
1131    while i < chars.len() {
1132        let c = chars[i];
1133        if matches!(c, '\u{0622}' | '\u{0623}' | '\u{0625}')
1134            && chars.get(i + 1) == Some(&'\u{0644}')
1135            && i > 0
1136            && is_arabic_letter(chars[i - 1])
1137            // A preceding lam means this alef-variant is *already* the logical
1138            // `lam + alef` ligature; the following lam is the next syllable's
1139            // letter, not a reversed ligature — swapping it corrupts `لآل` → `للآ`
1140            // (e.g. التعلم الآلي → الآلي, not اللآي).
1141            && chars[i - 1] != '\u{0644}'
1142        {
1143            a.push('\u{0644}');
1144            a.push(c);
1145            i += 2;
1146            continue;
1147        }
1148        a.push(c);
1149        i += 1;
1150    }
1151    // Pass 2: insert a space at Arabic↔Latin boundaries (bidi script switch) that
1152    // pdfium runs together — docling separates the embedded Latin run (`وPython`
1153    // → `و Python`).
1154    let mut out: Vec<char> = Vec::with_capacity(a.len());
1155    for (j, &c) in a.iter().enumerate() {
1156        if j > 0 {
1157            let p = a[j - 1];
1158            if (is_arabic_letter(p) && c.is_ascii_alphabetic())
1159                || (p.is_ascii_alphabetic() && is_arabic_letter(c))
1160            {
1161                out.push(' ');
1162            }
1163        }
1164        out.push(c);
1165    }
1166    out.into_iter().collect()
1167}
1168
1169/// docling's `PageAssembleModel._match_hyperlink`: the URI whose link
1170/// annotations cover at least half of the region's box, or `None`. Coverage is
1171/// intersection-over-region-area, **accumulated per URI** — a URL that wraps
1172/// across lines carries several annotation rects that sum toward the same
1173/// target. Ties resolve to the first-seen URI (Python's `max` over dict
1174/// insertion order); the winner still needs `>= 0.5`
1175/// (`_HYPERLINK_COVERAGE_THRESHOLD`).
1176pub(crate) fn region_hyperlink(
1177    region: &Region,
1178    links: &[crate::pdfium_backend::LinkAnnot],
1179) -> Option<String> {
1180    if links.is_empty() {
1181        return None;
1182    }
1183    let area = (region.r - region.l).max(0.0) * (region.b - region.t).max(0.0);
1184    if area <= 0.0 {
1185        return None;
1186    }
1187    let mut coverage: Vec<(&str, f32)> = Vec::new();
1188    for link in links {
1189        let ix = (region.r.min(link.r) - region.l.max(link.l)).max(0.0);
1190        let iy = (region.b.min(link.b) - region.t.max(link.t)).max(0.0);
1191        let c = ix * iy / area;
1192        match coverage.iter_mut().find(|(uri, _)| *uri == link.uri) {
1193            Some((_, acc)) => *acc += c,
1194            None => coverage.push((&link.uri, c)),
1195        }
1196    }
1197    let mut best: Option<(&str, f32)> = None;
1198    for (uri, c) in coverage {
1199        // Strictly greater keeps the first-seen URI on ties, like Python's max.
1200        if best.is_none_or(|(_, bc)| c > bc) {
1201            best = Some((uri, c));
1202        }
1203    }
1204    let (uri, c) = best?;
1205    (c >= 0.5).then(|| normalize_uri(uri))
1206}
1207
1208/// The pydantic-`AnyUrl` normalization docling's hyperlink value passes
1209/// through on its way to the serializer: a URL with an authority but no path
1210/// gains a trailing `/` (`https://arxiv.org` → `https://arxiv.org/`). Other
1211/// AnyUrl canonicalizations (scheme/host lowercasing, percent-encoding) don't
1212/// occur in PDF link annotations in practice, so they are not reproduced.
1213fn normalize_uri(uri: &str) -> String {
1214    if let Some((_, rest)) = uri.split_once("://") {
1215        if !rest.is_empty() && !rest.contains(['/', '?', '#']) {
1216            return format!("{uri}/");
1217        }
1218    }
1219    uri.to_string()
1220}
1221
1222/// Resolve each page hyperlink to the visible text it covers, as `(anchor, uri)`
1223/// in reading order. The anchor is the cells whose centre falls in the link rect,
1224/// joined left-to-right and cleaned the same way prose is (so it matches the
1225/// serialized text), deduped against the immediately-preceding link so pdfium's
1226/// occasional duplicate annotation doesn't double-list. Empty anchors are dropped.
1227pub(crate) fn resolve_link_anchors(page: &PdfPage) -> Vec<(String, String)> {
1228    let mut out: Vec<(String, String)> = Vec::new();
1229    // Use per-word cells, not the line-merged `cells`: a link rect covers a few
1230    // words on a line, and a whole merged line cell would over-capture (its centre
1231    // lands in one link's rect, grabbing the entire line as that link's anchor).
1232    let words = if page.word_cells.is_empty() {
1233        &page.cells
1234    } else {
1235        &page.word_cells
1236    };
1237    for link in &page.links {
1238        // A cell participates when its centre row is inside the rect and it
1239        // overlaps the rect horizontally. A cell can be *wider* than the rect:
1240        // PDFs often draw a whole header line as one text run ("LinkedIn |
1241        // GitHub | Credly"), which docling-parse's word grouping keeps as one
1242        // cell even though each label carries its own link annotation —
1243        // centre-in-rect alone would hand the entire line to every link.
1244        // [`cell_text_in_rect`] clips such a cell to the tokens under the rect.
1245        let mut inside: Vec<(&TextCell, String)> = words
1246            .iter()
1247            .filter(|c| {
1248                let cy = (c.t + c.b) / 2.0;
1249                cy >= link.t && cy <= link.b && c.r.min(link.r) > c.l.max(link.l)
1250            })
1251            .filter_map(|c| {
1252                let text = cell_text_in_rect(c, link.l, link.r);
1253                (!text.is_empty()).then_some((c, text))
1254            })
1255            .collect();
1256        // Reading order: top band then left-to-right (link anchors are LTR).
1257        let band = inside
1258            .iter()
1259            .map(|(c, _)| (c.b - c.t).abs())
1260            .fold(0.0f32, f32::max)
1261            .max(1.0);
1262        inside.sort_by_key(|(c, _)| ((c.t / band).round() as i64, (c.l * 10.0) as i64));
1263        let anchor = clean_text(
1264            &inside
1265                .iter()
1266                .map(|(_, t)| t.trim())
1267                .filter(|t| !t.is_empty())
1268                .collect::<Vec<_>>()
1269                .join(" "),
1270        );
1271        if anchor.is_empty() {
1272            continue;
1273        }
1274        if out
1275            .last()
1276            .is_some_and(|(a, u)| a == &anchor && u == &link.uri)
1277        {
1278            continue;
1279        }
1280        out.push((anchor, link.uri.clone()));
1281    }
1282    out
1283}
1284
1285/// The part of a cell's text that lies under a link rect's x-range. A cell
1286/// fully inside the rect (by centre) returns its whole text. A wider cell is
1287/// split into whitespace tokens whose x-spans are estimated proportionally to
1288/// their character positions (kerning makes this approximate, so selection
1289/// snaps to whole tokens, never characters); tokens whose estimated centre
1290/// falls inside the rect are kept. Returns "" when nothing falls inside.
1291fn cell_text_in_rect(c: &TextCell, l: f32, r: f32) -> String {
1292    let cx = (c.l + c.r) / 2.0;
1293    if cx >= l && cx <= r && c.l >= l - (c.r - c.l) * 0.25 && c.r <= r + (c.r - c.l) * 0.25 {
1294        return c.text.trim().to_string();
1295    }
1296    let chars: Vec<char> = c.text.chars().collect();
1297    let n = chars.len();
1298    if n == 0 || c.r <= c.l {
1299        return String::new();
1300    }
1301    let per = (c.r - c.l) / n as f32;
1302    let mut out: Vec<String> = Vec::new();
1303    let mut token = String::new();
1304    let mut start = 0usize;
1305    // A trailing sentinel space flushes the last token.
1306    for (i, &ch) in chars.iter().enumerate().chain(std::iter::once((n, &' '))) {
1307        if ch.is_whitespace() {
1308            if !token.is_empty() {
1309                let mid = c.l + (start as f32 + (i - start) as f32 / 2.0) * per;
1310                if mid >= l && mid <= r {
1311                    out.push(std::mem::take(&mut token));
1312                } else {
1313                    token.clear();
1314                }
1315            }
1316        } else {
1317            if token.is_empty() {
1318                start = i;
1319            }
1320            token.push(ch);
1321        }
1322    }
1323    out.join(" ")
1324}
1325
1326/// Cells assigned to a region (best container), in reading order, joined.
1327fn region_text(region: &Region, cells: &[TextCell]) -> String {
1328    let inside: Vec<&TextCell> = cells
1329        .iter()
1330        .filter(|c| {
1331            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1332            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1333        })
1334        .collect();
1335    cells_text(inside)
1336}
1337
1338/// docling's exclusive cell assignment (`_assign_cells_to_clusters`): every
1339/// non-empty cell goes to the single best-overlapping *regular* region at
1340/// intersection-over-self > 0.2, and each region serializes exactly its
1341/// assigned cells. A cell under two overlapping boxes is emitted once (by the
1342/// better-covering one), and a cell only partially under its region — e.g.
1343/// normal_4pages' big section numeral, ~30 % inside the heading box — still
1344/// joins it (`## 들어가며 1`) instead of leaking as an orphan. Pictures and
1345/// wrappers never claim (docling walks regular clusters only); ties go to the
1346/// first region, like docling's strict `>` best-overlap scan.
1347pub fn region_texts_exclusive(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
1348    let owned = assign_cells(regions, cells);
1349    // Non-claimers (tables/wrappers/pictures) keep the inclusive > 0.5 text:
1350    // docling fills a special cluster's cells from its contained children, and
1351    // downstream table assembly gates on that text being non-empty.
1352    regions
1353        .iter()
1354        .zip(owned)
1355        .map(|(r, cs)| {
1356            if claims_cells(r) {
1357                cells_text(cs.iter().map(|&i| &cells[i]).collect())
1358            } else {
1359                region_text(r, cells)
1360            }
1361        })
1362        .collect()
1363}
1364
1365/// A *regular* region in docling's sense — one that claims cells. Pictures and
1366/// the wrappers (`table`, `document_index`, `form`, `key_value_region`) fill
1367/// their cells from contained children instead.
1368fn claims_cells(r: &Region) -> bool {
1369    r.label != "picture" && !is_wrapper(r.label)
1370}
1371
1372/// docling's `_assign_cells_to_clusters`: each non-empty cell's index goes to
1373/// the single best-overlapping regular region at intersection-over-self > 0.2
1374/// (ties to the first region, like docling's strict `>` scan). One entry per
1375/// region, in region order.
1376fn assign_cells(regions: &[Region], cells: &[TextCell]) -> Vec<Vec<usize>> {
1377    let mut owned: Vec<Vec<usize>> = vec![Vec::new(); regions.len()];
1378    for (ci, c) in cells.iter().enumerate() {
1379        if c.text.trim().is_empty() {
1380            continue;
1381        }
1382        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1383        let mut best: Option<(usize, f32)> = None;
1384        for (i, r) in regions.iter().enumerate() {
1385            if !claims_cells(r) {
1386                continue;
1387            }
1388            let ov = inter(r, c.l, c.t, c.r, c.b) / ca;
1389            if ov > 0.2 && best.is_none_or(|(_, b)| ov > b) {
1390                best = Some((i, ov));
1391            }
1392        }
1393        if let Some((i, _)) = best {
1394            owned[i].push(ci);
1395        }
1396    }
1397    owned
1398}
1399
1400/// docling's regular-cluster refinement after cell assignment
1401/// (`LayoutPostprocessor._process_regular_clusters`, #419), run once the page's
1402/// cells are final and before reading order:
1403///
1404/// 1. every regular region's box becomes the union of the cells it claimed
1405///    (`_adjust_cluster_bboxes` — a regular cluster's bbox *is* its cells'
1406///    bbox; a table's is the union with the model box, and pictures keep
1407///    theirs, so neither is touched here);
1408/// 2. a regular region that claimed no cell is dropped (`keep_empty_clusters`
1409///    is off; a `formula` is kept, as upstream keeps it);
1410/// 3. an orphan text region (`score == 0.0`, from [`add_orphan_regions`]) that
1411///    now sits > 0.8 inside another regular region's fitted box is folded into
1412///    it (`_remove_overlapping_clusters` at containment 0.8, the larger box
1413///    winning the group) — up to three rounds, like upstream's loop.
1414///
1415/// Why it matters: the layout model's box can end partway through a line. That
1416/// line fails the 0.2 claim and becomes an orphan — recoverable — but the
1417/// *model* box still overlaps the orphan's line by a few points, so the
1418/// reading-order graph, which links only strictly-above pairs, gets no edge
1419/// between them and may emit the next paragraph first, stranding the line
1420/// after the paragraph it belongs in (1540 of 6050 text blocks on the #419
1421/// book began mid-sentence). Fitted to its cells, the box ends on a line
1422/// boundary and the orphan slots in between; an orphan the fitted box
1423/// swallows joins the paragraph outright. Cell assignment is untouched: a
1424/// region's fitted box contains every cell it claimed, so
1425/// [`region_texts_exclusive`] hands it the same cells afterwards.
1426///
1427/// A page with no cells yet (a scan before OCR) is left alone: dropping every
1428/// text region for want of cells would be wrong, and the OCR paths call this
1429/// again once the cells exist.
1430pub fn fit_regions_to_cells(regions: &mut Vec<Region>, cells: &[TextCell]) {
1431    if !cells.iter().any(|c| !c.text.trim().is_empty()) {
1432        return;
1433    }
1434    for _ in 0..3 {
1435        let owned = assign_cells(regions, cells);
1436        let mut fitted: Vec<Region> = Vec::with_capacity(regions.len());
1437        for (r, own) in regions.iter().zip(&owned) {
1438            if !claims_cells(r) {
1439                fitted.push(r.clone());
1440                continue;
1441            }
1442            if own.is_empty() {
1443                if r.label == "formula" {
1444                    fitted.push(r.clone());
1445                }
1446                continue;
1447            }
1448            let mut f = r.clone();
1449            f.l = own
1450                .iter()
1451                .map(|&i| cells[i].l)
1452                .fold(f32::INFINITY, f32::min);
1453            f.t = own
1454                .iter()
1455                .map(|&i| cells[i].t)
1456                .fold(f32::INFINITY, f32::min);
1457            f.r = own
1458                .iter()
1459                .map(|&i| cells[i].r)
1460                .fold(f32::NEG_INFINITY, f32::max);
1461            f.b = own
1462                .iter()
1463                .map(|&i| cells[i].b)
1464                .fold(f32::NEG_INFINITY, f32::max);
1465            fitted.push(f);
1466        }
1467        let mut changed = fitted.len() != regions.len();
1468        // Fold orphans into the regular region whose fitted box holds them.
1469        let mut drop = vec![false; fitted.len()];
1470        for i in 0..fitted.len() {
1471            let o = &fitted[i];
1472            if !(o.score == 0.0 && o.label == "text") {
1473                continue;
1474            }
1475            let oa = area(o.l, o.t, o.r, o.b).max(1.0);
1476            let mut best: Option<(usize, f32)> = None;
1477            for (j, r) in fitted.iter().enumerate() {
1478                if j == i || drop[j] || r.score == 0.0 || !claims_cells(r) {
1479                    continue;
1480                }
1481                let ov = inter(r, o.l, o.t, o.r, o.b) / oa;
1482                if ov > 0.8 && best.is_none_or(|(_, b)| ov > b) {
1483                    best = Some((j, ov));
1484                }
1485            }
1486            if let Some((j, _)) = best {
1487                let (l, t, r, b) = (o.l, o.t, o.r, o.b);
1488                let host = &mut fitted[j];
1489                host.l = host.l.min(l);
1490                host.t = host.t.min(t);
1491                host.r = host.r.max(r);
1492                host.b = host.b.max(b);
1493                drop[i] = true;
1494                changed = true;
1495            }
1496        }
1497        let mut drop = drop.into_iter();
1498        fitted.retain(|_| !drop.next().expect("aligned"));
1499        *regions = fitted;
1500        if !changed {
1501            break;
1502        }
1503    }
1504}
1505
1506/// Join a prefiltered cell list into the region's text (docling's
1507/// `sanitize_text` on the docling-parse path, gap-aware band join on legacy).
1508fn cells_text(mut inside: Vec<&TextCell>) -> String {
1509    // Quantize the top coordinate into ~line bands so cells on the same line
1510    // sort in reading order; this is a strict total order (a raw fuzzy comparator
1511    // is not transitive and makes Rust's sort panic). For a right-to-left
1512    // (Arabic-majority) region, cells on a line read right→left, so sort the band
1513    // by descending left edge.
1514    let band = inside
1515        .iter()
1516        .map(|c| (c.b - c.t).abs())
1517        .fold(0.0f32, f32::max)
1518        .max(1.0);
1519    let arabic = inside
1520        .iter()
1521        .flat_map(|c| c.text.chars())
1522        .filter(|&c| ('\u{0600}'..='\u{06FF}').contains(&c))
1523        .count();
1524    let latin = inside
1525        .iter()
1526        .flat_map(|c| c.text.chars())
1527        .filter(|c| c.is_ascii_alphabetic())
1528        .count();
1529    let rtl = arabic > latin;
1530    let dp = crate::pdfium_backend::use_dp_lines();
1531    if dp {
1532        // docling orders a cluster's cells by their docling-parse cell index
1533        // alone (`LayoutPostprocessor._sort_cells`: `sorted(cells, key=c.index)`)
1534        // — the sanitizer's output order, which our `cells` slice already is.
1535        // No geometric re-sort: normal_4pages' big section numerals paint
1536        // *after* their heading text, and docling's `## 들어가며 1` (numeral
1537        // last) only falls out of pure index order — a band sort dragged the
1538        // numeral to the front. The overlap-grouped line restore this replaced
1539        // measured strictly worse on the corpus (it fixed nothing the index
1540        // order broke, and broke the numerals).
1541    } else {
1542        inside.sort_by_key(|c| {
1543            let x = (c.l * 10.0) as i64;
1544            ((c.t / band).round() as i64, if rtl { -x } else { x })
1545        });
1546    }
1547    let joined = if dp {
1548        // docling's `PageAssembleModel.sanitize_text`, ported verbatim over the
1549        // parse-index-ordered lines: append a separating space to a line —
1550        // unless it ends with `-`. A dash-ending line whose last word and the
1551        // next line's first word are both alphanumeric is a wrapped word: the
1552        // dash is dropped and the lines fuse (`platforms-` + `reflects` →
1553        // `platformsreflects`, `pp. 545-` + `561` → `545561`). Any other
1554        // dash-ending line — e.g. the *bare* `-` cell a superscript ORCID or an
1555        // inline `–` bullet splits off (its word list is empty, so the fuse
1556        // test fails) — keeps its dash and still takes no trailing space:
1557        // `[0000` `-` `0002` joins as docling's `[0000 -0002`, and the OTSL
1558        // list's `-` + `"C" cell -` + `a new table cell` collapses to
1559        // `-"C" cell a new table cell`. Our cells still carry the raw dash
1560        // family (docling-parse normalizes to `-` before this; clean_text does
1561        // it after), so the endswith test matches them all.
1562        let texts: Vec<&str> = inside
1563            .iter()
1564            .map(|c| c.text.trim())
1565            // Skip whitespace-only cells (a justified line's trailing space
1566            // glyph): an empty line would double the separator.
1567            .filter(|t| !t.is_empty())
1568            .collect();
1569        let last_word_alnum = |s: &str| {
1570            s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
1571                .rfind(|w| !w.is_empty())
1572                .is_some_and(|w| w.chars().all(char::is_alphanumeric))
1573        };
1574        let first_word_alnum = |s: &str| {
1575            s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
1576                .find(|w| !w.is_empty())
1577                .is_some_and(|w| w.chars().all(char::is_alphanumeric))
1578        };
1579        let mut out = String::new();
1580        for (i, t) in texts.iter().enumerate() {
1581            if i > 0 {
1582                let prev = texts[i - 1];
1583                let dashish = matches!(
1584                    prev.chars().last(),
1585                    Some(
1586                        '-' | '\u{2010}'
1587                            | '\u{2011}'
1588                            | '\u{2012}'
1589                            | '\u{2013}'
1590                            | '\u{2014}'
1591                            | '\u{2015}'
1592                            | '\u{2212}'
1593                    )
1594                );
1595                // docling#4052 (2.122): a dash only splits a word when it is
1596                // *attached* to one — the character before it is alphanumeric.
1597                // A dash that follows whitespace (a separator dash, a bullet
1598                // marker, a wrapped `-prefixed` token, the bare `-` cell an
1599                // ORCID splits off) is a literal character: it is kept and the
1600                // lines join with the ordinary space.
1601                let attached = prev.chars().rev().nth(1).is_some_and(char::is_alphanumeric);
1602                if dashish && attached {
1603                    if last_word_alnum(prev) && first_word_alnum(t) {
1604                        out.pop(); // wrapped word: fuse without the dash
1605                    }
1606                    // an attached dash never takes a separating space
1607                } else {
1608                    out.push(' ');
1609                }
1610            }
1611            out.push_str(t);
1612        }
1613        out
1614    } else {
1615        // Legacy reconstruction: join same-band cells with a space only across a
1616        // real gap, because it can split a word into abutting segments
1617        // (`الت`|`ي` → `التي`).
1618        let mut out = String::new();
1619        let mut prev: Option<&&TextCell> = None;
1620        for c in &inside {
1621            let t = c.text.trim();
1622            if t.is_empty() {
1623                continue;
1624            }
1625            if let Some(p) = prev {
1626                let same_band = ((p.t / band).round() as i64) == ((c.t / band).round() as i64);
1627                let h = (c.b - c.t).abs().max((p.b - p.t).abs()).max(1.0);
1628                let gap = if rtl { p.l - c.r } else { c.l - p.r };
1629                if !same_band || gap > h * 0.25 {
1630                    out.push(' ');
1631                }
1632            }
1633            out.push_str(t);
1634            prev = Some(c);
1635        }
1636        out
1637    };
1638    clean_text(&joined)
1639}
1640
1641/// Tighten the spaces pdfium leaves around tight punctuation in a code line
1642/// (`console .log` → `console.log`, `add (3 , 5)` → `add(3, 5)`), matching
1643/// docling-parse's source spacing.
1644fn tighten_code_punct(s: &str) -> String {
1645    s.replace(" .", ".")
1646        .replace(" ,", ",")
1647        .replace(" ;", ";")
1648        .replace(" )", ")")
1649        .replace(" (", "(")
1650}
1651
1652/// Assemble a **code** region's text with its line structure preserved.
1653///
1654/// Unlike [`region_text`] — which joins every cell with a single space, the right
1655/// thing for prose reflow — a code block's line breaks and indentation are
1656/// significant. The `code_cells` are already one physical source line each
1657/// (grouped space-glyph-only, so monospace runs keep their spacing), so this:
1658///
1659/// 1. groups the cells into vertical line bands and orders them top→bottom,
1660///    left→right;
1661/// 2. joins the lines with `\n` (rather than spaces), keeping the carriage
1662///    returns; and
1663/// 3. reconstructs each line's leading indentation from its left offset, in units
1664///    of the block's estimated monospace character width, so nesting survives.
1665///
1666/// Typography is normalized per line via [`clean_text`] (smart quotes, dashes,
1667/// ellipsis), which never merges lines. Returns an empty string if the region has
1668/// no code cells (the caller falls back to the prose text).
1669fn code_region_text(region: &Region, cells: &[TextCell]) -> String {
1670    let mut inside: Vec<&TextCell> = cells
1671        .iter()
1672        .filter(|c| {
1673            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1674            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1675        })
1676        .filter(|c| !c.text.trim().is_empty())
1677        .collect();
1678    if inside.is_empty() {
1679        return String::new();
1680    }
1681
1682    // Quantize the top edge into ~line bands (like `region_text`), then order the
1683    // cells by band (top→bottom) and, within a band, by left edge.
1684    let band = inside
1685        .iter()
1686        .map(|c| (c.b - c.t).abs())
1687        .fold(0.0f32, f32::max)
1688        .max(1.0);
1689    let line_of = |c: &TextCell| (c.t / band).round() as i64;
1690    inside.sort_by_key(|c| (line_of(c), (c.l * 10.0) as i64));
1691
1692    // Estimate one monospace character's width (total ink width / total glyphs) to
1693    // convert a line's left offset into a count of leading spaces. Measured over
1694    // all lines so a single short line can't skew it.
1695    let (mut total_w, mut total_chars) = (0.0f32, 0usize);
1696    for c in &inside {
1697        let n = c.text.trim().chars().count();
1698        if n > 0 {
1699            total_w += (c.r - c.l).max(0.0);
1700            total_chars += n;
1701        }
1702    }
1703    let char_w = if total_chars > 0 {
1704        (total_w / total_chars as f32).max(1.0)
1705    } else {
1706        1.0
1707    };
1708    // The block's own left margin is the zero-indent baseline.
1709    let base_l = inside.iter().map(|c| c.l).fold(f32::INFINITY, f32::min);
1710
1711    let mut lines: Vec<String> = Vec::new();
1712    let mut cur: Option<i64> = None;
1713    for c in &inside {
1714        // Tighten pdfium's spaced punctuation per line (on the trimmed content, so
1715        // the reconstructed leading indentation is never nibbled).
1716        let text = tighten_code_punct(&clean_text(c.text.trim()));
1717        if Some(line_of(c)) == cur {
1718            // A second cell sharing this band (rare — e.g. split columns): keep it
1719            // on the same source line, separated by a space.
1720            if let Some(last) = lines.last_mut() {
1721                last.push(' ');
1722                last.push_str(&text);
1723            }
1724            continue;
1725        }
1726        let indent = ((c.l - base_l) / char_w).round().max(0.0) as usize;
1727        lines.push(format!("{}{}", " ".repeat(indent), text));
1728        cur = Some(line_of(c));
1729    }
1730    lines.join("\n")
1731}
1732
1733/// Reconstruct a table's grid geometrically from the text cells inside its
1734/// region: cluster cells into rows (by vertical centre) and columns (by clustered
1735/// left edges), then place each cell. A model-free stand-in for TableFormer that
1736/// recovers grid-aligned tables from the precise PDF text layer (it does not
1737/// resolve row/column spans).
1738pub fn reconstruct_table(region: &Region, cells: &[TextCell]) -> Vec<Vec<String>> {
1739    let mut inside: Vec<&TextCell> = cells
1740        .iter()
1741        .filter(|c| {
1742            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1743            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1744        })
1745        .collect();
1746    if inside.is_empty() {
1747        return Vec::new();
1748    }
1749    inside.sort_by(|a, b| a.t.total_cmp(&b.t));
1750
1751    // Rows: consecutive cells whose vertical centre is within ~0.7 line height.
1752    let mut rows: Vec<(f32, Vec<&TextCell>)> = Vec::new();
1753    for c in &inside {
1754        let cyc = (c.t + c.b) / 2.0;
1755        let lh = (c.b - c.t).abs().max(1.0);
1756        if let Some((ryc, row)) = rows.last_mut() {
1757            if (cyc - *ryc).abs() < lh * 0.7 {
1758                row.push(c);
1759                continue;
1760            }
1761        }
1762        rows.push((cyc, vec![c]));
1763    }
1764
1765    // Columns: cluster left edges (merge those within a tolerance).
1766    let tol = {
1767        let mut hs: Vec<f32> = inside.iter().map(|c| (c.b - c.t).abs()).collect();
1768        hs.sort_by(f32::total_cmp);
1769        hs[hs.len() / 2].max(4.0) * 1.5
1770    };
1771    let mut lefts: Vec<f32> = inside.iter().map(|c| c.l).collect();
1772    lefts.sort_by(f32::total_cmp);
1773    let mut col_starts: Vec<f32> = Vec::new();
1774    for l in lefts {
1775        if col_starts.last().is_none_or(|&last| l - last > tol) {
1776            col_starts.push(l);
1777        }
1778    }
1779    let ncols = col_starts.len().max(1);
1780    let col_of = |l: f32| -> usize {
1781        col_starts
1782            .iter()
1783            .rposition(|&s| l + tol * 0.5 >= s)
1784            .unwrap_or(0)
1785            .min(ncols - 1)
1786    };
1787
1788    let mut grid = Vec::with_capacity(rows.len());
1789    for (_, mut row) in rows {
1790        row.sort_by(|a, b| a.l.total_cmp(&b.l));
1791        let mut cols = vec![String::new(); ncols];
1792        for c in row {
1793            let ci = col_of(c.l);
1794            // Strip the wrap-hyphen control char so it never lands in a cell.
1795            let t = c.text.trim().replace(['\u{2}', '\u{ad}'], "");
1796            if cols[ci].is_empty() {
1797                cols[ci] = t;
1798            } else {
1799                cols[ci].push(' ');
1800                cols[ci].push_str(&t);
1801            }
1802        }
1803        grid.push(cols);
1804    }
1805    grid
1806}
1807
1808/// Does the geometric reconstruction of a table look trustworthy enough to use
1809/// as-is, instead of paying for TableFormer?
1810///
1811/// [`reconstruct_table`] derives columns by clustering cell **left edges**. On a
1812/// clean grid that is exact, but when a column's entries are not left-aligned
1813/// (or the OCR boxes wobble) the clustering splits one real column into several,
1814/// and the result is a wide, mostly-empty grid — the "spurious empty columns"
1815/// failure TableFormer exists to fix.
1816///
1817/// Two symptoms separate the two cases, and both are properties of the grid
1818/// alone (no model needed):
1819/// * **density** — a real table is mostly full; a split-up one is mostly holes;
1820/// * **thin columns** — a column carrying at most one entry across several rows
1821///   is almost always a split artefact rather than a real column.
1822///
1823/// Deliberately conservative: it answers `true` only for grids that are plainly
1824/// well-formed, so the expensive path stays the default whenever there is doubt.
1825/// A caller that skips TableFormer on `true` trades no quality for the time.
1826pub fn geometric_table_is_reliable(rows: &[Vec<String>]) -> bool {
1827    let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
1828    // Fewer than two columns is not a grid this heuristic can vouch for: it is
1829    // exactly the shape a collapsed table takes, and TableFormer may recover
1830    // real structure from it.
1831    if rows.len() < 2 || ncols < 2 {
1832        return false;
1833    }
1834    let filled = |c: &String| !c.trim().is_empty();
1835    let total = rows.len() * ncols;
1836    let full = rows.iter().flatten().filter(|c| filled(c)).count();
1837    if (full as f32) < MIN_TABLE_FILL * total as f32 {
1838        return false;
1839    }
1840    // A column used by at most one row, when there are rows enough to tell.
1841    if rows.len() >= 3 {
1842        for ci in 0..ncols {
1843            let used = rows
1844                .iter()
1845                .filter(|r| r.get(ci).is_some_and(filled))
1846                .count();
1847            if used <= 1 {
1848                return false;
1849            }
1850        }
1851    }
1852    true
1853}
1854
1855/// Share of a geometric grid's cells that must carry text for it to be trusted
1856/// without TableFormer. Chosen well above the density a left-edge split
1857/// produces (those land nearer a third) and below what a genuine table with a
1858/// few blank cells reaches.
1859const MIN_TABLE_FILL: f32 = 0.6;
1860
1861/// The union bbox of the text cells assigned to a region (same >50%-overlap
1862/// rule as [`region_text`]), or `None` when no cell lands in it. docling's
1863/// LayoutPostprocessor shrinks a regular cluster's bbox to its cells, and the
1864/// enrichment crops are taken from that cell-tight box — cropping the raw
1865/// detector box instead hands the VLM surrounding chrome (e.g. the `Listing N:`
1866/// caption under a code block) that changes its output.
1867pub fn region_cell_bbox(region: &Region, cells: &[TextCell]) -> Option<[f32; 4]> {
1868    let mut bbox: Option<[f32; 4]> = None;
1869    for c in cells {
1870        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1871        if inter(region, c.l, c.t, c.r, c.b) / ca <= 0.5 {
1872            continue;
1873        }
1874        bbox = Some(match bbox {
1875            None => [c.l, c.t, c.r, c.b],
1876            Some([l, t, r, b]) => [l.min(c.l), t.min(c.t), r.max(c.r), b.max(c.b)],
1877        });
1878    }
1879    bbox
1880}
1881
1882/// One region's enrichment-model result, produced by the pipeline's opt-in
1883/// passes (issue #76) and applied during assembly.
1884#[derive(Debug, Clone)]
1885pub enum Enrichment {
1886    /// DocumentPictureClassifier predictions, descending confidence.
1887    PictureClasses(Vec<PictureClass>),
1888    /// CodeFormulaV2 output for a `code` region: the rewritten source text and
1889    /// the `<_language_>` prefix (when the model emitted one).
1890    Code {
1891        language: Option<String>,
1892        text: String,
1893    },
1894    /// CodeFormulaV2 output for a `formula` region: the decoded LaTeX.
1895    Formula { latex: String },
1896}
1897
1898/// Crop a region (page points, already expanded by the caller if needed) from
1899/// the rendered page image and resize it to `target_scale` pixels per point —
1900/// the enrichment-model equivalent of docling's
1901/// `page.get_image(scale=…, cropbox=…)`, sourced from the existing
1902/// [`crate::pdfium_backend::RENDER_SCALE`] render instead of a fresh pdfium
1903/// pass (the page bitmap is already the exact docling render at scale 2).
1904#[cfg(feature = "ml")]
1905pub fn crop_region_scaled(page: &PdfPage, bbox: [f32; 4], target_scale: f32) -> Option<RgbImage> {
1906    let s = page.scale;
1907    let [l, t, r, b] = bbox;
1908    let (iw, ih) = (page.image.width(), page.image.height());
1909    let x = (l * s).max(0.0) as u32;
1910    let y = (t * s).max(0.0) as u32;
1911    if x >= iw || y >= ih {
1912        return None;
1913    }
1914    let w = (((r - l.max(0.0)) * s) as u32).min(iw - x);
1915    let h = (((b - t.max(0.0)) * s) as u32).min(ih - y);
1916    if w == 0 || h == 0 {
1917        return None;
1918    }
1919    let crop = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
1920    // docling renders the crop at `target_scale` directly; from the scale-2
1921    // page render that is a resize to the same pixel geometry
1922    // (`round(width_points * scale)`, PIL's BICUBIC ≙ CatmullRom).
1923    let tw = ((w as f32 / s) * target_scale).round().max(1.0) as u32;
1924    let th = ((h as f32 / s) * target_scale).round().max(1.0) as u32;
1925    if (tw, th) == (w, h) {
1926        return Some(crop);
1927    }
1928    Some(image::imageops::resize(
1929        &crop,
1930        tw,
1931        th,
1932        image::imageops::FilterType::CatmullRom,
1933    ))
1934}
1935
1936/// Crop a layout region from the rendered page image and encode it as PNG (the
1937/// figure bytes docling stores on a `PictureItem`). Region coordinates are page
1938/// points; the image is rendered at `page.scale`.
1939#[cfg(feature = "ocr-prep")]
1940fn crop_region(page: &PdfPage, region: &Region) -> Option<PictureImage> {
1941    let s = page.scale;
1942    let (iw, ih) = (page.image.width(), page.image.height());
1943    let x = (region.l * s).max(0.0) as u32;
1944    let y = (region.t * s).max(0.0) as u32;
1945    if x >= iw || y >= ih {
1946        return None;
1947    }
1948    let w = (((region.r - region.l) * s) as u32).min(iw - x);
1949    let h = (((region.b - region.t) * s) as u32).min(ih - y);
1950    if w == 0 || h == 0 {
1951        return None;
1952    }
1953    let sub = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
1954    let mut buf = std::io::Cursor::new(Vec::new());
1955    sub.write_to(&mut buf, image::ImageFormat::Png).ok()?;
1956    Some(PictureImage {
1957        mimetype: "image/png".into(),
1958        width: w,
1959        height: h,
1960        data: buf.into_inner(),
1961    })
1962}
1963
1964/// For each `picture` region, find the `caption` region closest below it (and
1965/// horizontally overlapping); docling pairs them and emits the caption first.
1966/// Each caption is claimed by at most one picture.
1967fn pair_captions(regions: &[Region]) -> Vec<Option<usize>> {
1968    let mut pairs = vec![None; regions.len()];
1969    let mut taken = vec![false; regions.len()];
1970    for (pi, p) in regions.iter().enumerate() {
1971        if p.label != "picture" {
1972            continue;
1973        }
1974        let mut best: Option<(usize, f32)> = None;
1975        for (ci, c) in regions.iter().enumerate() {
1976            if c.label != "caption" || taken[ci] {
1977                continue;
1978            }
1979            let line_h = (c.b - c.t).abs().max(1.0);
1980            let gap = c.t - p.b; // caption sits below the picture
1981            let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
1982            if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
1983                let dist = gap.abs();
1984                if best.is_none_or(|(_, bd)| dist < bd) {
1985                    best = Some((ci, dist));
1986                }
1987            }
1988        }
1989        if let Some((ci, _)) = best {
1990            pairs[pi] = Some(ci);
1991            taken[ci] = true;
1992        }
1993    }
1994    pairs
1995}
1996
1997/// Pair each `code` region with the `caption` region just **above** it (a
1998/// `Listing N:` label). docling renders the code block first, then its caption,
1999/// so the caption is consumed from its own (earlier) reading-order slot and
2000/// re-emitted after the code.
2001fn pair_code_captions(regions: &[Region]) -> Vec<Option<usize>> {
2002    let mut pairs = vec![None; regions.len()];
2003    let mut taken = vec![false; regions.len()];
2004    for (pi, p) in regions.iter().enumerate() {
2005        if p.label != "code" {
2006            continue;
2007        }
2008        let mut best: Option<(usize, f32)> = None;
2009        for (ci, c) in regions.iter().enumerate() {
2010            if c.label != "caption" || taken[ci] {
2011                continue;
2012            }
2013            let line_h = (c.b - c.t).abs().max(1.0);
2014            let gap = p.t - c.b; // caption sits above the code
2015            let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
2016            if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
2017                let dist = gap.abs();
2018                if best.is_none_or(|(_, bd)| dist < bd) {
2019                    best = Some((ci, dist));
2020                }
2021            }
2022        }
2023        if let Some((ci, _)) = best {
2024            pairs[pi] = Some(ci);
2025            taken[ci] = true;
2026        }
2027    }
2028    pairs
2029}
2030
2031/// Pair each `table`/`document_index` region with its `caption` (#265) the way
2032/// docling's `ReadingOrderPredictor._find_to_captions` does: by **reading-order
2033/// adjacency**, not geometry. A caption claims the media element
2034/// (table/picture/code) immediately next to it in the ordered region sequence,
2035/// and only when exactly one side holds one — a caption sandwiched between two
2036/// media elements stays unattached, and a text paragraph between caption and
2037/// table breaks the bond. This is what lets a flush-left "Table 3: …" label
2038/// bind a centered grid it doesn't horizontally overlap, while a caption in
2039/// the neighbouring column of a two-column page — geometrically close — never
2040/// pairs across the gutter. Runs after the picture and code pairings (the
2041/// picture/code arms of the same upstream matcher), so a caption they claimed
2042/// stays claimed. docling attaches these as `TableItem.captions` refs; the
2043/// paired caption is consumed from its own reading-order slot and rides on the
2044/// table node instead.
2045fn pair_table_captions(regions: &[Region], taken: &mut [bool]) -> Vec<Option<usize>> {
2046    let is_media = |label: &str| is_table_like(label) || matches!(label, "picture" | "code");
2047    let mut pairs: Vec<Option<usize>> = vec![None; regions.len()];
2048    for ci in 0..regions.len() {
2049        if regions[ci].label != "caption" || taken[ci] {
2050            continue;
2051        }
2052        // Furniture (headers/footers, form chrome) is not part of docling's
2053        // body-element sequence, so it neither bonds nor blocks.
2054        let prev = regions[..ci].iter().rposition(|r| !is_skipped(r.label));
2055        let next = regions[ci + 1..]
2056            .iter()
2057            .position(|r| !is_skipped(r.label))
2058            .map(|off| ci + 1 + off);
2059        let prev_media = prev.is_some_and(|j| is_media(regions[j].label));
2060        let next_media = next.is_some_and(|j| is_media(regions[j].label));
2061        let target = match (prev_media, next_media) {
2062            (true, false) => prev,
2063            (false, true) => next,
2064            // Ambiguous (media on both sides) or no media at all: leave the
2065            // caption in its own reading-order slot, as docling does.
2066            _ => None,
2067        };
2068        if let Some(ti) = target {
2069            // A first claim wins (a table with captions above *and* below
2070            // keeps the earlier one — docling's nearest-first tiebreak).
2071            if is_table_like(regions[ti].label) && pairs[ti].is_none() {
2072                pairs[ti] = Some(ci);
2073                taken[ci] = true;
2074            }
2075        }
2076    }
2077    pairs
2078}
2079
2080/// Assemble one page from its (already overlap-resolved) layout regions and
2081/// text cells.
2082/// Normalize a layout region (page points, top-left origin) to DocLang's 0–511
2083/// location grid: `clamp(round(512 · coord / page_dim), 0, 511)`, per axis,
2084/// order `[x0, y0, x1, y1]`. Mirrors docling_core's
2085/// `_doclang_utils._create_location_tokens_for_bbox` (resolution 512) so the
2086/// emitted `<location>` tokens line up with the Python groundtruth. Our heron
2087/// cluster boxes match docling's to within ~1 grid unit; the residual (mainly
2088/// the aspect-ratio-stretch vs letterbox preprocessing difference) is absorbed
2089/// by the conformance harness's geometry tolerance.
2090fn norm_loc(region: &Region, page_w: f32, page_h: f32) -> [u16; 4] {
2091    let q = |v: f32, dim: f32| -> u16 {
2092        if dim <= 0.0 {
2093            return 0;
2094        }
2095        let g = (512.0 * (v as f64) / (dim as f64)).round() as i64;
2096        g.clamp(0, 511) as u16
2097    };
2098    [
2099        q(region.l, page_w),
2100        q(region.t, page_h),
2101        q(region.r, page_w),
2102        q(region.b, page_h),
2103    ]
2104}
2105
2106/// Wrap a node in its layout provenance so the DocLang serializer emits the four
2107/// `<location>` tokens as the element's head (Markdown/JSON render `inner`
2108/// unchanged).
2109fn located(loc: [u16; 4], inner: Node) -> Node {
2110    Node::Located {
2111        location: loc,
2112        inner: Box::new(inner),
2113    }
2114}
2115
2116/// Stamp the real 1-based page number onto a page's leading marker (see
2117/// [`assemble_page`], which emits it with `page_no: 0` because only the
2118/// document-level collector knows the true index — `--pages` windows shift it).
2119pub fn stamp_page_no(nodes: &mut [Node], page_no: usize) {
2120    if let Some(Node::PageInfo { page_no: p, .. }) = nodes.first_mut() {
2121        *p = page_no;
2122    }
2123}
2124
2125/// A dense table grid plus its first-class cells (#240): `rows` is the text
2126/// grid every serializer renders (spans replicate their anchor's text);
2127/// `cells` are the docling-parity per-cell records (text, page-point bbox,
2128/// span rectangle, OTSL header roles). Produced by the TableFormer paths
2129/// (`tf_core`); lives in this always-compiled module so the pure-text (wasm
2130/// `pdf-text`) build sees the type.
2131#[derive(Clone, Debug)]
2132pub struct TableGrid {
2133    pub rows: Vec<Vec<String>>,
2134    pub cells: Vec<docling_core::TableCell>,
2135}
2136
2137/// docling's `_RICH_CELL_PICTURE_COVERAGE_THRESHOLD`.
2138const RICH_CELL_PICTURE_COVERAGE: f32 = 0.8;
2139
2140/// docling `ReadingOrderModel._match_table_pictures` (#3906, 2.118.1): every
2141/// picture ≥ 80 % inside a TableFormer-structured table on the page is matched
2142/// to the cell covering it, and returned per table as `cell index → pictures`.
2143/// A picture that pairs with a caption stays a standalone figure (upstream
2144/// would nest it and lose the caption; keeping the caption is the better
2145/// failure). Tables without first-class cells (geometric fallback) have no cell
2146/// boxes to match against and nest nothing.
2147fn match_table_pictures(
2148    regions: &[Region],
2149    table_rows: &[Option<TableGrid>],
2150    caption_for: &[Option<usize>],
2151) -> std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> {
2152    let mut out: std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> =
2153        std::collections::HashMap::new();
2154    for (p, pic) in regions.iter().enumerate() {
2155        if pic.label != "picture" || caption_for.get(p).is_some_and(Option::is_some) {
2156            continue;
2157        }
2158        let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2159        let mut best: Option<(f32, usize, usize)> = None; // (coverage, table, cell)
2160        for (t, tbl) in regions.iter().enumerate() {
2161            if !is_table_like(tbl.label) {
2162                continue;
2163            }
2164            let Some(grid) = table_rows.get(t).and_then(Option::as_ref) else {
2165                continue;
2166            };
2167            if inter(pic, tbl.l, tbl.t, tbl.r, tbl.b) / pa < RICH_CELL_PICTURE_COVERAGE {
2168                continue;
2169            }
2170            if let Some((cov, cell)) = match_picture_to_cell(pic, &grid.cells) {
2171                if best.is_none_or(|(b, _, _)| cov > b) {
2172                    best = Some((cov, t, cell));
2173                }
2174            }
2175        }
2176        if let Some((_, t, cell)) = best {
2177            let entry = out.entry(t).or_default();
2178            match entry.iter_mut().find(|(c, _)| *c == cell) {
2179                Some((_, pics)) => pics.push(p),
2180                None => entry.push((cell, vec![p])),
2181            }
2182        }
2183    }
2184    out
2185}
2186
2187/// docling `_match_picture_to_table_cell`: among the cells covering ≥ 80 % of
2188/// the picture, prefer the one at the picture's inferred grid position (the
2189/// row / column whose median cell center is nearest the picture's center —
2190/// cell boxes can overlap across logical rows and columns), else the best
2191/// coverage. Returns `(coverage, cell index)`.
2192fn match_picture_to_cell(pic: &Region, cells: &[docling_core::TableCell]) -> Option<(f32, usize)> {
2193    let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2194    let cover = |b: &[f32; 4]| inter(pic, b[0], b[1], b[2], b[3]) / pa;
2195    let eligible: Vec<(f32, usize)> = cells
2196        .iter()
2197        .enumerate()
2198        .filter_map(|(i, c)| {
2199            let b = c.bbox.as_ref()?;
2200            let cov = cover(b);
2201            (cov >= RICH_CELL_PICTURE_COVERAGE).then_some((cov, i))
2202        })
2203        .collect();
2204    if eligible.is_empty() {
2205        return None;
2206    }
2207    let mut row_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2208    let mut col_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2209    for c in cells {
2210        let Some(b) = c.bbox.as_ref() else { continue };
2211        for r in c.start_row..c.start_row + c.row_span {
2212            row_centers.entry(r).or_default().push((b[1] + b[3]) / 2.0);
2213        }
2214        for k in c.start_col..c.start_col + c.col_span {
2215            col_centers.entry(k).or_default().push((b[0] + b[2]) / 2.0);
2216        }
2217    }
2218    let median = |v: &mut Vec<f32>| -> f32 {
2219        v.sort_by(f32::total_cmp);
2220        let n = v.len();
2221        if n % 2 == 1 {
2222            v[n / 2]
2223        } else {
2224            (v[n / 2 - 1] + v[n / 2]) / 2.0
2225        }
2226    };
2227    let (px, py) = ((pic.l + pic.r) / 2.0, (pic.t + pic.b) / 2.0);
2228    let nearest = |centers: &mut std::collections::BTreeMap<usize, Vec<f32>>, target: f32| {
2229        centers
2230            .iter_mut()
2231            .map(|(&i, v)| (i, (median(v) - target).abs()))
2232            .min_by(|a, b| a.1.total_cmp(&b.1))
2233            .map(|(i, _)| i)
2234    };
2235    let row = nearest(&mut row_centers, py);
2236    let col = nearest(&mut col_centers, px);
2237    let logical: Vec<(f32, usize)> = eligible
2238        .iter()
2239        .copied()
2240        .filter(|&(_, i)| {
2241            let c = &cells[i];
2242            row.is_some_and(|r| c.start_row <= r && r < c.start_row + c.row_span)
2243                && col.is_some_and(|k| c.start_col <= k && k < c.start_col + c.col_span)
2244        })
2245        .collect();
2246    let pool = if logical.is_empty() {
2247        &eligible
2248    } else {
2249        &logical
2250    };
2251    // Python's `max` over `(coverage, cell_index, cell)` tuples: highest
2252    // coverage, ties to the higher index.
2253    pool.iter()
2254        .copied()
2255        .max_by(|a, b| a.0.total_cmp(&b.0).then(a.1.cmp(&b.1)))
2256}
2257
2258/// The DocLang structure overlay derived from first-class cells: span
2259/// continuations (`lcel`/`ucel`/`xcel`) and per-cell header roles, so the
2260/// PDF path's DCLX carries real spans instead of a flat grid.
2261fn structure_from_cells(
2262    cells: &[docling_core::TableCell],
2263    nrows: usize,
2264    ncols: usize,
2265) -> docling_core::TableStructure {
2266    let grid = || vec![vec![false; ncols]; nrows];
2267    let mut col_cont = grid();
2268    let mut row_cont = grid();
2269    let mut row_header = grid();
2270    let mut col_header = grid();
2271    for c in cells {
2272        for r in c.start_row..(c.start_row + c.row_span).min(nrows) {
2273            for k in c.start_col..(c.start_col + c.col_span).min(ncols) {
2274                col_cont[r][k] = k > c.start_col;
2275                row_cont[r][k] = r > c.start_row;
2276                row_header[r][k] = c.row_header;
2277                col_header[r][k] = c.column_header;
2278            }
2279        }
2280    }
2281    docling_core::TableStructure {
2282        header_row: Vec::new(),
2283        col_continuation: col_cont,
2284        row_continuation: row_cont,
2285        row_header,
2286        col_header,
2287    }
2288}
2289
2290pub fn assemble_page(
2291    page: &PdfPage,
2292    regions: Vec<Region>,
2293    table_rows: &[Option<TableGrid>],
2294    enrichments: &[Option<Enrichment>],
2295) -> (Vec<Node>, Vec<(String, String)>) {
2296    let mut nodes: Vec<Node> = Vec::new();
2297    // Every page opens with an invisible page marker carrying its size in
2298    // points — what the JSON export needs to build docling's `pages` map and
2299    // denormalize the 0–511 `<location>` grid into point bboxes (#171). The
2300    // page *number* is stamped by the document-level collector (which knows
2301    // the real 1-based index, `--pages` windows included); every serializer
2302    // except JSON skips the marker, so Markdown/DocLang stay byte-identical.
2303    nodes.push(Node::PageInfo {
2304        page_no: 0,
2305        width: page.width,
2306        height: page.height,
2307    });
2308    // Recover this page's hyperlinks (anchor-precise pairs for strict
2309    // Markdown; whole-item docling-parity links are baked below and their
2310    // pairs dropped from this list so strict output doesn't double-wrap).
2311    let mut links = resolve_link_anchors(page);
2312    // Pair each region with its precomputed TableFormer grid and enrichment
2313    // (indexed by original order) and order by reading order together, so they
2314    // stay aligned.
2315    // docling's assembly order of the regions — what its reading-order
2316    // predictor knows as `cid` (#424) — before they are shuffled.
2317    let cids = cluster_cids(&regions, &page.cells);
2318    type RegionItem = (Region, Option<TableGrid>, Option<Enrichment>);
2319    let mut items: Vec<RegionItem> = regions
2320        .into_iter()
2321        .enumerate()
2322        .map(|(i, r)| {
2323            (
2324                r,
2325                table_rows.get(i).cloned().flatten(),
2326                enrichments.get(i).cloned().flatten(),
2327            )
2328        })
2329        .collect();
2330    order_with_containers(&mut items, &cids, page.width, page.height, |it| &it.0);
2331    // Float a margin page number to the front of reading order (docling parity:
2332    // right_to_left_02's bottom `11` is its first item). Stable, so everything
2333    // else keeps its order; no-op on pages without such a region.
2334    let page_h = page.height;
2335    items.sort_by_key(|(r, _, _)| !is_page_number(r, &page.cells, page_h));
2336    let table_rows: Vec<Option<TableGrid>> = items.iter().map(|(_, t, _)| t.clone()).collect();
2337    let enrichments: Vec<Option<Enrichment>> = items.iter().map(|(_, _, e)| e.clone()).collect();
2338    let regions: Vec<Region> = items.into_iter().map(|(r, _, _)| r).collect();
2339    // docling emits a figure's caption *before* the image marker. Pair each
2340    // picture with the caption region nearest below it and consume that caption,
2341    // so it isn't also emitted in its own (lower) reading-order position.
2342    let caption_for = pair_captions(&regions);
2343    let code_caption_for = pair_code_captions(&regions);
2344    let mut consumed = vec![false; regions.len()];
2345    for ci in caption_for.iter().flatten() {
2346        consumed[*ci] = true;
2347    }
2348    for ci in code_caption_for.iter().flatten() {
2349        consumed[*ci] = true;
2350    }
2351    // Table captions (#265) claim from what the picture/code pairings left.
2352    let mut caption_taken = consumed.clone();
2353    let table_caption_for = pair_table_captions(&regions, &mut caption_taken);
2354    for ci in table_caption_for.iter().flatten() {
2355        consumed[*ci] = true;
2356    }
2357    // Pictures inside a table become rich-cell content (docling#3906, 2.118.1):
2358    // the picture is nested in the cell it covers and not emitted standalone.
2359    let rich_cell_pictures = match_table_pictures(&regions, &table_rows, &caption_for);
2360    for (_, pics) in rich_cell_pictures.values().flatten() {
2361        for &p in pics {
2362            consumed[p] = true;
2363        }
2364    }
2365    // A code block's language label (`XML`, `C#`, …) is chrome, not content — the
2366    // detector emits it as its own region above the code; consume it.
2367    for (i, is_label) in code_language_labels(&regions, &page.cells)
2368        .into_iter()
2369        .enumerate()
2370    {
2371        if is_label {
2372            consumed[i] = true;
2373        }
2374    }
2375
2376    // docling `ReadingOrderPredictor.predict_merges`: join a text fragment with a
2377    // following text fragment strictly to its right (an author column that wraps
2378    // into the next, a paragraph continuing in the next column) into one block —
2379    // the intra-page half of docling's reading-order merges (cross-page/vertical
2380    // continuations stay with [`merge_continuations`]). Already-consumed regions
2381    // (paired captions, code labels) are excluded.
2382    // Exclusive docling cell assignment: computed once for the ordered region
2383    // list and reused for every serialization below, so a cell can never render
2384    // in two regions.
2385    let region_texts: Vec<String> = region_texts_exclusive(&regions, &page.cells);
2386    let is_text: Vec<bool> = regions
2387        .iter()
2388        .enumerate()
2389        .map(|(i, r)| r.label == "text" && !consumed[i])
2390        .collect();
2391    let is_skip: Vec<bool> = regions
2392        .iter()
2393        .enumerate()
2394        .map(|(i, r)| {
2395            consumed[i]
2396                || matches!(
2397                    r.label,
2398                    "page_header" | "page_footer" | "table" | "picture" | "caption" | "footnote"
2399                )
2400        })
2401        .collect();
2402    let boxes: Vec<(f32, f32, f32, f32)> = regions.iter().map(|r| (r.l, r.t, r.r, r.b)).collect();
2403    if docling_core::env::flag("DOCLING_RS_DEBUG_MERGES") {
2404        for (i, r) in regions.iter().enumerate() {
2405            eprintln!(
2406                "MRG {i:2} {} text={} skip={} [{:.0},{:.0},{:.0},{:.0}] {:?}",
2407                r.label,
2408                is_text[i],
2409                is_skip[i],
2410                r.l,
2411                r.t,
2412                r.r,
2413                r.b,
2414                region_texts[i].chars().take(40).collect::<String>()
2415            );
2416        }
2417    }
2418    let mut merge_suffix: Vec<String> = vec![String::new(); regions.len()];
2419    for (head, children) in
2420        crate::reading_order::predict_merges(&boxes, &region_texts, &is_text, &is_skip)
2421            .into_iter()
2422            .enumerate()
2423    {
2424        for c in children {
2425            let t = region_texts[c].trim();
2426            if !t.is_empty() {
2427                merge_suffix[head].push(' ');
2428                merge_suffix[head].push_str(t);
2429            }
2430            consumed[c] = true;
2431        }
2432    }
2433
2434    for (i, region) in regions.iter().enumerate() {
2435        if consumed[i] {
2436            continue;
2437        }
2438        // Page headers/footers: docling emits them as furniture blocks
2439        // (`<page_header>`/`<page_footer>` with a layer + location + text) at
2440        // their reading-order position, not as body — emit them, don't skip.
2441        if matches!(region.label, "page_header" | "page_footer") {
2442            let text = region_texts[i].clone();
2443            if !text.is_empty() {
2444                nodes.push(Node::PageFurniture {
2445                    footer: region.label == "page_footer",
2446                    location: norm_loc(region, page.width, page_h),
2447                    text: md_escape(&text),
2448                });
2449            }
2450            continue;
2451        }
2452        if is_skipped(region.label) {
2453            continue;
2454        }
2455        // Layout provenance for this region, normalized to docling's 0–511 grid.
2456        let loc = norm_loc(region, page.width, page_h);
2457        if region.label == "picture" {
2458            // The figure pixels are cropped from the page render for image export.
2459            // Captions are prose: markdown-escaped like a paragraph (the JSON
2460            // export unescapes back to the raw text, matching docling).
2461            let caption = caption_for[i]
2462                .map(|ci| md_escape(&region_texts[ci]))
2463                .filter(|t| !t.is_empty());
2464            let classification = match &enrichments[i] {
2465                Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
2466                _ => None,
2467            };
2468            // Without the page render (text-layer-only build) a picture keeps
2469            // its caption/classification but carries no cropped pixels.
2470            #[cfg(feature = "ocr-prep")]
2471            let image = crate::timing::timed("crop_region", || crop_region(page, region));
2472            #[cfg(not(feature = "ocr-prep"))]
2473            let image: Option<PictureImage> = None;
2474            nodes.push(located(
2475                loc,
2476                Node::Picture {
2477                    caption,
2478                    caption_href: None,
2479                    image,
2480                    classification,
2481                    // docling's layout pipeline parents a figure's caption to
2482                    // the picture itself (#390) — the one backend that does.
2483                    caption_parent: CaptionParent::Item,
2484                },
2485            ));
2486            continue;
2487        }
2488        let mut text = region_texts[i].clone();
2489        text.push_str(&merge_suffix[i]);
2490        if text.is_empty() {
2491            continue;
2492        }
2493        match region.label {
2494            // docling assembles checkboxes as TEXT_ELEM items (the region's
2495            // cells are the option label, e.g. right_to_left_03's بلی/خير)
2496            // and its Markdown serializer renders them as task-list lines
2497            // (`- [x] …`) — mirrored by [`Node::CheckboxItem`].
2498            "checkbox_selected" | "checkbox_unselected" => nodes.push(Node::CheckboxItem {
2499                checked: region.label == "checkbox_selected",
2500                text: md_escape(&text),
2501            }),
2502            // docling renders both the document title and section headers as
2503            // `##` (it never emits a top-level `#` for PDFs), so match that.
2504            "title" | "section_header" => nodes.push(located(
2505                loc,
2506                Node::Heading {
2507                    level: 2,
2508                    text: md_escape(&text),
2509                },
2510            )),
2511            // docling drops the rendered bullet glyph; the Markdown serializer
2512            // adds its own `- ` marker. An item whose text opens with an `N.`
2513            // enumeration marker is an ordered item (rendered `N. text`).
2514            // A leading dash stays: it is an ordinary text glyph that
2515            // docling-parse keeps, and docling's items carry it into the
2516            // Markdown (2305's OTSL list renders `- -"C" cell …`) — only the
2517            // symbol-font bullets docling-parse filters out are stripped.
2518            "list_item" => {
2519                let stripped = text
2520                    .trim_start_matches(['•', '◦', '▪', '·', '*'])
2521                    .trim_start()
2522                    .to_string();
2523                if let Some((number, rest)) = parse_ordered_marker(&stripped) {
2524                    nodes.push(Node::ListItem {
2525                        ordered: true,
2526                        number,
2527                        first_in_list: false,
2528                        text: md_escape(&rest),
2529                        level: 0,
2530                        marker: None,
2531                        location: Some(loc),
2532                        dclx: None,
2533                        href: None,
2534                        layer: None,
2535                    });
2536                } else {
2537                    nodes.push(Node::ListItem {
2538                        ordered: false,
2539                        number: 0,
2540                        first_in_list: false,
2541                        text: md_escape(&stripped),
2542                        level: 0,
2543                        // docling keeps the bullet as the DocLang list marker
2544                        // (`<ldiv><marker>·</marker></ldiv>`); Markdown ignores it.
2545                        marker: Some("·".into()),
2546                        location: Some(loc),
2547                        dclx: None,
2548                        href: None,
2549                        layer: None,
2550                    });
2551                }
2552            }
2553            // TableFormer structure (cells + spans, text matched from word cells)
2554            // when available; otherwise geometric grid reconstruction; finally a
2555            // single cell.
2556            "table" | "document_index" => {
2557                // TableFormer grids carry first-class cells (#240: text +
2558                // page-point bbox + span rectangle + OTSL header roles) into
2559                // the public model, and the DocLang structure overlay derives
2560                // from them so DCLX emits real span/header tokens. The
2561                // geometric fallback has no per-cell records.
2562                let (mut rows, cells, structure) = match table_rows[i].clone() {
2563                    Some(grid) => {
2564                        let nrows = grid.rows.len();
2565                        let ncols = grid.rows.first().map_or(0, Vec::len);
2566                        let structure = structure_from_cells(&grid.cells, nrows, ncols);
2567                        (grid.rows, Some(grid.cells), Some(structure))
2568                    }
2569                    None => {
2570                        let rows = reconstruct_table(region, &page.cells);
2571                        let rows = if rows.iter().any(|r| r.len() > 1) {
2572                            rows
2573                        } else {
2574                            vec![vec![text.clone()]]
2575                        };
2576                        (rows, None, None)
2577                    }
2578                };
2579                // The paired caption (#265) rides on the table — docling's
2580                // TableItem.captions ref; Markdown prints it above the grid,
2581                // the JSON export emits the $ref, DocLang the <caption>.
2582                let caption = table_caption_for[i]
2583                    .map(|ci| md_escape(&region_texts[ci]))
2584                    .filter(|t| !t.is_empty());
2585                // Rich cells (docling#3906): the covering cell's blocks are its
2586                // text followed by the nested picture(s). docling's Markdown
2587                // renders a `RichTableCell` through the serializer — the
2588                // group's children joined by blank lines, newlines flattened
2589                // to spaces — so the flat `rows` text becomes
2590                // `text  <!-- image -->`; the first-class `cells` (the JSON
2591                // `table_cells` / `grid`) keep the plain text, as upstream.
2592                let mut cell_blocks: Option<Vec<Vec<Vec<Node>>>> = None;
2593                if let (Some(by_cell), Some(fc)) = (rich_cell_pictures.get(&i), cells.as_ref()) {
2594                    let nrows = rows.len();
2595                    let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
2596                    let mut blocks = vec![vec![Vec::<Node>::new(); ncols]; nrows];
2597                    for (cell_idx, pics) in by_cell {
2598                        let cell = &fc[*cell_idx];
2599                        let (r, c) = (cell.start_row, cell.start_col);
2600                        if r >= nrows || c >= ncols {
2601                            continue;
2602                        }
2603                        let mut parts: Vec<String> = Vec::new();
2604                        let mut cell_nodes: Vec<Node> = Vec::new();
2605                        if !cell.text.trim().is_empty() {
2606                            parts.push(cell.text.clone());
2607                            cell_nodes.push(Node::Paragraph {
2608                                text: cell.text.clone(),
2609                            });
2610                        }
2611                        for &p in pics {
2612                            parts.push("<!-- image -->".to_string());
2613                            let classification = match &enrichments[p] {
2614                                Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
2615                                _ => None,
2616                            };
2617                            #[cfg(feature = "ocr-prep")]
2618                            let image = crop_region(page, &regions[p]);
2619                            #[cfg(not(feature = "ocr-prep"))]
2620                            let image: Option<PictureImage> = None;
2621                            cell_nodes.push(located(
2622                                norm_loc(&regions[p], page.width, page_h),
2623                                Node::Picture {
2624                                    caption: None,
2625                                    caption_href: None,
2626                                    image,
2627                                    classification,
2628                                    caption_parent: Default::default(),
2629                                },
2630                            ));
2631                        }
2632                        let rendered = parts.join("  ");
2633                        for row in rows.iter_mut().skip(r).take(cell.row_span) {
2634                            for slot in row.iter_mut().skip(c).take(cell.col_span) {
2635                                *slot = rendered.clone();
2636                            }
2637                        }
2638                        blocks[r][c] = cell_nodes;
2639                    }
2640                    cell_blocks = Some(blocks);
2641                }
2642                nodes.push(located(
2643                    loc,
2644                    Node::Table(Table {
2645                        rows,
2646                        location: None,
2647                        structure,
2648                        cell_blocks,
2649                        cells,
2650                        caption,
2651                        // As for pictures: the caption is the table's child.
2652                        caption_parent: CaptionParent::Item,
2653                    }),
2654                ));
2655            }
2656            // With formula enrichment the CodeFormula model decodes the region
2657            // to LaTeX; otherwise docling emits a placeholder comment rather
2658            // than the (garbled) raw glyph text.
2659            "formula" => match &enrichments[i] {
2660                Some(Enrichment::Formula { latex }) => nodes.push(Node::Formula {
2661                    latex: latex.clone(),
2662                    orig: text.clone(),
2663                    location: Some(loc),
2664                }),
2665                _ => nodes.push(Node::Paragraph {
2666                    text: "<!-- formula-not-decoded -->".into(),
2667                }),
2668            },
2669            // Code blocks: use the space-glyph-only grouping (monospace keeps its
2670            // source spacing) and emit a fenced block, preserving the line breaks
2671            // and indentation of the source (unlike prose, which reflows). pdfium
2672            // still inserts spaces around tight punctuation (`console .log`,
2673            // `add (3 , 5)`); tighten them to match docling-parse's source spacing.
2674            "code" => {
2675                // `code_region_text` preserves line breaks/indentation and tightens
2676                // each line itself; the fallback prose `text` is tightened here.
2677                let code = code_region_text(region, &page.code_cells);
2678                let code = if code.is_empty() {
2679                    tighten_code_punct(&text)
2680                } else {
2681                    code
2682                };
2683                // With code enrichment the CodeFormula model rewrites the block
2684                // (and names its language); `orig` keeps the raw extraction in
2685                // docling's shape — its parser has no line-preserving code
2686                // path, so its `orig` is the same code with the lines joined
2687                // by single spaces (indentation collapsed).
2688                // docling's parser has no line-preserving code path — its code
2689                // items carry the lines joined by single spaces. That flat
2690                // form is what every byte-conformance surface serializes
2691                // (legacy Markdown, JSON, DocLang); the line-preserving
2692                // extraction rides in `pretty` for strict Markdown only.
2693                let flat = code
2694                    .lines()
2695                    .map(str::trim)
2696                    .filter(|l| !l.is_empty())
2697                    .collect::<Vec<_>>()
2698                    .join(" ");
2699                let node = match &enrichments[i] {
2700                    Some(Enrichment::Code {
2701                        language,
2702                        text: enriched,
2703                    }) => Node::Code {
2704                        language: language.clone(),
2705                        text: enriched.clone(),
2706                        orig: Some(flat),
2707                        pretty: None,
2708                    },
2709                    _ => Node::Code {
2710                        language: None,
2711                        text: flat,
2712                        orig: None,
2713                        pretty: Some(code),
2714                    },
2715                };
2716                nodes.push(located(loc, node));
2717                // docling emits the `Listing N:` caption after the code block.
2718                if let Some(ci) = code_caption_for[i] {
2719                    let cap = md_escape(&region_texts[ci]);
2720                    if !cap.is_empty() {
2721                        nodes.push(Node::Paragraph { text: cap });
2722                    }
2723                }
2724            }
2725            // text, caption, footnote → paragraph
2726            _ => {
2727                // docling parity (`PageAssembleModel._match_hyperlink`): when
2728                // link annotations cover ≥ half of the region's box, the
2729                // hyperlink attaches to the item and the legacy Markdown
2730                // serializer wraps its full text — 2206.01062's footnote URLs
2731                // render as `[1 https://…](https://…)`. Sparse in-paragraph
2732                // citation links stay below the 0.5 coverage threshold and
2733                // remain plain text, exactly like docling.
2734                //
2735                // Scope: **footnote regions only.** Upstream's page_assemble
2736                // matches every TEXT_ELEM label, but published docling
2737                // observably carries the hyperlink into the document only for
2738                // footnote items — in both committed groundtruth generations
2739                // (docling-JSON and Markdown, independent runs) the fully
2740                // covered plain-text DOI line of 2206.01062 page 1 has
2741                // `hyperlink: None` while the equally covered footnotes carry
2742                // theirs. The corpus is the conformance reference, so match
2743                // the observed behavior; widen the label set if a future
2744                // groundtruth refresh starts linking plain text too.
2745                let escaped = md_escape(&text);
2746                let hyperlink = (region.label == "footnote")
2747                    .then(|| region_hyperlink(region, &page.links))
2748                    .flatten();
2749                let text = match hyperlink {
2750                    Some(uri) => {
2751                        // The strict-mode anchor pairs this item covers are
2752                        // superseded by the baked whole-item link.
2753                        links.retain(|(anchor, href)| {
2754                            !(href == &uri && region_texts[i].contains(anchor.as_str()))
2755                        });
2756                        format!("[{escaped}]({uri})")
2757                    }
2758                    None => escaped,
2759                };
2760                nodes.push(located(loc, Node::Paragraph { text }))
2761            }
2762        }
2763    }
2764    // A `/Rotate`-normalized scanned page (see `pdfium_backend`) was assembled
2765    // in upright space; rotate the finished geometry back so locations and the
2766    // page size are display-space, like docling and every viewer report them.
2767    if page.rotation != 0 {
2768        rotate_nodes_to_display(&mut nodes, page.rotation);
2769    }
2770    (nodes, links)
2771}
2772
2773/// Rotate one 0–511 location bbox 90° clockwise on the grid (top-left origin):
2774/// `(x, y) → (511 - y, x)`.
2775fn rot_loc_cw(l: [u16; 4]) -> [u16; 4] {
2776    [511 - l[3], l[0], 511 - l[1], l[2]]
2777}
2778
2779/// Map upright-space geometry back to display space for a page whose `/Rotate`
2780/// was normalized away before inference: every `<location>` rotates `rot`°
2781/// clockwise on the 0–511 grid (the grid is per-axis normalized, so no page
2782/// dims are needed), and the `PageInfo` size returns to the display box. Node
2783/// text and order are untouched — reading order was decided upright, which is
2784/// the whole point.
2785fn rotate_nodes_to_display(nodes: &mut [Node], rot: u16) {
2786    let quarter_turns = (rot / 90) as usize;
2787    let rot_loc = |l: &mut [u16; 4]| {
2788        for _ in 0..quarter_turns {
2789            *l = rot_loc_cw(*l);
2790        }
2791    };
2792    fn walk(node: &mut Node, rot_loc: &impl Fn(&mut [u16; 4]), swap_dims: bool) {
2793        match node {
2794            Node::PageInfo { width, height, .. } => {
2795                if swap_dims {
2796                    std::mem::swap(width, height);
2797                }
2798            }
2799            Node::Located { location, inner } => {
2800                rot_loc(location);
2801                walk(inner, rot_loc, swap_dims);
2802            }
2803            Node::Furniture { inner, .. } => walk(inner, rot_loc, swap_dims),
2804            Node::Group { children, .. } => {
2805                for c in children {
2806                    walk(c, rot_loc, swap_dims);
2807                }
2808            }
2809            Node::ListItem { location, .. }
2810            | Node::Formula { location, .. }
2811            | Node::Chart { location, .. } => {
2812                if let Some(l) = location {
2813                    rot_loc(l);
2814                }
2815            }
2816            Node::PageFurniture { location, .. } => rot_loc(location),
2817            Node::Table(t) => {
2818                if let Some(l) = &mut t.location {
2819                    rot_loc(l);
2820                }
2821            }
2822            _ => {}
2823        }
2824    }
2825    let swap_dims = quarter_turns % 2 == 1;
2826    for node in nodes {
2827        walk(node, &rot_loc, swap_dims);
2828    }
2829}
2830
2831/// Merge paragraph fragments split across a column or page break. docling joins a
2832/// paragraph whose previous fragment ends mid-sentence (a letter, not sentence
2833/// punctuation) with a lowercase continuation: `…definition of` + `lists in…` →
2834/// `…definition of lists in…`. The fragments are consecutive paragraphs, or
2835/// separated only by figure(s) the text wraps around: a column whose body flows
2836/// past a figure resumes below it (`…The wing type that is` ⟶[figure]⟶ `the most
2837/// common…`), and docling emits the whole paragraph before the figure. A heading,
2838/// table, or list between them ends the paragraph (no merge).
2839/// A paragraph that is really a figure/table caption (`Fig. 1. …`, `Table 2 …`).
2840/// Used to skip an unpaired caption when stitching a paragraph that wraps around
2841/// a figure.
2842fn looks_like_caption(text: &str) -> bool {
2843    let head: String = text.trim_start().chars().take(14).collect();
2844    (head.starts_with("Fig") || head.starts_with("Table"))
2845        && head.contains(|c: char| c.is_ascii_digit())
2846}
2847
2848/// A paragraph fragment is "open" — i.e. it might continue into the next
2849/// paragraph — when it ends mid-word (a letter) or with a wrap hyphen/dash.
2850/// docling joins `vocab-` + `ulary` → `vocab- ulary`.
2851fn paragraph_is_open(text: &str) -> bool {
2852    // docling's merge head test (`.+([a-z,\-\u00AD])\s*`): at least two chars,
2853    // ending in an ASCII lowercase letter, a comma, a hyphen, or a soft
2854    // hyphen. The comma matters: 2206's "…In phase four," resumes across the
2855    // page break. Uppercase/non-Latin endings do not merge, exactly as
2856    // upstream (the dash family is already `-` here — clean_text normalized).
2857    let t = text.trim_end();
2858    t.chars().count() >= 2
2859        && t.chars()
2860            .next_back()
2861            .is_some_and(|c| matches!(c, 'a'..='z' | ',' | '-' | '\u{ad}'))
2862}
2863
2864/// The paragraph text inside a node, looking through a [`Node::Located`]
2865/// provenance wrapper (PDF body paragraphs are wrapped since they carry a
2866/// `<location>`). Returns `None` for non-paragraph nodes.
2867fn as_paragraph(n: &Node) -> Option<&str> {
2868    match n {
2869        Node::Paragraph { text } => Some(text),
2870        Node::Located { inner, .. } => match inner.as_ref() {
2871            Node::Paragraph { text } => Some(text),
2872            _ => None,
2873        },
2874        _ => None,
2875    }
2876}
2877
2878/// Whether a node is a picture, looking through a [`Node::Located`] wrapper.
2879fn is_picture_node(n: &Node) -> bool {
2880    match n {
2881        Node::Picture { .. } => true,
2882        Node::Located { inner, .. } => matches!(inner.as_ref(), Node::Picture { .. }),
2883        _ => false,
2884    }
2885}
2886
2887/// A node a forward paragraph merge looks straight past: a figure or *table*
2888/// the text wraps around, or a page header/footer that falls between the two
2889/// fragments of a paragraph continuing across a page break (docling's merge
2890/// skip-labels: page_header, page_footer, table, picture, caption, footnote —
2891/// 2206's "…In phase four," resumes after a full caption+table+figure block).
2892fn is_merge_trailer(n: &Node) -> bool {
2893    is_picture_node(n)
2894        || matches!(
2895            n,
2896            Node::PageFurniture { .. } | Node::PageInfo { .. } | Node::Table(_)
2897        )
2898        || matches!(n, Node::Located { inner, .. } if matches!(inner.as_ref(), Node::Table(_)))
2899        || as_paragraph(n).is_some_and(looks_like_caption)
2900}
2901
2902/// Rebuild node `i` as a paragraph with `text`, preserving its `<location>`
2903/// wrapper (and thus provenance) if it had one.
2904fn reparagraph(node: &Node, text: String) -> Node {
2905    match node {
2906        Node::Located { location, .. } => located(*location, Node::Paragraph { text }),
2907        _ => Node::Paragraph { text },
2908    }
2909}
2910
2911pub(crate) fn merge_continuations(nodes: &mut Vec<Node>) {
2912    let mut i = 0;
2913    while i + 1 < nodes.len() {
2914        let Some(a) = as_paragraph(&nodes[i]) else {
2915            i += 1;
2916            continue;
2917        };
2918        // A figure/table caption is a self-contained unit; body text resuming
2919        // after a figure is the continuation case, not the caption itself. Never
2920        // stitch *from* a caption — otherwise a caption that ends in a lone glyph
2921        // (`Fig. 5. … PubTabNet. μ`) would swallow a following stray figure label
2922        // (a standalone `μ`) into `… μ μ`.
2923        if looks_like_caption(a) {
2924            i += 1;
2925            continue;
2926        }
2927        if !paragraph_is_open(a) {
2928            i += 1;
2929            continue;
2930        }
2931        // The continuation is the next paragraph, looking past any figures the
2932        // text wraps around — and a figure/table caption that was emitted as its
2933        // own paragraph (an above-the-figure caption that didn't pair), since the
2934        // body text resumes after the whole figure+caption block.
2935        let mut j = i + 1;
2936        while nodes.get(j).is_some_and(is_merge_trailer) {
2937            j += 1;
2938        }
2939        // docling's continuation regex allows either case, but its merge runs
2940        // over the pre-assembly element stream; at node level an uppercase
2941        // start is overwhelmingly a new sentence/heading fragment (allowing it
2942        // swallowed 2305's formula blocks and redp's chapter openers), so the
2943        // continuation stays lowercase-start here.
2944        let cont = nodes.get(j).and_then(as_paragraph).is_some_and(|b| {
2945            b.trim_start()
2946                .chars()
2947                .next()
2948                .is_some_and(char::is_lowercase)
2949        });
2950        if cont {
2951            let a = as_paragraph(&nodes[i]).unwrap().trim_end().to_string();
2952            let b = as_paragraph(&nodes[j]).unwrap().trim_start().to_string();
2953            // A soft hyphen -- or a hard hyphen followed by a lowercase
2954            // continuation (guaranteed lowercase by the `cont` gate above) --
2955            // is a word split across the break: strip it and join without a
2956            // space, docling#3888 ("vocab-" + "ulary" -> "vocabulary");
2957            // docling's older serializer kept the artifact ("vocab- ulary").
2958            // Everything else joins with the space, as before.
2959            let merged = match a.strip_suffix('\u{ad}').or_else(|| a.strip_suffix('-')) {
2960                Some(stem) => format!("{stem}{b}"),
2961                None => format!("{a} {b}"),
2962            };
2963            // Keep node i's provenance wrapper; docling's merged paragraph keeps
2964            // the first fragment's geometry as its primary location.
2965            nodes[i] = reparagraph(&nodes[i], merged);
2966            nodes.remove(j);
2967            // Re-check i: the merged paragraph may continue further.
2968        } else {
2969            i += 1;
2970        }
2971    }
2972}
2973
2974/// How many leading nodes of `nodes` are safe to flush now — i.e. cannot be
2975/// rewritten by a future [`merge_continuations`] once more pages are appended.
2976///
2977/// A forward merge can only start from an "open" paragraph (ends mid-word) and
2978/// only reaches across trailing pictures and figure/table captions. So we scan
2979/// from the end past those skippable trailers: if the first non-skippable node is
2980/// an open paragraph, it (and the trailers after it) must be held; anything else —
2981/// a closed paragraph, a heading, a table, a list — blocks any forward merge, so
2982/// the whole buffer is safe to flush.
2983fn hold_start(nodes: &[Node]) -> usize {
2984    for k in (0..nodes.len()).rev() {
2985        // Skippable trailers (figures, page furniture, captions): a forward merge
2986        // looks straight past them.
2987        if is_merge_trailer(&nodes[k]) {
2988            continue;
2989        }
2990        match as_paragraph(&nodes[k]) {
2991            // An open body paragraph might still pull a continuation off the next
2992            // page — hold from here to the end.
2993            Some(text) if paragraph_is_open(text) => return k,
2994            // A closed paragraph, heading, table, list, etc. ends the paragraph:
2995            // nothing after it can merge backwards across it. Flush everything.
2996            _ => return nodes.len(),
2997        }
2998    }
2999    // Only skippable trailers (or empty) and no open paragraph to anchor a merge.
3000    nodes.len()
3001}
3002
3003/// Streaming counterpart of [`merge_continuations`]: feed per-page node batches in
3004/// document order and get back the prefix that is final (its cross-page merges are
3005/// resolved and no future page can change it), holding back only the small tail
3006/// that might still merge into the next page. Concatenating every flushed batch
3007/// (then [`finish`](Self::finish)) yields exactly the same nodes as running
3008/// [`merge_continuations`] once over the whole document.
3009pub(crate) struct StreamAssembler {
3010    pending: Vec<Node>,
3011}
3012
3013impl StreamAssembler {
3014    pub(crate) fn new() -> Self {
3015        Self {
3016            pending: Vec::new(),
3017        }
3018    }
3019
3020    /// Append one page's nodes, resolve merges within the buffer, and return the
3021    /// now-final prefix to emit (possibly empty).
3022    pub(crate) fn push(&mut self, mut nodes: Vec<Node>) -> Vec<Node> {
3023        self.pending.append(&mut nodes);
3024        merge_continuations(&mut self.pending);
3025        let cut = hold_start(&self.pending);
3026        let tail = self.pending.split_off(cut);
3027        std::mem::replace(&mut self.pending, tail)
3028    }
3029
3030    /// Flush whatever is left after the last page (the held tail is final once no
3031    /// more pages can follow).
3032    pub(crate) fn finish(self) -> Vec<Node> {
3033        self.pending
3034    }
3035}
3036
3037#[cfg(test)]
3038mod tests {
3039    use super::{cells_text, clean_text};
3040    use super::{code_region_text, merge_continuations, resolve_link_anchors, StreamAssembler};
3041    use crate::layout::Region;
3042    use crate::pdfium_backend::{LinkAnnot, PdfPage, TextCell};
3043    use docling_core::Node;
3044
3045    /// The int8-layout guard's coverage metric: cells under detections count,
3046    /// cells outside don't, whitespace cells are ignored, and a cell-less page
3047    /// reads as fully covered (nothing to rescue).
3048    #[test]
3049    fn layout_cell_coverage_counts_claimed_text_cells() {
3050        let cell = |text: &str, l: f32, t: f32| TextCell {
3051            text: text.into(),
3052            l,
3053            t,
3054            r: l + 40.0,
3055            b: t + 10.0,
3056        };
3057        let region = Region {
3058            label: "text",
3059            score: 0.9,
3060            l: 0.0,
3061            t: 0.0,
3062            r: 100.0,
3063            b: 50.0,
3064        };
3065        let cells = vec![
3066            cell("inside", 10.0, 10.0),
3067            cell("also inside", 10.0, 30.0),
3068            cell("outside", 10.0, 200.0),
3069            cell("   ", 10.0, 210.0), // whitespace: not counted at all
3070        ];
3071        let cov = super::layout_cell_coverage(std::slice::from_ref(&region), &cells);
3072        assert!((cov - 2.0 / 3.0).abs() < 1e-6, "got {cov}");
3073        assert_eq!(super::layout_cell_coverage(&[], &[]), 1.0);
3074        assert_eq!(super::layout_cell_coverage(&[], &cells), 0.0);
3075    }
3076
3077    /// #165: a picture no longer claims cells at 0.2 intersection-over-self.
3078    /// A line straddling the figure border (≤80 % contained) becomes an orphan
3079    /// region and survives the contained-regulars drop — before the fix its
3080    /// cells were silently erased. A line fully inside the picture is still
3081    /// re-dropped, matching docling's Markdown (a picture's children never
3082    /// reach its serializer's output).
3083    #[test]
3084    fn border_straddling_lines_survive_picture_interior_is_still_dropped() {
3085        let pic = Region {
3086            label: "picture",
3087            score: 0.9,
3088            l: 0.0,
3089            t: 0.0,
3090            r: 100.0,
3091            b: 100.0,
3092        };
3093        // ~35 % of this cell overlaps the picture (l=90..120 of 0..100): above
3094        // the old 0.2 claim (was swallowed), below full containment (survives).
3095        let straddler = TextCell {
3096            text: "axis label".into(),
3097            l: 90.0,
3098            t: 40.0,
3099            r: 120.0,
3100            b: 48.0,
3101        };
3102        let interior = TextCell {
3103            text: "in-figure callout".into(),
3104            l: 10.0,
3105            t: 10.0,
3106            r: 60.0,
3107            b: 18.0,
3108        };
3109        let mut regions = vec![pic];
3110        super::add_orphan_regions(&mut regions, &[straddler, interior]);
3111        assert_eq!(
3112            regions.iter().filter(|r| r.label == "text").count(),
3113            2,
3114            "both unclaimed lines become orphans"
3115        );
3116        super::drop_contained_regulars(&mut regions);
3117        let texts: Vec<(f32, f32)> = regions
3118            .iter()
3119            .filter(|r| r.label == "text")
3120            .map(|r| (r.l, r.r))
3121            .collect();
3122        assert_eq!(
3123            texts,
3124            [(90.0, 120.0)],
3125            "the straddler is emitted, the fully-contained callout is not"
3126        );
3127    }
3128
3129    /// docling#3906's concern, pinned on our side: a picture detected fully
3130    /// inside a table region must survive the containment drop (upstream now
3131    /// attaches it to the table's cell; we keep it as a body sibling — either
3132    /// way it must not vanish). The text region inside the same table is the
3133    /// control: regulars are the ones the drop swallows.
3134    #[test]
3135    fn picture_inside_a_table_region_survives_the_containment_drop() {
3136        let mut regions = vec![
3137            region("table", 0.9, 0.0, 0.0, 200.0, 200.0),
3138            region("picture", 0.9, 20.0, 20.0, 120.0, 120.0),
3139            region("text", 0.9, 20.0, 140.0, 180.0, 180.0),
3140        ];
3141        super::drop_contained_regulars(&mut regions);
3142        let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
3143        assert_eq!(
3144            labels,
3145            ["table", "picture"],
3146            "the in-table picture stays; the in-table regular is the special's child"
3147        );
3148    }
3149
3150    /// Table–caption pairing (#265) is reading-order adjacency, docling's
3151    /// `_find_to_captions`: a caption binds the table directly next to it in
3152    /// the region sequence — above-caption and below-caption both work, and
3153    /// geometry is irrelevant (a same-page caption in the other column of a
3154    /// two-column layout is *not* adjacent, however close its box is). A
3155    /// caption with media on both sides, or separated from the table by a
3156    /// text paragraph, stays unattached.
3157    #[test]
3158    fn table_captions_pair_by_reading_order_adjacency() {
3159        // caption → table (above-caption), then table → caption (below-caption),
3160        // then a caption fenced off by a paragraph, then one between two tables.
3161        let regions = vec![
3162            region("text", 0.9, 0.0, 0.0, 100.0, 10.0), // 0 body text
3163            region("caption", 0.9, 0.0, 12.0, 60.0, 20.0), // 1 above-caption
3164            region("table", 0.9, 20.0, 22.0, 90.0, 60.0), // 2 ← pairs with 1
3165            region("table", 0.9, 0.0, 70.0, 100.0, 110.0), // 3 ← pairs with 4
3166            region("caption", 0.9, 0.0, 112.0, 60.0, 120.0), // 4 below-caption
3167            region("text", 0.9, 0.0, 130.0, 100.0, 140.0), // 5 body text
3168            region("caption", 0.9, 0.0, 142.0, 60.0, 150.0), // 6 fenced by 5/7
3169            region("text", 0.9, 0.0, 152.0, 100.0, 162.0), // 7 body text
3170            region("table", 0.9, 0.0, 170.0, 100.0, 200.0), // 8 unpaired
3171            region("caption", 0.9, 0.0, 202.0, 60.0, 210.0), // 9 ambiguous
3172            region("table", 0.9, 0.0, 212.0, 100.0, 240.0), // 10 unpaired
3173        ];
3174        let mut taken = vec![false; regions.len()];
3175        let pairs = super::pair_table_captions(&regions, &mut taken);
3176        assert_eq!(pairs[2], Some(1), "caption directly above its table pairs");
3177        assert_eq!(pairs[3], Some(4), "caption directly below its table pairs");
3178        assert_eq!(
3179            pairs[8], None,
3180            "a text paragraph between caption and table breaks the bond"
3181        );
3182        assert_eq!(
3183            pairs[10], None,
3184            "a caption between two tables is ambiguous and stays loose"
3185        );
3186        assert!(taken[1] && taken[4] && !taken[6] && !taken[9]);
3187    }
3188
3189    /// A colored terms-and-conditions panel detected as `picture` demotes into
3190    /// per-paragraph `text` regions (the blank line between C.7 and C.8 splits
3191    /// them); a chart whose only text is a few narrow axis labels keeps its
3192    /// crop untouched.
3193    #[test]
3194    fn text_panels_demote_to_paragraphs_but_charts_keep_their_crop() {
3195        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3196            text: text.to_string(),
3197            l,
3198            t,
3199            r,
3200            b,
3201        };
3202        let panel = Region {
3203            label: "picture",
3204            score: 0.9,
3205            l: 0.0,
3206            t: 0.0,
3207            r: 100.0,
3208            b: 100.0,
3209        };
3210        // Three tight lines, a blank-line gap, two more: two paragraphs.
3211        let cells = vec![
3212            cell(
3213                "C.7. Wenn Sie diesen Vertrag widerrufen,",
3214                5.0,
3215                10.0,
3216                95.0,
3217                18.0,
3218            ),
3219            cell(
3220                "haben wir Ihnen alle Zahlungen, die wir",
3221                5.0,
3222                20.0,
3223                95.0,
3224                28.0,
3225            ),
3226            cell(
3227                "von Ihnen erhalten haben, zurückzuzahlen.",
3228                5.0,
3229                30.0,
3230                90.0,
3231                38.0,
3232            ),
3233            cell(
3234                "C.8. Wir können die Rückzahlung verweigern,",
3235                5.0,
3236                52.0,
3237                95.0,
3238                60.0,
3239            ),
3240            cell(
3241                "bis wir die Waren wieder zurückerhalten haben.",
3242                5.0,
3243                62.0,
3244                92.0,
3245                70.0,
3246            ),
3247        ];
3248        let mut regions = vec![panel.clone()];
3249        super::recover_text_panels(&mut regions, &cells);
3250        assert_eq!(
3251            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3252            ["text", "text"],
3253            "dense panel must demote into one text region per paragraph"
3254        );
3255        assert!(regions[0].b < regions[1].t, "paragraphs split at the gap");
3256        // Sparse narrow labels (a chart): picture survives.
3257        let labels = vec![
3258            cell("0", 5.0, 90.0, 8.0, 95.0),
3259            cell("50", 5.0, 50.0, 10.0, 55.0),
3260            cell("100", 5.0, 10.0, 12.0, 15.0),
3261            cell("t, s", 45.0, 96.0, 55.0, 100.0),
3262        ];
3263        let mut regions = vec![panel];
3264        super::recover_text_panels(&mut regions, &labels);
3265        assert_eq!(
3266            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3267            ["picture"]
3268        );
3269    }
3270
3271    /// An uncaptioned chart on a scanned page whose title, axis labels, and
3272    /// OCR boxes over the plot area are dense and wide enough to pass the
3273    /// coverage/width gates still keeps its crop: its line heights are ragged
3274    /// (title face vs tick labels vs bar-area OCR), failing the uniform-leading
3275    /// gate — a real text panel is set with constant leading (#173).
3276    #[test]
3277    fn dense_titled_chart_keeps_its_crop() {
3278        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3279            text: text.to_string(),
3280            l,
3281            t,
3282            r,
3283            b,
3284        };
3285        let chart = Region {
3286            label: "picture",
3287            score: 0.9,
3288            l: 0.0,
3289            t: 0.0,
3290            r: 100.0,
3291            b: 100.0,
3292        };
3293        // Five wide lines at wildly different heights: a 12-pt title, 20-pt OCR
3294        // boxes over the bars, 4–5-pt tick/axis labels. Coverage and median
3295        // width both clear the panel thresholds.
3296        let cells = vec![
3297            cell("Underground Water Storage", 10.0, 5.0, 90.0, 17.0),
3298            cell("aquifer recharge zone", 15.0, 30.0, 75.0, 50.0),
3299            cell("confined | unconfined | perched", 12.0, 55.0, 80.0, 59.0),
3300            cell("saturated thickness", 8.0, 70.0, 60.0, 90.0),
3301            cell("distance from well, km", 20.0, 92.0, 85.0, 97.0),
3302        ];
3303        let mut regions = vec![chart];
3304        super::recover_text_panels(&mut regions, &cells);
3305        assert_eq!(
3306            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3307            ["picture"],
3308            "ragged line heights mark a figure, not a text panel"
3309        );
3310    }
3311
3312    /// docling serializes a cluster's cells in docling-parse index order
3313    /// (`_sort_cells`) and joins them with `PageAssembleModel.sanitize_text`:
3314    /// a space after every line except one ending in `-`, which either fuses a
3315    /// wrapped word (alnum on both sides — dash dropped) or glues verbatim (a
3316    /// bare `-` cell: `[0000` `-` `0002` → `[0000 -0002`, the 2305 ORCID line;
3317    /// `-` + `"C" cell -` + `a new table cell` → `-"C" cell a new table cell`,
3318    /// its OTSL list). Verified against the corpus: pure index order beats any
3319    /// geometric re-sort (normal_4pages' heading numerals paint after their
3320    /// text and belong last: `## 들어가며 1`).
3321    #[test]
3322    fn cells_join_in_index_order_with_sanitize_text_rules() {
3323        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3324            text: text.to_string(),
3325            l,
3326            t,
3327            r,
3328            b,
3329        };
3330        let region = Region {
3331            label: "text",
3332            score: 1.0,
3333            l: 0.0,
3334            t: 95.0,
3335            r: 200.0,
3336            b: 130.0,
3337        };
3338        // ORCID superscript: a bare dash cell is a *detached* dash — kept, and
3339        // since docling#4052 (2.122) it joins with the ordinary space on both
3340        // sides (`[0000 -0002 -6960]` before that fix).
3341        let orcid = vec![
3342            cell("[0000", 10.0, 100.0, 30.0, 110.0),
3343            cell("−", 30.0, 100.0, 34.0, 110.0),
3344            cell("0002", 34.0, 100.0, 50.0, 110.0),
3345            cell("−", 50.0, 100.0, 54.0, 110.0),
3346            cell("6960]", 54.0, 100.0, 70.0, 110.0),
3347        ];
3348        assert_eq!(super::region_text(&region, &orcid), "[0000 - 0002 - 6960]");
3349        // Wrapped word: dash dropped, lines fused (both boundary words alnum).
3350        let wrapped = vec![
3351            cell("platforms-", 10.0, 100.0, 60.0, 110.0),
3352            cell("reflects the design", 10.0, 112.0, 90.0, 122.0),
3353        ];
3354        assert_eq!(
3355            super::region_text(&region, &wrapped),
3356            "platformsreflects the design"
3357        );
3358        // Dash-ending lines that are *detached* dashes (a bare bullet cell, a
3359        // `cell -` separator): the dash stays and the lines join with a space
3360        // — docling#4052; before it they glued (`-"C" cell a new table cell`,
3361        // 2305's OTSL list bullets).
3362        let otsl = vec![
3363            cell("–", 10.0, 100.0, 14.0, 110.0),
3364            cell("\"C\" cell -", 16.0, 100.0, 60.0, 110.0),
3365            cell("a new table cell", 10.0, 112.0, 80.0, 122.0),
3366        ];
3367        assert_eq!(
3368            super::region_text(&region, &otsl),
3369            "- \"C\" cell - a new table cell"
3370        );
3371        // Index order is authoritative — no geometric re-sort.
3372        let numeral = vec![
3373            cell("들어가며", 30.0, 100.0, 80.0, 110.0),
3374            cell("1", 10.0, 98.0, 25.0, 112.0), // big numeral painted last
3375        ];
3376        assert_eq!(super::region_text(&region, &numeral), "들어가며 1");
3377    }
3378
3379    /// The geometric-reliability gate, on the two shapes it has to tell apart.
3380    #[test]
3381    fn geometric_reliability_rejects_split_column_grids() {
3382        let g = |rows: &[&[&str]]| -> Vec<Vec<String>> {
3383            rows.iter()
3384                .map(|r| r.iter().map(|c| c.to_string()).collect())
3385                .collect()
3386        };
3387        // A genuine grid: dense, every column carrying entries. Nothing for
3388        // TableFormer to improve, so geometry is used as-is.
3389        assert!(super::geometric_table_is_reliable(&g(&[
3390            &["Datum", "Leistung", "Anzahl", "Kosten"],
3391            &["04.07", "Internet", "1", "40.30"],
3392            &["04.07", "Telefon", "2", "8.06"],
3393        ])));
3394        // The left-edge split artefact (the shape a scanned invoice produced):
3395        // one real label column plus values scattered across three sparse ones.
3396        assert!(!super::geometric_table_is_reliable(&g(&[
3397            &["www.magenta.at/faq", "", "", ""],
3398            &["Serviceteam", "", "", ""],
3399            &["Telefon", "0676/2000", "", ""],
3400            &["Kundennummer", "", "", "1.21699482"],
3401            &["Rechnungsnummer", "", "922769430725", ""],
3402            &["Rechnungsdatum", "", "", "04.07.2025"],
3403        ])));
3404        // A column only one row ever uses is a split artefact even when the
3405        // grid is otherwise dense.
3406        assert!(!super::geometric_table_is_reliable(&g(&[
3407            &["a", "b", ""],
3408            &["c", "d", ""],
3409            &["e", "f", "g"],
3410        ])));
3411        // Degenerate shapes are never vouched for — TableFormer may recover
3412        // structure a collapsed reconstruction lost.
3413        assert!(!super::geometric_table_is_reliable(&g(&[&[
3414            "only one column"
3415        ]])));
3416        assert!(!super::geometric_table_is_reliable(&[]));
3417    }
3418
3419    /// A `picture` region is cropped out of the rendered page, whatever built
3420    /// that page. The browser pipeline (#157) has no pdfium but does hand over
3421    /// the rasterized bitmap through `from_cells_with_image`, so it must get
3422    /// the same figure bytes the native path does — that is what makes
3423    /// `images = "embedded"` inline real pixels instead of a placeholder.
3424    #[cfg(feature = "ocr-prep")]
3425    #[test]
3426    fn picture_regions_are_cropped_from_a_host_supplied_page_image() {
3427        let mut img = image::RgbImage::new(200, 200);
3428        // Paint the figure area so the crop is distinguishable from the page.
3429        for y in 100..160 {
3430            for x in 20..120 {
3431                img.put_pixel(x, y, image::Rgb([255, 0, 0]));
3432            }
3433        }
3434        // scale 2.0: the region is in page points, the bitmap in pixels.
3435        let page = PdfPage::from_cells_with_image(100.0, 100.0, 2.0, Vec::new(), img);
3436        let region = Region {
3437            label: "picture",
3438            score: 0.9,
3439            l: 10.0,
3440            t: 50.0,
3441            r: 60.0,
3442            b: 80.0,
3443        };
3444        let (nodes, _) = super::assemble_page(&page, vec![region], &[None], &[None]);
3445        // Layout-derived nodes carry provenance, so the picture arrives wrapped.
3446        let image = nodes
3447            .iter()
3448            .find_map(|n| match n {
3449                Node::Located { inner, .. } => match &**inner {
3450                    Node::Picture { image, .. } => image.as_ref(),
3451                    _ => None,
3452                },
3453                Node::Picture { image, .. } => image.as_ref(),
3454                _ => None,
3455            })
3456            .expect("a picture node with cropped pixels");
3457        assert_eq!(image.mimetype, "image/png");
3458        assert_eq!((image.width, image.height), (100, 60), "region × scale");
3459        assert!(!image.data.is_empty(), "PNG bytes were encoded");
3460    }
3461
3462    #[test]
3463    fn link_anchors_split_a_shared_word_cell_between_adjacent_links() {
3464        // A common header layout: one text run holds several pipe-separated
3465        // labels, each carrying its own link annotation. Every link must get
3466        // its own label as the anchor (and the "|" separators must belong to
3467        // none), not the whole run.
3468        let annot = |l: f32, r: f32, uri: &str| LinkAnnot {
3469            l,
3470            t: 100.0,
3471            r,
3472            b: 114.0,
3473            uri: uri.into(),
3474        };
3475        let page = PdfPage {
3476            width: 600.0,
3477            height: 800.0,
3478            scale: 2.0,
3479            cells: Vec::new(),
3480            code_cells: Vec::new(),
3481            // "LinkedIn | GitHub | Credly" = 26 chars over x 100..360.
3482            word_cells: vec![cell(
3483                "LinkedIn | GitHub | Credly",
3484                100.0,
3485                100.0,
3486                360.0,
3487                114.0,
3488            )],
3489            image: image::RgbImage::new(1, 1),
3490            image_layout: None,
3491            links: vec![
3492                annot(100.0, 180.0, "https://l"),
3493                annot(200.0, 260.0, "https://g"),
3494                annot(290.0, 360.0, "https://c"),
3495            ],
3496            rotation: 0,
3497        };
3498        assert_eq!(
3499            resolve_link_anchors(&page),
3500            vec![
3501                ("LinkedIn".to_string(), "https://l".to_string()),
3502                ("GitHub".to_string(), "https://g".to_string()),
3503                ("Credly".to_string(), "https://c".to_string()),
3504            ]
3505        );
3506    }
3507
3508    /// A one-line code cell at `[l, r] × [t, b]` (top-left coords).
3509    fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
3510        TextCell {
3511            text: text.into(),
3512            l,
3513            t,
3514            r,
3515            b,
3516        }
3517    }
3518
3519    fn region(label: &'static str, score: f32, l: f32, t: f32, r: f32, b: f32) -> Region {
3520        Region {
3521            label,
3522            score,
3523            l,
3524            t,
3525            r,
3526            b,
3527        }
3528    }
3529
3530    #[test]
3531    fn resolve_collapses_nested_code_keeping_the_larger_box() {
3532        // A tight high-score `code` box and a taller lower-score near-duplicate that
3533        // contains it must collapse to one — the *larger* box, so every cell stays
3534        // covered and nothing leaks out as orphan text.
3535        let tight = region("code", 0.95, 78.0, 292.0, 300.0, 330.0);
3536        let wide = region("code", 0.66, 63.0, 260.0, 320.0, 346.0);
3537        let kept = super::resolve(vec![tight, wide]);
3538        assert_eq!(kept.len(), 1, "nested code boxes must collapse to one");
3539        assert!(
3540            kept[0].l == 63.0 && kept[0].b == 346.0,
3541            "the larger containing box is kept"
3542        );
3543    }
3544
3545    #[test]
3546    fn resolve_keeps_distinct_and_differently_typed_regions() {
3547        // A text box fully inside a lower-score *table* must NOT be collapsed (the
3548        // code dedup is code-only), and two separate code blocks stay separate.
3549        let text = region("text", 0.95, 90.0, 210.0, 200.0, 230.0);
3550        let table = region("table", 0.60, 80.0, 200.0, 400.0, 500.0);
3551        assert_eq!(super::resolve(vec![text, table]).len(), 2);
3552
3553        let code_a = region("code", 0.9, 78.0, 100.0, 300.0, 140.0);
3554        let code_b = region("code", 0.9, 78.0, 300.0, 300.0, 360.0); // far below, no overlap
3555        assert_eq!(super::resolve(vec![code_a, code_b]).len(), 2);
3556    }
3557
3558    #[test]
3559    fn code_language_label_above_code_is_detected() {
3560        // A bare "XML" token directly above a code box is a language label; a real
3561        // heading above the same code is not; a language word with no code below is
3562        // left alone.
3563        let label = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
3564        let code = region("code", 0.7, 77.0, 552.0, 290.0, 640.0);
3565        let heading = region("section_header", 0.9, 76.0, 500.0, 260.0, 512.0);
3566        let cells = vec![
3567            cell("XML", 78.0, 541.0, 94.0, 548.0),       // inside `label`
3568            cell("Overview", 78.0, 501.0, 250.0, 511.0), // inside `heading`
3569        ];
3570        let drop = super::code_language_labels(&[label, code, heading], &cells);
3571        assert_eq!(drop, vec![true, false, false], "only the label is consumed");
3572
3573        // Same label with no code region present → not consumed.
3574        let label2 = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
3575        let only = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
3576        assert_eq!(super::code_language_labels(&[label2], &only), vec![false]);
3577
3578        // A label swallowed into the top of a wider code box (negative gap) is still
3579        // recognized.
3580        let inside_lbl = region("text", 0.9, 76.0, 540.0, 96.0, 549.0);
3581        let wide_code = region("code", 0.7, 63.0, 531.0, 320.0, 654.0);
3582        let cells2 = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
3583        assert_eq!(
3584            super::code_language_labels(&[inside_lbl, wide_code], &cells2),
3585            vec![true, false]
3586        );
3587
3588        assert!(super::is_code_language("XML") && super::is_code_language("c#"));
3589        assert!(!super::is_code_language("Configure") && !super::is_code_language("XML schema"));
3590    }
3591
3592    #[test]
3593    fn code_region_text_keeps_lines_and_indentation() {
3594        // Three source lines; each glyph is 6 units wide (width / chars = 6), so the
3595        // `int X;` line indented to x=22 is (22-10)/6 = 2 spaces in.
3596        let region = Region {
3597            label: "code",
3598            score: 1.0,
3599            l: 0.0,
3600            t: -5.0,
3601            r: 100.0,
3602            b: 40.0,
3603        };
3604        let cells = vec![
3605            cell("struct P {", 10.0, 0.0, 70.0, 10.0),
3606            cell("int X;", 22.0, 12.0, 58.0, 22.0),
3607            cell("}", 10.0, 24.0, 16.0, 34.0),
3608        ];
3609        assert_eq!(code_region_text(&region, &cells), "struct P {\n  int X;\n}");
3610    }
3611
3612    #[test]
3613    fn code_region_text_tightens_punctuation_without_eating_indentation() {
3614        // A fluent `.Foo()` line at x=22 (2 chars in). Per-line tightening must not
3615        // consume the leading indent space by matching " ." across it.
3616        let region = Region {
3617            label: "code",
3618            score: 1.0,
3619            l: 0.0,
3620            t: -5.0,
3621            r: 100.0,
3622            b: 40.0,
3623        };
3624        let cells = vec![
3625            cell("builder", 10.0, 0.0, 52.0, 10.0),
3626            // pdfium spaced the call: ".Foo (x)" tightens to ".Foo(x)", still 2-indented.
3627            cell(".Foo (x)", 22.0, 12.0, 70.0, 22.0),
3628        ];
3629        assert_eq!(code_region_text(&region, &cells), "builder\n  .Foo(x)");
3630    }
3631
3632    #[test]
3633    fn code_region_text_orders_out_of_order_cells_and_ignores_blank_lines() {
3634        let region = Region {
3635            label: "code",
3636            score: 1.0,
3637            l: 0.0,
3638            t: -5.0,
3639            r: 100.0,
3640            b: 60.0,
3641        };
3642        // Fed bottom-up and with a whitespace-only cell; output is top-down, no blank.
3643        let cells = vec![
3644            cell("b();", 10.0, 24.0, 34.0, 34.0),
3645            cell("   ", 10.0, 12.0, 20.0, 22.0),
3646            cell("a();", 10.0, 0.0, 34.0, 10.0),
3647        ];
3648        assert_eq!(code_region_text(&region, &cells), "a();\nb();");
3649        // No code cells → empty, so the caller falls back to the prose text.
3650        assert_eq!(code_region_text(&region, &[]), "");
3651    }
3652
3653    fn para(text: &str) -> Node {
3654        Node::Paragraph { text: text.into() }
3655    }
3656
3657    /// Run a node sequence through [`StreamAssembler`] with the given page splits
3658    /// and assert the flushed result equals one-shot [`merge_continuations`].
3659    fn assert_stream_eq(nodes: &[Node], splits: &[usize]) {
3660        let mut want = nodes.to_vec();
3661        merge_continuations(&mut want);
3662
3663        let mut asm = StreamAssembler::new();
3664        let mut got = Vec::new();
3665        let mut start = 0;
3666        for &end in splits {
3667            got.extend(asm.push(nodes[start..end].to_vec()));
3668            start = end;
3669        }
3670        got.extend(asm.push(nodes[start..].to_vec()));
3671        got.extend(asm.finish());
3672        assert_eq!(got, want, "stream assembly diverged (splits={splits:?})");
3673    }
3674
3675    #[test]
3676    fn stream_assembler_matches_merge_continuations() {
3677        // Open fragment + lowercase continuation split across a page boundary.
3678        let cross = [para("the definition of"), para("lists in scope")];
3679        assert_stream_eq(&cross, &[1]);
3680        assert_stream_eq(&cross, &[]);
3681
3682        // Continuation that wraps around a figure (+ its caption) on the boundary.
3683        let wrap = [
3684            para("the wing type that is"),
3685            Node::Picture {
3686                caption: None,
3687                caption_href: None,
3688                image: None,
3689                classification: None,
3690                caption_parent: Default::default(),
3691            },
3692            para("Fig. 1. a diagram"),
3693            para("the most common kind"),
3694        ];
3695        for splits in [&[][..], &[1][..], &[2][..], &[3][..], &[1, 3][..]] {
3696            assert_stream_eq(&wrap, splits);
3697        }
3698
3699        // A heading between fragments blocks the merge (must still flush correctly).
3700        let blocked = [
3701            para("ends mid word and"),
3702            Node::Heading {
3703                level: 2,
3704                text: "New Section".into(),
3705            },
3706            para("more body here"),
3707        ];
3708        for splits in [&[][..], &[1][..], &[2][..]] {
3709            assert_stream_eq(&blocked, splits);
3710        }
3711
3712        // A chain across three pages: each page is one open lowercase fragment.
3713        let chain = [
3714            para("alpha beta"),
3715            para("gamma delta"),
3716            para("epsilon zeta"),
3717        ];
3718        assert_stream_eq(&chain, &[1, 2]);
3719    }
3720
3721    #[test]
3722    fn clean_text_dehyphenates_and_normalizes_typography() {
3723        // U+0002 line-wrap hyphen + the join space → merged word (like docling).
3724        assert_eq!(clean_text("com\u{2} pact"), "compact");
3725        assert_eq!(clean_text("end-to\u{2} end deep"), "end-toend deep");
3726        // A stray wrap hyphen (no following join) is dropped.
3727        assert_eq!(clean_text("word\u{2}"), "word");
3728        // Typographic punctuation → ASCII: every curly quote becomes `'`
3729        // (docling-parse's sanitizer table), a literal `"` stays.
3730        assert_eq!(
3731            clean_text("Graph\u{2019}s \u{201c}x\u{201d} \"y\""),
3732            "Graph's 'x' \"y\""
3733        );
3734        assert_eq!(clean_text("a\u{2026}"), "a...");
3735        // The dp default (the docling-parse sanitizer) preserves internal spacing
3736        // it placed deliberately; line breaks/tabs normalize to a space, ends trim.
3737        assert_eq!(clean_text("a   b\nc"), "a   b c");
3738    }
3739
3740    /// docling#4064: a form's children are emitted together where the form
3741    /// sits in the top-level order, not interleaved with surrounding text.
3742    #[test]
3743    fn form_children_stay_together_in_reading_order() {
3744        let reg = |label: &'static str, l: f32, t: f32, r: f32, b: f32| Region {
3745            label,
3746            score: 0.9,
3747            l,
3748            t,
3749            r,
3750            b,
3751        };
3752        // Page: intro text, then a form spanning the left column with two
3753        // fields and a table inside, while a right-column paragraph sits
3754        // level with the form's first field (it would otherwise be read
3755        // between the form's children).
3756        let mut items = vec![
3757            reg("text", 50.0, 50.0, 550.0, 70.0),    // 0 intro
3758            reg("form", 50.0, 100.0, 300.0, 400.0),  // 1 container
3759            reg("text", 60.0, 110.0, 290.0, 130.0),  // 2 field A (child)
3760            reg("text", 320.0, 110.0, 550.0, 130.0), // 3 right column paragraph
3761            reg("table", 60.0, 150.0, 290.0, 300.0), // 4 table (child)
3762            reg("text", 60.0, 320.0, 290.0, 340.0),  // 5 field B (child)
3763            reg("text", 50.0, 450.0, 550.0, 470.0),  // 6 outro
3764        ];
3765        let cids = super::cluster_cids(&items, &[]);
3766        super::order_with_containers(&mut items, &cids, 600.0, 800.0, |r| r);
3767        let order: Vec<(&str, f32)> = items.iter().map(|r| (r.label, r.t)).collect();
3768        // The form block (container, then its children top-down) is one unit.
3769        let form_pos = order.iter().position(|(l, _)| *l == "form").unwrap();
3770        assert_eq!(
3771            &order[form_pos..form_pos + 4],
3772            &[
3773                ("form", 100.0),
3774                ("text", 110.0),
3775                ("table", 150.0),
3776                ("text", 320.0)
3777            ]
3778        );
3779        assert_eq!(order[0], ("text", 50.0));
3780        assert_eq!(order[order.len() - 1], ("text", 450.0));
3781        // Without a container the plain order interleaves by geometry.
3782        let mut flat: Vec<Region> = items
3783            .iter()
3784            .filter(|r| r.label != "form")
3785            .cloned()
3786            .collect();
3787        let cids = super::cluster_cids(&flat, &[]);
3788        super::order_regions(&mut flat, &cids, 600.0, 800.0, |r| r);
3789        assert_ne!(
3790            flat.iter().map(|r| r.t).collect::<Vec<_>>(),
3791            order
3792                .iter()
3793                .filter(|(l, _)| *l != "form")
3794                .map(|(_, t)| *t)
3795                .collect::<Vec<_>>()
3796        );
3797    }
3798
3799    /// docling#3906: a picture inside a table lands in the covering cell,
3800    /// chosen by the picture's inferred grid position when cell boxes overlap.
3801    #[test]
3802    fn picture_matches_the_cell_at_its_grid_position() {
3803        let cell = |r: usize, c: usize, bbox: [f32; 4]| docling_core::TableCell {
3804            text: format!("r{r}c{c}"),
3805            bbox: Some(bbox),
3806            start_row: r,
3807            start_col: c,
3808            row_span: 1,
3809            col_span: 1,
3810            column_header: false,
3811            row_header: false,
3812            row_section: false,
3813        };
3814        // 2×2 grid; the (1,0) cell box is generous and also covers the picture.
3815        let cells = vec![
3816            cell(0, 0, [0.0, 0.0, 100.0, 50.0]),
3817            cell(0, 1, [100.0, 0.0, 200.0, 50.0]),
3818            cell(1, 0, [0.0, 50.0, 100.0, 100.0]),
3819            cell(1, 1, [100.0, 50.0, 200.0, 100.0]),
3820        ];
3821        let pic = Region {
3822            label: "picture",
3823            score: 0.9,
3824            l: 110.0,
3825            t: 60.0,
3826            r: 190.0,
3827            b: 95.0,
3828        };
3829        assert_eq!(super::match_picture_to_cell(&pic, &cells), Some((1.0, 3)));
3830        // A picture only half inside any cell is not nested.
3831        let straddling = Region {
3832            label: "picture",
3833            score: 0.9,
3834            l: 60.0,
3835            t: 60.0,
3836            r: 160.0,
3837            b: 95.0,
3838        };
3839        assert_eq!(super::match_picture_to_cell(&straddling, &cells), None);
3840    }
3841
3842    /// docling#4052 (2.122): a line-final dash fuses the wrapped word only
3843    /// when attached to it; a detached dash is a literal and the lines join
3844    /// with a space.
3845    #[test]
3846    fn line_final_hyphen_fuses_only_when_attached_to_a_word() {
3847        let line = |text: &str, t: f32| TextCell {
3848            text: text.to_string(),
3849            l: 0.0,
3850            t,
3851            r: 100.0,
3852            b: t + 10.0,
3853        };
3854        // `algo-` / `rithms`: attached hyphen, alnum on both sides → fused.
3855        assert_eq!(
3856            cells_text(vec![&line("algo-", 0.0), &line("rithms", 12.0)]),
3857            "algorithms"
3858        );
3859        // `pp. 545-` / `561`: attached, digits count as alnum → `545561` (upstream).
3860        assert_eq!(
3861            cells_text(vec![&line("pp. 545-", 0.0), &line("561", 12.0)]),
3862            "pp. 545561"
3863        );
3864        // A dash after whitespace — a separator or a lone `-` cell — is kept and
3865        // the lines take the ordinary joining space.
3866        assert_eq!(
3867            cells_text(vec![&line("range -", 0.0), &line("wide", 12.0)]),
3868            "range - wide"
3869        );
3870        assert_eq!(
3871            cells_text(vec![&line("-", 0.0), &line("item", 12.0)]),
3872            "- item"
3873        );
3874        // Attached but the next line opens with no word (`x-` / `...`): dash
3875        // kept and, as before, no separating space.
3876        assert_eq!(
3877            cells_text(vec![&line("x-", 0.0), &line("...", 12.0)]),
3878            "x-..."
3879        );
3880    }
3881
3882    #[test]
3883    fn lam_alef_only_swaps_a_genuinely_reversed_ligature() {
3884        // A mid-word `alef-variant + lam` is pdfium's reversed lam-alef ligature and
3885        // is swapped back to logical `lam + alef-variant` (`ب أ ل` → `ب ل أ`).
3886        assert_eq!(
3887            clean_text("\u{0628}\u{0623}\u{0644}"),
3888            "\u{0628}\u{0644}\u{0623}"
3889        );
3890        // But when the alef-variant is *already* preceded by a lam it is the logical
3891        // ligature `لآ`; the following lam is the next syllable's letter and must not
3892        // move. `التعلم الآلي` must stay `الآلي`, not become `اللآي`.
3893        assert_eq!(
3894            clean_text("\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"),
3895            "\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"
3896        );
3897    }
3898
3899    /// The #419 page, in points: three layout boxes over one paragraph, two of
3900    /// them ending partway through a line. The sliced lines miss the 0.2 claim
3901    /// and become orphans; the third model box starts above the second orphan,
3902    /// so unfitted the reading order emits that box first and strands the line.
3903    fn sliced_paragraph() -> (Vec<Region>, Vec<TextCell>) {
3904        let line = |text: &str, t: f32, r: f32| cell(text, 60.0, t, r, t + 11.0);
3905        let cells = vec![
3906            line("The mission of this series is to improve", 135.0, 458.0),
3907            line("The books in this series are technical,", 147.0, 458.0),
3908            line("substantial. The authors are", 159.0, 458.0),
3909            line("highly experienced craftsmen and", 171.5, 458.0), // sliced: 1.5/11 under box A
3910            line("actually works in practice, as opposed", 185.0, 458.0),
3911            line("about what the author has done, not", 197.0, 458.0),
3912            line("about programming, there will be lots", 210.5, 458.0), // sliced: 1.5/11 under box B
3913            line("will be lots of case studies from real", 223.0, 206.0), // C's line
3914        ];
3915        let regions = vec![
3916            region("text", 0.9, 60.0, 132.0, 458.0, 173.0), // A: three lines + a sliver of the 4th
3917            region("text", 0.9, 60.0, 184.0, 458.0, 212.0), // B: two lines + a sliver of the 7th
3918            region("text", 0.9, 60.0, 216.0, 206.0, 227.0), // C: last line, box opening 5.5pt too early
3919        ];
3920        (regions, cells)
3921    }
3922
3923    fn ordered_texts(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
3924        let mut items: Vec<Region> = regions.to_vec();
3925        let cids = super::cluster_cids(&items, cells);
3926        super::order_regions(&mut items, &cids, 500.0, 700.0, |r| r);
3927        super::region_texts_exclusive(&items, cells)
3928            .into_iter()
3929            .map(|t| t.chars().take(9).collect())
3930            .collect()
3931    }
3932
3933    /// #419: fitted to its cells, a model box that cut a line in half no longer
3934    /// overlaps the orphan that line became, so the orphan orders where it
3935    /// reads; unfitted, the same page strands the line after the paragraph.
3936    #[test]
3937    fn fitting_boxes_to_cells_puts_a_sliced_line_back_in_order() {
3938        let (mut regions, cells) = sliced_paragraph();
3939        super::add_orphan_regions(&mut regions, &cells);
3940        assert_eq!(regions.len(), 5, "two orphan lines");
3941        // The defect, for the record: C (top 216) is not strictly below the
3942        // orphan at 210.5–221.5, so the graph orders C first.
3943        assert_eq!(
3944            ordered_texts(&regions, &cells).last().map(String::as_str),
3945            Some("about pro")
3946        );
3947
3948        super::fit_regions_to_cells(&mut regions, &cells);
3949        assert_eq!(regions.len(), 5);
3950        // A ends on its last claimed line, C starts on its only one.
3951        assert_eq!((regions[0].t, regions[0].b), (135.0, 170.0));
3952        assert_eq!((regions[2].t, regions[2].b), (223.0, 234.0));
3953        assert_eq!(
3954            ordered_texts(&regions, &cells),
3955            [
3956                "The missi",
3957                "highly ex",
3958                "actually ",
3959                "about pro",
3960                "will be l"
3961            ]
3962        );
3963    }
3964
3965    /// An orphan the fitted paragraph box surrounds (a short middle line the
3966    /// narrow model box missed while claiming the lines around it) is folded
3967    /// into the paragraph; an empty regular box goes away, a formula stays, a
3968    /// picture is never refitted, and a page with no cells is left untouched.
3969    #[test]
3970    fn fitting_folds_surrounded_orphans_and_drops_empty_regulars() {
3971        let wide = |text: &str, t: f32| cell(text, 60.0, t, 400.0, t + 11.0);
3972        let cells = vec![
3973            wide("first line of the paragraph", 100.0),
3974            cell("stray", 250.0, 112.0, 400.0, 123.0), // clear of the narrow box
3975            wide("third line of the paragraph", 124.0),
3976        ];
3977        let mut regions = vec![
3978            // Narrow box: claims the wide lines at 0.41, misses the short one.
3979            region("text", 0.9, 60.0, 98.0, 200.0, 136.0),
3980            region("section_header", 0.8, 60.0, 300.0, 200.0, 320.0), // no cells
3981            region("formula", 0.8, 60.0, 340.0, 200.0, 360.0),        // no cells, kept
3982            region("picture", 0.8, 0.0, 400.0, 500.0, 600.0),
3983        ];
3984        super::add_orphan_regions(&mut regions, &cells);
3985        assert_eq!(regions.len(), 5, "the short line became an orphan");
3986        super::fit_regions_to_cells(&mut regions, &cells);
3987        let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
3988        assert_eq!(labels, ["text", "formula", "picture"]);
3989        let para = &regions[0];
3990        assert_eq!(
3991            (para.l, para.t, para.r, para.b),
3992            (60.0, 100.0, 400.0, 135.0)
3993        );
3994        assert_eq!(
3995            super::region_texts_exclusive(&regions, &cells)[0],
3996            "first line of the paragraph stray third line of the paragraph"
3997        );
3998        assert_eq!(
3999            (regions[2].t, regions[2].b),
4000            (400.0, 600.0),
4001            "picture untouched"
4002        );
4003
4004        let mut untouched = vec![region("text", 0.9, 0.0, 0.0, 10.0, 10.0)];
4005        super::fit_regions_to_cells(&mut untouched, &[]);
4006        assert_eq!(untouched.len(), 1, "no cells yet: nothing dropped");
4007    }
4008}