Skip to main content

docling_pdf/
assemble.rs

1//! Layout-driven assembly: map detected [`Region`]s + text cells to a
2//! [`DoclingDocument`], mirroring docling's page-assembly + reading-order.
3//!
4//! Overlapping detections are resolved greedily by score, each text cell is
5//! assigned to its best-containing region, regions are ordered in reading order
6//! (two-column aware), and each becomes a typed node by its layout label.
7
8use docling_core::{CaptionParent, Node, PictureClass, PictureImage, Table};
9#[cfg(any(feature = "ml", feature = "ocr-prep"))]
10use image::RgbImage;
11
12use crate::layout::Region;
13use crate::pdfium_backend::{PdfPage, TextCell};
14
15fn area(l: f32, t: f32, r: f32, b: f32) -> f32 {
16    ((r - l).max(0.0)) * ((b - t).max(0.0))
17}
18
19/// Intersection area of two boxes.
20fn inter(a: &Region, l: f32, t: f32, r: f32, b: f32) -> f32 {
21    let il = a.l.max(l);
22    let it = a.t.max(t);
23    let ir = a.r.min(r);
24    let ib = a.b.min(b);
25    area(il, it, ir, ib)
26}
27
28/// Wrapper (structured-region) labels, ported from docling
29/// `LayoutPostprocessor.WRAPPER_TYPES`: a region that *contains* other regions
30/// and renders as a structured block (a table / table-of-contents index), not as
31/// its own flat text.
32fn is_wrapper(label: &str) -> bool {
33    matches!(
34        label,
35        "table" | "document_index" | "form" | "key_value_region"
36    )
37}
38
39/// Labels docling's table-structure (TableFormer) model runs on and that render
40/// as a Markdown table: a plain `table` and a `document_index` (a table of
41/// contents), which docling assembles as a `TableItem` too.
42pub fn is_table_like(label: &str) -> bool {
43    matches!(label, "table" | "document_index")
44}
45
46/// Greedily keep regions by descending score, dropping a region that is mostly
47/// covered by an already-kept one (RT-DETR emits overlapping duplicates).
48fn greedy(mut regions: Vec<Region>) -> Vec<Region> {
49    regions.sort_by(|a, b| b.score.total_cmp(&a.score));
50    let mut kept: Vec<Region> = Vec::new();
51    for r in regions {
52        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
53        let covered = kept.iter().any(|k| {
54            let i = inter(&r, k.l, k.t, k.r, k.b);
55            let ka = area(k.l, k.t, k.r, k.b).max(1.0);
56            // drop if most of r is inside k, or they strongly mutually overlap
57            i / ra > 0.7 || i / (ra + ka - i) > 0.5
58        });
59        if !covered {
60            kept.push(r);
61        }
62    }
63    kept
64}
65
66/// Resolve overlapping RT-DETR detections, ported from the bucket structure of
67/// docling's `LayoutPostprocessor`: regular, picture and wrapper clusters live in
68/// **separate** spatial indexes and are de-overlapped independently, so a
69/// high-score picture never suppresses a lower-score table or table-of-contents
70/// index (the redp5110 TOC that was otherwise replaced by a picture box). A
71/// cross-type pass first drops a picture that nearly coincides with a table
72/// (`_handle_cross_type_overlaps`), keeping the structured table.
73/// docling's `_remove_overlapping_clusters("picture")`: same-label picture
74/// detections whose boxes heavily overlap (IoU > 0.8, or either box > 80 %
75/// contained in the other) form one group, and a single survivor is kept per
76/// group. Survivor selection ports `_should_prefer_cluster` /
77/// `_select_best_cluster_from_group` with the picture params
78/// (`area_threshold` 2.0, `conf_threshold` 0.3): a candidate is rejected only
79/// when a rival is both comparable in size (candidate ≤ 2× its area) and
80/// clearly more confident (> 0.3); among the survivors the *larger* box wins
81/// unless it is > 0.3 less confident. Net effect on the corpus: a figure the
82/// detector proposes both whole and as its sub-panels (2206's four-thumbnail
83/// Figure 1) collapses to the whole-figure box, exactly like docling.
84pub(crate) fn dedup_pictures(regions: &mut Vec<Region>) {
85    remove_overlapping_specials(regions, |l| l == "picture", 2.0, 0.3);
86}
87
88/// docling's `_remove_overlapping_clusters` for one special bucket (the
89/// regions whose label satisfies `in_bucket`), with that bucket's
90/// `OVERLAP_PARAMS`: picture (2.0, 0.3) — see [`dedup_pictures`] — or wrapper
91/// (2.0, 0.2) for the table bucket in [`resolve`]. Regions outside the bucket
92/// are untouched; a bucket member overlapping nothing is always kept.
93fn remove_overlapping_specials(
94    regions: &mut Vec<Region>,
95    in_bucket: impl Fn(&str) -> bool,
96    area_threshold: f32,
97    conf_threshold: f32,
98) {
99    let idx: Vec<usize> = (0..regions.len())
100        .filter(|&i| in_bucket(regions[i].label))
101        .collect();
102    if idx.len() < 2 {
103        return;
104    }
105    // Union-find over the bucket.
106    let mut parent: Vec<usize> = (0..idx.len()).collect();
107    fn find(parent: &mut [usize], i: usize) -> usize {
108        let mut root = i;
109        while parent[root] != root {
110            root = parent[root];
111        }
112        let mut cur = i;
113        while parent[cur] != root {
114            let next = parent[cur];
115            parent[cur] = root;
116            cur = next;
117        }
118        root
119    }
120    let boxed = |r: &Region| (r.l, r.t, r.r, r.b);
121    for a in 0..idx.len() {
122        for b in (a + 1)..idx.len() {
123            let (ra, rb) = (&regions[idx[a]], &regions[idx[b]]);
124            let (al, at, ar, ab_) = boxed(ra);
125            let (bl, bt, br, bb) = boxed(rb);
126            let ix = (ar.min(br) - al.max(bl)).max(0.0);
127            let iy = (ab_.min(bb) - at.max(bt)).max(0.0);
128            let inter = ix * iy;
129            let aa = area(al, at, ar, ab_).max(f32::EPSILON);
130            let ba = area(bl, bt, br, bb).max(f32::EPSILON);
131            let iou = inter / (aa + ba - inter).max(f32::EPSILON);
132            if iou > 0.8 || inter / aa > 0.8 || inter / ba > 0.8 {
133                let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
134                if pa != pb {
135                    parent[pa] = pb;
136                }
137            }
138        }
139    }
140    // Per group, run docling's pairwise preference + larger-wins selection.
141    let mut groups: std::collections::HashMap<usize, Vec<usize>> = std::collections::HashMap::new();
142    for i in 0..idx.len() {
143        let root = find(&mut parent, i);
144        groups.entry(root).or_default().push(i);
145    }
146    let mut drop = vec![false; regions.len()];
147    for group in groups.values() {
148        if group.len() < 2 {
149            continue;
150        }
151        let area_of = |i: usize| {
152            let r = &regions[idx[i]];
153            area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
154        };
155        let mut best: Option<usize> = None;
156        for &cand in group {
157            let passes = group.iter().all(|&other| {
158                if other == cand {
159                    return true;
160                }
161                let area_ratio = area_of(cand) / area_of(other);
162                let conf_diff = regions[idx[other]].score - regions[idx[cand]].score;
163                !(area_ratio <= area_threshold && conf_diff > conf_threshold)
164            });
165            if passes {
166                best = Some(match best {
167                    None => cand,
168                    Some(cur) => {
169                        if area_of(cand) > area_of(cur)
170                            && regions[idx[cur]].score - regions[idx[cand]].score <= conf_threshold
171                        {
172                            cand
173                        } else {
174                            cur
175                        }
176                    }
177                });
178            }
179        }
180        // Every candidate rejected can't happen with docling's rule (rejection
181        // needs a strictly better rival); guard with highest score anyway.
182        let keep = best.unwrap_or_else(|| {
183            *group
184                .iter()
185                .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(&regions[idx[b]].score))
186                .expect("non-empty group")
187        });
188        for &i in group {
189            if i != keep {
190                drop[idx[i]] = true;
191            }
192        }
193    }
194    let mut keep_iter = drop.into_iter();
195    regions.retain(|_| !keep_iter.next().expect("aligned"));
196}
197
198/// docling's `_remove_overlapping_clusters("regular")` for what [`greedy`]
199/// leaves standing, run on OCR'd pages before the region-scoped OCR (found
200/// with #471's synthetic scan: every line of the page came out twice).
201///
202/// `greedy` keeps regions by descending score and drops a candidate mostly
203/// inside an already-kept one — so a *lower*-score block that contains
204/// several higher-score line boxes (RT-DETR's favourite reading of a sparse
205/// scanned page: every line, plus the paragraph) survives alongside them.
206/// On a digital page that is harmless: [`fit_regions_to_cells`] hands each
207/// text cell to one owner and drops the regions left empty. On a scanned
208/// page the cells don't exist yet — they come from OCR of *each* region's
209/// crop — so the block and its lines were each recognized, and the block,
210/// which then owned both cell sets, read every line twice.
211///
212/// Upstream never has this problem because its OCR runs over the bitmap
213/// before layout postprocessing and each cell is assigned once; the
214/// postprocessor's regular pass then groups clusters that overlap (IoU > 0.8,
215/// or either > 80 % contained in the other) with a union-find, keeps one
216/// survivor per group via `_should_prefer_cluster` /
217/// `_select_best_cluster_from_group` (`area_threshold` 1.3, `conf_threshold`
218/// 0.05; a LIST_ITEM beats a same-sized TEXT, a CODE box beats what it
219/// contains) and merges the losers' cells into it, whose box is then fitted
220/// to those cells. The equivalent for region-scoped OCR: one survivor per
221/// group, keeping its label and score, with the group's **union** box so the
222/// single crop still covers every merged line. Pictures and wrappers are not
223/// regulars and are left alone ([`dedup_pictures`], [`resolve`]).
224pub(crate) fn merge_overlapping_regulars(regions: &mut Vec<Region>) {
225    let idx: Vec<usize> = (0..regions.len())
226        .filter(|&i| regions[i].label != "picture" && !is_wrapper(regions[i].label))
227        .collect();
228    if idx.len() < 2 {
229        return;
230    }
231    let mut parent: Vec<usize> = (0..idx.len()).collect();
232    fn find(parent: &mut [usize], i: usize) -> usize {
233        let mut root = i;
234        while parent[root] != root {
235            root = parent[root];
236        }
237        let mut cur = i;
238        while parent[cur] != root {
239            let next = parent[cur];
240            parent[cur] = root;
241            cur = next;
242        }
243        root
244    }
245    for a in 0..idx.len() {
246        for b in (a + 1)..idx.len() {
247            let (ra, rb) = (&regions[idx[a]], &regions[idx[b]]);
248            let i = inter(ra, rb.l, rb.t, rb.r, rb.b);
249            let aa = area(ra.l, ra.t, ra.r, ra.b).max(f32::EPSILON);
250            let ba = area(rb.l, rb.t, rb.r, rb.b).max(f32::EPSILON);
251            if i / (aa + ba - i).max(f32::EPSILON) > 0.8 || i / aa > 0.8 || i / ba > 0.8 {
252                let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
253                if pa != pb {
254                    parent[pa] = pb;
255                }
256            }
257        }
258    }
259    let mut groups: std::collections::BTreeMap<usize, Vec<usize>> =
260        std::collections::BTreeMap::new();
261    for i in 0..idx.len() {
262        let root = find(&mut parent, i);
263        groups.entry(root).or_default().push(i);
264    }
265    const AREA_THRESHOLD: f32 = 1.3;
266    const CONF_THRESHOLD: f32 = 0.05;
267    let area_of = |i: usize| {
268        let r = &regions[idx[i]];
269        area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
270    };
271    // `_should_prefer_cluster(candidate, other)` with the regular params.
272    let prefer = |cand: usize, other: usize| -> bool {
273        let (c, o) = (&regions[idx[cand]], &regions[idx[other]]);
274        let area_ratio = area_of(cand) / area_of(other);
275        if c.label == "list_item" && o.label == "text" && (1.0 - area_ratio).abs() < 0.2 {
276            return true;
277        }
278        if c.label == "code" && inter(o, c.l, c.t, c.r, c.b) / area_of(other) > 0.8 {
279            return true;
280        }
281        !(area_ratio <= AREA_THRESHOLD && o.score - c.score > CONF_THRESHOLD)
282    };
283    let mut drop = vec![false; regions.len()];
284    let mut unions: Vec<(usize, (f32, f32, f32, f32))> = Vec::new();
285    for group in groups.values() {
286        if group.len() < 2 {
287            continue;
288        }
289        let mut best: Option<usize> = None;
290        for &cand in group {
291            if group
292                .iter()
293                .all(|&other| other == cand || prefer(cand, other))
294            {
295                best = Some(match best {
296                    None => cand,
297                    Some(cur)
298                        if area_of(cand) > area_of(cur)
299                            && regions[idx[cur]].score - regions[idx[cand]].score
300                                <= CONF_THRESHOLD =>
301                    {
302                        cand
303                    }
304                    Some(cur) => cur,
305                });
306            }
307        }
308        // docling falls back to the group's first cluster; the highest score
309        // is the deterministic equivalent for a set with no insertion order.
310        let keep = best.unwrap_or_else(|| {
311            *group
312                .iter()
313                .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(&regions[idx[b]].score))
314                .expect("non-empty group")
315        });
316        let mut u = (
317            f32::INFINITY,
318            f32::INFINITY,
319            f32::NEG_INFINITY,
320            f32::NEG_INFINITY,
321        );
322        for &i in group {
323            let r = &regions[idx[i]];
324            u = (u.0.min(r.l), u.1.min(r.t), u.2.max(r.r), u.3.max(r.b));
325            if i != keep {
326                drop[idx[i]] = true;
327            }
328        }
329        unions.push((idx[keep], u));
330    }
331    for (i, (l, t, r, b)) in unions {
332        let k = &mut regions[i];
333        (k.l, k.t, k.r, k.b) = (l, t, r, b);
334    }
335    let mut keep_iter = drop.into_iter();
336    regions.retain(|_| !keep_iter.next().expect("aligned"));
337}
338
339/// `intersection_over_union` of two regions.
340fn iou(a: &Region, b: &Region) -> f32 {
341    let i = inter(a, b.l, b.t, b.r, b.b);
342    let u = area(a.l, a.t, a.r, a.b) + area(b.l, b.t, b.r, b.b) - i;
343    if u > 0.0 {
344        i / u
345    } else {
346        0.0
347    }
348}
349
350/// docling's `_resolve_coincident_pairs` (#4059, 2.122): for every (loser,
351/// winner) pair at a near-identical box (IoU > 0.8) whose confidences are
352/// within 0.1 (`loser.score - winner.score < 0.1`), the loser label is dropped
353/// so the label with the richer downstream semantic survives. Nothing else —
354/// containment, area — is considered; a clearly more confident loser stays.
355fn coincident_losers(regions: &[Region], losers: &[usize], winners: &[usize]) -> Vec<usize> {
356    let mut out = Vec::new();
357    for &li in losers {
358        for &wi in winners {
359            if iou(&regions[li], &regions[wi]) > 0.8 && regions[li].score - regions[wi].score < 0.1
360            {
361                out.push(li);
362                break;
363            }
364        }
365    }
366    out
367}
368
369/// docling's `_handle_cross_type_overlaps` (2.122/2.123 shape): the layout
370/// model can emit one grounded region under several labels, and the picture /
371/// table / container buckets are de-overlapped independently, so such a region
372/// survives twice. Elect a winner for the near-identical pairs:
373///
374/// | pair                                  | loser     | winner              |
375/// |---------------------------------------|-----------|---------------------|
376/// | TABLE vs DOCUMENT_INDEX               | table     | document_index      |
377/// | PICTURE vs TABLE / DOCUMENT_INDEX     | picture   | the table-like      |
378/// | FORM / KEY_VALUE_REGION vs *surviving* TABLE / DOCUMENT_INDEX / PICTURE | container | structured element |
379///
380/// IoU (not containment) so a genuine small figure inside a large table region
381/// is not removed; the confidence tolerance keeps a clearly more confident
382/// loser (an earlier port dropped every coincident picture regardless).
383fn handle_cross_type_overlaps(regions: Vec<Region>) -> Vec<Region> {
384    let by = |pred: &dyn Fn(&str) -> bool| -> Vec<usize> {
385        (0..regions.len())
386            .filter(|&i| pred(regions[i].label))
387            .collect()
388    };
389    let tables = by(&|l| l == "table");
390    let doc_indices = by(&|l| l == "document_index");
391    let pictures = by(&|l| l == "picture");
392    let containers = by(&|l| matches!(l, "form" | "key_value_region"));
393    let mut drop = vec![false; regions.len()];
394    for i in coincident_losers(&regions, &tables, &doc_indices) {
395        drop[i] = true;
396    }
397    let table_like: Vec<usize> = tables.iter().chain(&doc_indices).copied().collect();
398    for i in coincident_losers(&regions, &pictures, &table_like) {
399        drop[i] = true;
400    }
401    let structured: Vec<usize> = table_like
402        .iter()
403        .chain(&pictures)
404        .copied()
405        .filter(|&i| !drop[i])
406        .collect();
407    for i in coincident_losers(&regions, &containers, &structured) {
408        drop[i] = true;
409    }
410    let mut drop = drop.into_iter();
411    let mut regions = regions;
412    regions.retain(|_| !drop.next().expect("aligned"));
413    regions
414}
415
416pub fn resolve(regions: Vec<Region>) -> Vec<Region> {
417    let regions = handle_cross_type_overlaps(regions);
418    // De-overlap each bucket on its own.
419    let pictures = greedy(
420        regions
421            .iter()
422            .filter(|r| r.label == "picture")
423            .cloned()
424            .collect(),
425    );
426    // Tables and containers are separate buckets since docling 2.123
427    // (`TABLE_TYPES` vs `CONTAINER_TYPES`, docling#4064): a form drawn around a
428    // table no longer competes with it for survival — the table nests inside
429    // the container instead (`order_with_containers`).
430    let mut tables = greedy(
431        regions
432            .iter()
433            .filter(|r| is_table_like(r.label))
434            .cloned()
435            .collect(),
436    );
437    // `greedy` only drops a table mostly inside a *more* confident one, so a
438    // low-score whole-page table proposed over the column tables it contains
439    // (a two-column glossary page: 0.53 over 0.71/0.67/0.66) survived next to them and every
440    // cell was emitted twice. docling's `_remove_overlapping_clusters(tables,
441    // "wrapper")` groups tables whose boxes overlap (IoU > 0.8, or either one
442    // > 80 % inside the other) and keeps one per group: run it on what
443    // `greedy` leaves, like `merge_overlapping_regulars` does for regulars.
444    remove_overlapping_specials(&mut tables, |_| true, 2.0, 0.2);
445    let containers = greedy(
446        regions
447            .iter()
448            .filter(|r| matches!(r.label, "form" | "key_value_region"))
449            .cloned()
450            .collect(),
451    );
452    let mut kept = greedy(
453        regions
454            .iter()
455            .filter(|r| r.label != "picture" && !is_wrapper(r.label))
456            .cloned()
457            .collect(),
458    );
459    dedup_nested_code(&mut kept);
460    kept.extend(pictures);
461    kept.extend(tables);
462    kept.extend(containers);
463    kept
464}
465
466/// Drop a regular region that is >80% contained in a surviving special region we
467/// render **as a single unit** — a table/table-of-contents index — ported from
468/// docling's "Remove regular clusters that are included in wrappers" step: the
469/// special absorbs it as a child (a table cell), so it must not also be emitted
470/// as its own paragraph/list-item. This stops the survey list-items from
471/// appearing both inside the detected table and again as bullets
472/// (`table_mislabeled_as_picture`).
473///
474/// `picture` regions are **not** in the swallow set: docling keeps a picture's
475/// contained clusters as the `PictureItem`'s *children* in the document JSON
476/// (`_set_cluster_children`, `ReadingOrderModel._add_child_elements`), while
477/// its `MarkdownPictureSerializer` prints only the caption and the image — the
478/// children never reach the Markdown (verified against the corpus groundtruth:
479/// `amt_handbook`'s in-figure callout labels are absent). So the regulars
480/// inside a picture stay in the region list, claim their cells and are fitted
481/// like any regular, and [`assemble_page`] lifts them out of the page's
482/// reading order into a [`Node::PictureChildren`] after their picture (see
483/// [`picture_parents`]). A line only partially under a figure box (straddling
484/// its border, ≤80 % contained) is no picture's child and is emitted, as
485/// docling emits it (#165).
486///
487/// `form` / `key_value_region` wrappers are deliberately **excluded**: this
488/// pipeline does not render them as a structured block (they are skipped), so
489/// their textual content comes precisely from the contained regular regions —
490/// A `page_footer` that is really the body of the last heading on the page
491/// becomes `text`.
492///
493/// The layout model labels the bottom margin by position as much as by
494/// content: a one-line paragraph that happens to sit where a running footer
495/// would — a CV's `Languages` line under its `## Languages` heading, the last
496/// entry of a section that runs to the page edge — comes back as
497/// `page_footer` (0.88 against 0.49 for `text` on the reporting file, with
498/// the fp32 model and either renderer alike), and Markdown drops furniture,
499/// so the document loses real content while the JSON keeps it under the
500/// wrong label. docling fails the same way. Deliberate deviation, gated so a
501/// real footer keeps its label: the nearest `section_header` above the footer
502/// must overlap it horizontally, end within 2.5 footer heights of its top,
503/// and be the **last body element** of its column — no other non-furniture
504/// region starts at or below the heading's bottom edge over the same span —
505/// and the footer must be a line of text, not a page number: at least 40 %
506/// of the page width. A running footer never follows a heading that has no
507/// body of its own, so the combination is the misread heading body.
508pub fn reclaim_heading_body_footers(regions: &mut [Region], page_w: f32) {
509    let n = regions.len();
510    let overlap_x = |a: &Region, b: &Region| a.r.min(b.r) - a.l.max(b.l) > 0.0;
511    for fi in 0..n {
512        let f = regions[fi].clone();
513        if f.label != "page_footer" || f.r - f.l < 0.4 * page_w {
514            continue;
515        }
516        let fh = (f.b - f.t).max(1.0);
517        // The nearest heading above the footer, over the footer's span.
518        let heading = (0..n)
519            .filter(|&j| {
520                let h = &regions[j];
521                j != fi && h.label == "section_header" && h.b <= f.t + 0.5 * fh && overlap_x(h, &f)
522            })
523            .min_by(|&a, &b| regions[b].b.total_cmp(&regions[a].b));
524        let Some(hi) = heading else {
525            continue;
526        };
527        let h = regions[hi].clone();
528        if f.t - h.b > 2.5 * fh {
529            continue;
530        }
531        // The heading must have no body of its own: nothing but the footer
532        // starts at or below its bottom edge over the heading's or footer's
533        // span (a heading whose paragraph follows is not this case, and a
534        // heading with the footer far below it was filtered above).
535        let has_body = (0..n).any(|j| {
536            let r = &regions[j];
537            j != fi
538                && j != hi
539                && !matches!(r.label, "page_footer" | "page_header")
540                && r.t >= h.b - 0.5 * fh
541                && (overlap_x(r, &h) || overlap_x(r, &f))
542        });
543        if has_body {
544            continue;
545        }
546        regions[fi].label = "text";
547    }
548}
549
550/// dropping those would erase the page (e.g. `right_to_left_03`'s form-heavy
551/// pages). Runs *after* [`drop_false_pictures`] so a phantom picture can't
552/// swallow real text on its way out.
553pub fn drop_contained_regulars(regions: &mut Vec<Region>) {
554    let specials: Vec<(f32, f32, f32, f32)> = regions
555        .iter()
556        .filter(|r| is_table_like(r.label))
557        .map(|r| (r.l, r.t, r.r, r.b))
558        .collect();
559    if specials.is_empty() {
560        return;
561    }
562    regions.retain(|r| {
563        if r.label == "picture" || is_wrapper(r.label) {
564            return true;
565        }
566        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
567        !specials
568            .iter()
569            .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.8)
570    });
571}
572
573/// docling's `_set_cluster_children` for pictures: a regular region (one that
574/// claims cells) whose box is > 80 % inside a picture's box
575/// (`intersection_over_self`) is that picture's child — not a page element
576/// (it leaves the reading order and the layout score), but an item under the
577/// `PictureItem` in the JSON. Returns, per region, the index of its parent
578/// picture: the smallest containing one when pictures nest (upstream would
579/// attach it to each).
580pub fn picture_parents(regions: &[Region]) -> Vec<Option<usize>> {
581    regions
582        .iter()
583        .map(|r| {
584            if !claims_cells(r) {
585                return None;
586            }
587            let ra = area(r.l, r.t, r.r, r.b).max(1.0);
588            regions
589                .iter()
590                .enumerate()
591                .filter(|(_, p)| p.label == "picture" && inter(r, p.l, p.t, p.r, p.b) / ra > 0.8)
592                .min_by(|(_, a), (_, b)| {
593                    area(a.l, a.t, a.r, a.b).total_cmp(&area(b.l, b.t, b.r, b.b))
594                })
595                .map(|(i, _)| i)
596        })
597        .collect()
598}
599
600/// True for a bare, single-token source-code language label (`XML`, `C#`, `JSON`,
601/// `bash`, …) — the little header the docs render above a code block. Matched
602/// case-insensitively; anything with whitespace or longer than a token is out.
603fn is_code_language(t: &str) -> bool {
604    let t = t.trim();
605    if t.is_empty() || t.chars().any(char::is_whitespace) || t.chars().count() > 12 {
606        return false;
607    }
608    const LANGS: &[&str] = &[
609        "xml",
610        "html",
611        "xhtml",
612        "json",
613        "jsonc",
614        "yaml",
615        "yml",
616        "toml",
617        "ini",
618        "c#",
619        "csharp",
620        "f#",
621        "fsharp",
622        "vb",
623        "c",
624        "c++",
625        "cpp",
626        "java",
627        "kotlin",
628        "scala",
629        "go",
630        "golang",
631        "rust",
632        "swift",
633        "javascript",
634        "js",
635        "typescript",
636        "ts",
637        "jsx",
638        "tsx",
639        "python",
640        "py",
641        "ruby",
642        "rb",
643        "php",
644        "perl",
645        "lua",
646        "r",
647        "dart",
648        "bash",
649        "sh",
650        "shell",
651        "powershell",
652        "zsh",
653        "batch",
654        "cmd",
655        "sql",
656        "tsql",
657        "plsql",
658        "graphql",
659        "dockerfile",
660        "makefile",
661        "css",
662        "scss",
663        "sass",
664        "less",
665        "markdown",
666        "md",
667        "tex",
668        "latex",
669        "diff",
670        "proto",
671        "razor",
672        "cshtml",
673        "xaml",
674        "aspx",
675        "http",
676    ];
677    let lower = t.to_ascii_lowercase();
678    LANGS.contains(&lower.as_str())
679}
680
681/// Mark the region indices that are a code block's **language label** — a bare
682/// `XML`/`C#`/… token sitting directly above a `code` region — so they are consumed
683/// rather than emitted as their own stray paragraph/heading. The label may also be
684/// captured inside a wider code box (rendered as the fence's first line); dropping
685/// the standalone copy just removes the duplicate.
686fn code_language_labels(regions: &[Region], cells: &[TextCell]) -> Vec<bool> {
687    let mut drop = vec![false; regions.len()];
688    for (i, r) in regions.iter().enumerate() {
689        if matches!(r.label, "code" | "picture" | "table") {
690            continue;
691        }
692        if !is_code_language(&region_text(r, cells)) {
693            continue;
694        }
695        // The label sits just above the code (a blank line's gap) or is swallowed
696        // into the top of a wider code box; either way it is that block's label.
697        // The window is generous because the label's own font is small, so a
698        // one-line gap is several times its height.
699        let line_h = (r.b - r.t).abs().max(1.0);
700        let window = (line_h * 4.0).max(28.0);
701        let labels_code = regions.iter().enumerate().any(|(j, c)| {
702            if j == i || c.label != "code" {
703                return false;
704            }
705            let gap = c.t - r.b; // >0 when the code is below the label
706            let h_overlap = (r.r.min(c.r) - r.l.max(c.l)).max(0.0);
707            gap > -line_h * 3.0 && gap < window && h_overlap > 0.0
708        });
709        if labels_code {
710            drop[i] = true;
711        }
712    }
713    drop
714}
715
716/// Collapse `code` regions where one is nested inside another, keeping the larger.
717///
718/// RT-DETR sometimes emits a tight code box *and* a wider near-duplicate that also
719/// captures the block's language label (`XML`, `C#`, …). When the tight box scores
720/// higher it is kept first, and the wider container — not "mostly inside" the tight
721/// box — survives [`resolve`]'s greedy pass, so the block is emitted twice. Keeping
722/// the **larger** box (rather than dropping it) collapses the pair without leaking
723/// the container's extra cells back out as orphan text, since the larger box still
724/// covers every cell. Restricted to `code` so genuinely distinct nested regions of
725/// other kinds are untouched.
726fn dedup_nested_code(kept: &mut Vec<Region>) {
727    let mut drop = vec![false; kept.len()];
728    for i in 0..kept.len() {
729        if kept[i].label != "code" {
730            continue;
731        }
732        let ai = area(kept[i].l, kept[i].t, kept[i].r, kept[i].b).max(1.0);
733        for j in 0..kept.len() {
734            if i == j || drop[j] || kept[j].label != "code" {
735                continue;
736            }
737            let aj = area(kept[j].l, kept[j].t, kept[j].r, kept[j].b).max(1.0);
738            // Drop i when it is mostly inside a strictly larger code box j.
739            let overlap = inter(&kept[i], kept[j].l, kept[j].t, kept[j].r, kept[j].b);
740            if aj > ai && overlap / ai > 0.7 {
741                drop[i] = true;
742                break;
743            }
744        }
745    }
746    let mut keep = drop.iter();
747    kept.retain(|_| !*keep.next().unwrap());
748}
749
750/// Fraction of the page's non-empty text cells that some detected region
751/// claims (>0.2 intersection-over-self, docling's assignment rule). 1.0 for a
752/// page without text cells.
753///
754/// The int8-layout guard keys off this: a dense digital page whose detections
755/// cover almost none of its text is the signature of quantized confidences
756/// flipping under the 0.5 label thresholds on this CPU's kernels — not of a
757/// genuinely empty layout — and is worth re-running on the fp32 graph.
758pub fn layout_cell_coverage(regions: &[Region], cells: &[TextCell]) -> f32 {
759    let mut total = 0usize;
760    let mut covered = 0usize;
761    for c in cells {
762        if c.text.trim().is_empty() {
763            continue;
764        }
765        total += 1;
766        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
767        if regions
768            .iter()
769            .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
770        {
771            covered += 1;
772        }
773    }
774    if total == 0 {
775        1.0
776    } else {
777        covered as f32 / total as f32
778    }
779}
780
781/// Append `text` regions for cells the layout left uncovered ("orphan cells"),
782/// the way docling's `LayoutPostprocessor` does (`create_orphan_clusters`): any
783/// non-empty cell that no kept region covers (>50% of the cell's area) becomes a
784/// text region of its own, so text the detector missed (a stray `.`, a small
785/// label) is still emitted instead of silently dropped. Adjacent orphan cells on a
786/// line are merged so a missed paragraph doesn't shatter into one block per line.
787pub fn add_orphan_regions(regions: &mut Vec<Region>, cells: &[TextCell]) {
788    // docling assigns each cell to its single best-overlapping cluster at
789    // intersection-over-self > 0.2 and serializes exactly the assigned cells —
790    // and since [`region_texts_exclusive`] now emits under that very rule, the
791    // claim test here matches it: any cell over 0.2 will actually render in
792    // its best region, everything else becomes an orphan. Completeness by
793    // construction, with no (0.2, 0.5] hole (the old > 0.5 serializer needed
794    // the claim test raised to > 0.5 to keep right_to_left_03's `20300` from
795    // vanishing; the exclusive port closes that structurally).
796    //
797    // Only *regular* clusters claim cells: docling's `_find_unassigned_cells`
798    // walks `regular_clusters` alone, so a cell under a `picture` or a wrapper
799    // (`table`/`document_index`/`form`/`key_value_region`) that no regular
800    // cluster covers still becomes an orphan text cluster (#165). The orphans
801    // that end up *fully* inside the special are re-dropped by
802    // [`drop_contained_regulars`] (docling's Markdown drops them the same way
803    // — a picture's children never reach its `MarkdownPictureSerializer`
804    // output, a table's text renders through the reconstructed grid). The
805    // observable fix is the border-straddlers: a line only partially under a
806    // figure box used to lose its cells to the picture's 0.2 claim and vanish
807    // — now it forms an orphan region and is emitted, as docling does.
808    let assigned = |c: &TextCell| {
809        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
810        regions
811            .iter()
812            .filter(|r| r.label != "picture" && !is_wrapper(r.label))
813            .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
814    };
815    // Collect orphan cells (non-empty, unassigned), in page order.
816    let mut orphans: Vec<&TextCell> = cells
817        .iter()
818        .filter(|c| !c.text.trim().is_empty() && !assigned(c))
819        .collect();
820    if orphans.is_empty() {
821        return;
822    }
823    orphans.sort_by(|a, b| a.t.total_cmp(&b.t).then(a.l.total_cmp(&b.l)));
824    // Merge cells that sit on the same line and nearly touch into one region, so a
825    // dropped multi-word line stays one block (docling's refinement merges these).
826    let mut merged: Vec<Region> = Vec::new();
827    for c in orphans {
828        let h = (c.b - c.t).abs().max(1.0);
829        if let Some(last) = merged.last_mut() {
830            let same_line = (last.t - c.t).abs() < h * 0.5;
831            let touching = c.l <= last.r + h && c.l >= last.l - h;
832            // Both tolerances scale with the cell's own height, so a run set
833            // vertically (the arXiv stamp up the margin, #528 — a cell a few
834            // points wide and hundreds tall) would read as "on the line" of
835            // whatever precedes it and glue a whole column into one region.
836            // Lines of one row differ by a drop cap's few multiples at most.
837            let lh = (last.b - last.t).abs().max(1.0);
838            let comparable = h <= 4.0 * lh && lh <= 4.0 * h;
839            if same_line && touching && comparable {
840                last.l = last.l.min(c.l);
841                last.r = last.r.max(c.r);
842                last.t = last.t.min(c.t);
843                last.b = last.b.max(c.b);
844                continue;
845            }
846        }
847        merged.push(Region {
848            label: "text",
849            score: 0.0,
850            l: c.l,
851            t: c.t,
852            r: c.r,
853            b: c.b,
854        });
855    }
856    regions.extend(merged);
857}
858
859/// Demote a `picture` region that is really a **text panel** — a paragraph block
860/// the layout model boxed as a figure because it is typeset on a colored
861/// background (terms-and-conditions callouts, quote boxes) — into ordinary
862/// `text` regions, one per paragraph, so its words are read instead of shipped
863/// as pixels. docling loses this text the same way (cells assigned to a picture
864/// cluster are never serialized); this is a deliberate improvement, not parity.
865///
866/// The gate is conservative so a genuine figure keeps its crop: the region must
867/// contain at least three text lines whose median width spans most of the panel
868/// (axis labels and chat bubbles are narrow and varied) and whose cells cover a
869/// substantial fraction of its area (a photo or chart with sparse labels does
870/// not). Paragraph boundaries are re-derived from the line pitch: a vertical gap
871/// clearly larger than the panel's own leading starts a new `text` region, so
872/// the panel doesn't collapse into one giant paragraph.
873///
874/// Works on any cell source — the digital text layer or OCR lines recognized
875/// from the picture crop — so the native and browser paths, with or without
876/// force-OCR, demote identically.
877pub fn recover_text_panels(regions: &mut Vec<Region>, cells: &[TextCell]) {
878    // A *captioned* picture is a genuine figure whatever it contains — the
879    // corpus is full of document screenshots ("Figure 3: …" above a page
880    // image) that are exactly as dense and wide as a text panel. Only an
881    // uncaptioned picture is a demotion candidate.
882    let captioned: Vec<bool> = regions
883        .iter()
884        .map(|r| {
885            r.label == "picture"
886                && regions.iter().any(|c| {
887                    c.label == "caption" && c.r.min(r.r) - c.l.max(r.l) > 0.0 && {
888                        let gap = if c.t >= r.b {
889                            c.t - r.b
890                        } else if r.t >= c.b {
891                            r.t - c.b
892                        } else {
893                            f32::MAX // vertically overlapping: not a caption
894                        };
895                        gap <= 25.0
896                    }
897                })
898        })
899        .collect();
900    let mut out: Vec<Region> = Vec::with_capacity(regions.len());
901    // Synthesized paragraphs and the demoted panels' boxes are kept separate
902    // from `out` until the end: the dedup filter below must not confuse a
903    // paragraph we just built with a pre-existing region inside the panel.
904    let mut demoted_paras: Vec<Region> = Vec::new();
905    let mut demoted_boxes: Vec<(f32, f32, f32, f32)> = Vec::new();
906    for (i, r) in regions.drain(..).enumerate() {
907        if r.label != "picture" || captioned[i] {
908            out.push(r);
909            continue;
910        }
911        let inside: Vec<&TextCell> = cells
912            .iter()
913            .filter(|c| {
914                !c.text.trim().is_empty() && {
915                    let ca = area(c.l, c.t, c.r, c.b).max(1.0);
916                    inter(&r, c.l, c.t, c.r, c.b) / ca > 0.5
917                }
918            })
919            .collect();
920        // Group the contained cells into lines by vertical overlap (the same
921        // rule region_text orders by), tracking each line's union box.
922        let mut lines: Vec<(f32, f32, f32, f32)> = Vec::new(); // (t, b, l, r)
923        for c in &inside {
924            let (ct, cb) = (c.t.min(c.b), c.t.max(c.b));
925            match lines.iter_mut().find(|(lt, lb, _, _)| {
926                let ov = cb.min(*lb) - ct.max(*lt);
927                ov > 0.5 * (cb - ct).min(*lb - *lt).max(1.0)
928            }) {
929                Some((lt, lb, ll, lr)) => {
930                    *lt = lt.min(ct);
931                    *lb = lb.max(cb);
932                    *ll = ll.min(c.l);
933                    *lr = lr.max(c.r);
934                }
935                None => lines.push((ct, cb, c.l, c.r)),
936            }
937        }
938        if lines.len() < 3 {
939            out.push(r);
940            continue;
941        }
942        let panel_w = (r.r - r.l).max(1.0);
943        let coverage = inside.iter().map(|c| area(c.l, c.t, c.r, c.b)).sum::<f32>()
944            / area(r.l, r.t, r.r, r.b).max(1.0);
945        let mut widths: Vec<f32> = lines.iter().map(|(_, _, l, rr)| rr - l).collect();
946        widths.sort_by(f32::total_cmp);
947        // A figure's text is ragged: a title line, small axis/tick labels, and
948        // OCR boxes over the plot area come out at wildly different heights,
949        // whereas a real text panel is set in one face with constant leading.
950        // Require near-uniform line heights (median absolute deviation ≤ 35%
951        // of the median) so an uncaptioned chart keeps its crop even when its
952        // labels are dense enough to pass the coverage gate (#173) — garbled
953        // OCR of its bars is not content.
954        let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
955        heights.sort_by(f32::total_cmp);
956        let h_med = heights[heights.len() / 2].max(1.0);
957        let mut devs: Vec<f32> = heights.iter().map(|h| (h - h_med).abs()).collect();
958        devs.sort_by(f32::total_cmp);
959        let uniform = devs[devs.len() / 2] <= 0.35 * h_med;
960        let text_panel = coverage >= 0.2 && widths[widths.len() / 2] >= 0.45 * panel_w && uniform;
961        if !text_panel {
962            out.push(r);
963            continue;
964        }
965        lines.sort_by(|a, b| a.0.total_cmp(&b.0));
966        let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
967        heights.sort_by(f32::total_cmp);
968        let h = heights[heights.len() / 2].max(1.0);
969        let mut gaps: Vec<f32> = lines
970            .windows(2)
971            .map(|w| (w[1].0 - w[0].1).max(0.0))
972            .collect();
973        gaps.sort_by(f32::total_cmp);
974        let leading = if gaps.is_empty() {
975            0.0
976        } else {
977            gaps[gaps.len() / 2]
978        };
979        let brk = (1.8 * leading).max(0.75 * h);
980        let mut para: Option<(f32, f32, f32, f32)> = None; // (l, t, r, b) union
981        for (t, b, l, rr) in &lines {
982            match &mut para {
983                Some((pl, _, pr, pb)) if *t - *pb <= brk => {
984                    *pl = pl.min(*l);
985                    *pr = pr.max(*rr);
986                    *pb = pb.max(*b);
987                }
988                _ => {
989                    if let Some((pl, pt, pr, pb)) = para.take() {
990                        demoted_paras.push(Region {
991                            label: "text",
992                            score: r.score,
993                            l: pl,
994                            t: pt,
995                            r: pr,
996                            b: pb,
997                        });
998                    }
999                    para = Some((*l, *t, *rr, *b));
1000                }
1001            }
1002        }
1003        if let Some((pl, pt, pr, pb)) = para {
1004            demoted_paras.push(Region {
1005                label: "text",
1006                score: r.score,
1007                l: pl,
1008                t: pt,
1009                r: pr,
1010                b: pb,
1011            });
1012        }
1013        demoted_boxes.push((r.l, r.t, r.r, r.b));
1014    }
1015    // The paragraphs are rebuilt from *all* of the panel's cells, so any
1016    // surviving text region inside a demoted panel (an orphan cluster or a
1017    // layout-detected fragment — pictures no longer swallow them, #165) would
1018    // say the same words twice. Consume those; wrappers and pictures stay.
1019    if !demoted_boxes.is_empty() {
1020        out.retain(|r| {
1021            r.label == "picture" || is_wrapper(r.label) || {
1022                let ra = area(r.l, r.t, r.r, r.b).max(1.0);
1023                !demoted_boxes
1024                    .iter()
1025                    .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.5)
1026            }
1027        });
1028    }
1029    // docling's "Remove regular clusters that are included in wrappers" (a
1030    // regular > 80 % inside a table is absorbed by it) already ran as
1031    // [`drop_contained_regulars`], but before this demotion created new
1032    // regulars. Apply it to them too: a panel that coincides with a table (a
1033    // dense data table detected as picture 0.80 and table 0.62 on one box;
1034    // `_handle_cross_type_overlaps` keeps both once the picture is ≥ 0.1 more
1035    // confident) rebuilds the table's words as a paragraph the grid already
1036    // renders. A panel inside another picture is left as it was.
1037    demoted_paras.retain(|p| {
1038        let pa = area(p.l, p.t, p.r, p.b).max(1.0);
1039        !out.iter()
1040            .any(|s| is_table_like(s.label) && inter(p, s.l, s.t, s.r, s.b) / pa > 0.8)
1041    });
1042    out.extend(demoted_paras);
1043    *regions = out;
1044}
1045
1046/// Drop a `picture` detection covering more than 90 % of the page — docling's
1047/// `LayoutPostprocessor._process_special_clusters` "Filter out full-page
1048/// pictures" (upstream since 2.15), applied to the thresholded detections
1049/// before overlap resolution. A box that big is the page itself, not a figure
1050/// on it: the layout model emits one for a whole-page diagram (a LaTeX figure
1051/// PDF cropped to its drawing), a plate, or a scan, and keeping it would swallow
1052/// every text cell on the page as picture children — the diagram's labels and
1053/// caption vanish behind a lone `<!-- image -->`, where docling reads them out as
1054/// text. `page_w`/`page_h` is the display-frame page box.
1055pub fn drop_full_page_pictures(regions: &mut Vec<Region>, page_w: f32, page_h: f32) {
1056    let page_area = (page_w * page_h).max(1.0);
1057    regions.retain(|r| r.label != "picture" || area(r.l, r.t, r.r, r.b) / page_area <= 0.90);
1058}
1059
1060/// Drop a `picture` detection that is a small, empty, low-confidence margin box on
1061/// a **text page** — a false positive the RT-DETR layout sometimes emits (e.g.
1062/// `right_to_left_02`'s phantom right-column picture, score 0.40); docling does not
1063/// emit it. The gate is deliberately narrow so a genuine figure is never dropped:
1064/// (1) only on pages with a digital text layer — image/scanned/figure pages have
1065/// no `cells` yet at this point (OCR runs later), so their pictures, which *are*
1066/// the content, are kept; (2) only a box covering < 25 % of the page (a margin
1067/// artifact, not a dominant figure); (3) only when it contains no text and scores
1068/// below 0.5 (real empty figures in the corpus all score ≥ 0.86).
1069pub fn drop_false_pictures(
1070    regions: &mut Vec<Region>,
1071    cells: &[TextCell],
1072    page_w: f32,
1073    page_h: f32,
1074) {
1075    if cells.iter().all(|c| c.text.trim().is_empty()) {
1076        return; // no digital text layer (image/scanned page) — keep all pictures
1077    }
1078    // A text-document page carries several text-bearing non-picture regions (so a
1079    // spurious margin picture is clearly extra). A slide / figure page has at most
1080    // one — there the picture is the content, so never drop it.
1081    let content_regions = regions
1082        .iter()
1083        .filter(|r| r.label != "picture" && !region_text(r, cells).trim().is_empty())
1084        .count();
1085    if content_regions < 2 {
1086        return;
1087    }
1088    let page_area = (page_w * page_h).max(1.0);
1089    regions.retain(|r| {
1090        if r.label != "picture" || r.score >= 0.5 {
1091            return true;
1092        }
1093        if area(r.l, r.t, r.r, r.b) / page_area >= 0.25 {
1094            return true; // a dominant figure, not a margin artifact
1095        }
1096        // Keep it if any text cell falls mostly inside (a real captioned/labelled
1097        // figure); drop only the genuinely empty low-confidence boxes.
1098        cells.iter().any(|c| {
1099            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1100            !c.text.trim().is_empty() && inter(r, c.l, c.t, c.r, c.b) / ca > 0.5
1101        })
1102    });
1103}
1104
1105/// A small digit-only region in the top/bottom margin: a page number. docling
1106/// emits `right_to_left_02`'s bottom `11` as the page's *first* text item (its
1107/// reading-order model floats the page number to the front), whereas our
1108/// position-based ordering would place a bottom region last.
1109fn is_page_number(region: &Region, cells: &[TextCell], page_h: f32) -> bool {
1110    let t = region_text(region, cells);
1111    let t = t.trim();
1112    !t.is_empty()
1113        && t.chars().all(|c| c.is_ascii_digit())
1114        && (region.b - region.t).abs() < 30.0
1115        && (region.t < page_h * 0.12 || region.b > page_h * 0.88)
1116}
1117
1118/// docling's `form` / `key_value_region` *containers* (2.123, docling#4064):
1119/// every region sitting > 0.8 inside one — text, list items, and since #4064
1120/// tables and pictures too — is that container's child. Children are
1121/// reading-ordered among themselves and emitted as one block where the
1122/// container falls in the page's top-level order (a `form_area` /
1123/// `key_value_area` group upstream), instead of interleaving with the text
1124/// around the form. A child inside several containers belongs to the smallest
1125/// (then most confident, then first); a container with children shrinks to
1126/// their union for the top-level ordering, like upstream's bbox adjustment.
1127///
1128/// The containers themselves are still not emitted (`is_skipped`), so the
1129/// Markdown is exactly upstream's — a group prints only its children.
1130///
1131/// `cids` are the items' positions in docling's assembly order
1132/// ([`cluster_cids`]) — the reading-order predictor's same-row rule (#424)
1133/// pairs consecutive ones, within the top level and within each container.
1134fn order_with_containers<T: Clone>(
1135    items: &mut Vec<T>,
1136    cids: &[usize],
1137    page_w: f32,
1138    page_h: f32,
1139    reg: impl Fn(&T) -> &Region,
1140) {
1141    let is_container = |r: &Region| matches!(r.label, "form" | "key_value_region");
1142    let containers: Vec<usize> = (0..items.len())
1143        .filter(|&i| is_container(reg(&items[i])))
1144        .collect();
1145    if containers.is_empty() {
1146        order_regions(items, cids, page_w, page_h, reg);
1147        return;
1148    }
1149    // Parent container per item (containers never nest in each other here —
1150    // upstream assigns regulars and tables/pictures only).
1151    let mut parent: Vec<Option<usize>> = vec![None; items.len()];
1152    for i in 0..items.len() {
1153        let r = reg(&items[i]);
1154        if is_container(r) {
1155            continue;
1156        }
1157        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
1158        let mut best: Option<(usize, f32, f32)> = None; // (idx, area, -score)
1159        for &c in &containers {
1160            let cr = reg(&items[c]);
1161            if inter(r, cr.l, cr.t, cr.r, cr.b) / ra > 0.8 {
1162                let key = (area(cr.l, cr.t, cr.r, cr.b), -cr.score);
1163                if best.is_none_or(|(_, a, s)| key.0 < a || (key.0 == a && key.1 < s)) {
1164                    best = Some((c, key.0, key.1));
1165                }
1166            }
1167        }
1168        parent[i] = best.map(|(c, _, _)| c);
1169    }
1170    // Top-level pass: non-children plus the containers, the latter shrunk to
1171    // their children's union.
1172    let mut top: Vec<(usize, Region)> = Vec::new();
1173    for i in 0..items.len() {
1174        if parent[i].is_some() {
1175            continue;
1176        }
1177        let mut r = reg(&items[i]).clone();
1178        if is_container(&r) {
1179            let kids: Vec<&Region> = (0..items.len())
1180                .filter(|&k| parent[k] == Some(i))
1181                .map(|k| reg(&items[k]))
1182                .collect();
1183            if !kids.is_empty() {
1184                r.l = kids.iter().map(|k| k.l).fold(f32::INFINITY, f32::min);
1185                r.t = kids.iter().map(|k| k.t).fold(f32::INFINITY, f32::min);
1186                r.r = kids.iter().map(|k| k.r).fold(f32::NEG_INFINITY, f32::max);
1187                r.b = kids.iter().map(|k| k.b).fold(f32::NEG_INFINITY, f32::max);
1188            }
1189        }
1190        top.push((i, r));
1191    }
1192    let top_cids: Vec<usize> = top.iter().map(|(i, _)| cids[*i]).collect();
1193    order_regions(&mut top, &top_cids, page_w, page_h, |it| &it.1);
1194    let mut out: Vec<T> = Vec::with_capacity(items.len());
1195    for (i, _) in top {
1196        if is_container(reg(&items[i])) {
1197            let kid_idx: Vec<usize> = (0..items.len()).filter(|&k| parent[k] == Some(i)).collect();
1198            let mut kids: Vec<T> = kid_idx.iter().map(|&k| items[k].clone()).collect();
1199            let kid_cids: Vec<usize> = kid_idx.iter().map(|&k| cids[k]).collect();
1200            order_regions(&mut kids, &kid_cids, page_w, page_h, &reg);
1201            out.push(items[i].clone());
1202            out.extend(kids);
1203        } else {
1204            out.push(items[i].clone());
1205        }
1206    }
1207    *items = out;
1208}
1209
1210/// Furniture / not-yet-emitted labels.
1211fn is_skipped(label: &str) -> bool {
1212    matches!(
1213        label,
1214        "page_header" | "page_footer" | "form" | "key_value_region"
1215    )
1216}
1217
1218/// Reading-order sort of a page's regions, via the ported rule-based
1219/// [`reading_order`](crate::reading_order) predictor (docling's
1220/// `ReadingOrderPredictor`): an up/down geometry graph with same-row links
1221/// between `cids`-consecutive elements (#424), horizontal dilation and a
1222/// depth-first traversal, with `page_header`/`page_footer` ordered as their own
1223/// groups (first/last) as docling does.
1224fn order_regions<T: Clone>(
1225    items: &mut Vec<T>,
1226    cids: &[usize],
1227    page_w: f32,
1228    page_h: f32,
1229    reg: impl Fn(&T) -> &Region,
1230) {
1231    let boxes: Vec<(f32, f32, f32, f32)> = items
1232        .iter()
1233        .map(|it| {
1234            let r = reg(it);
1235            (r.l, r.t, r.r, r.b)
1236        })
1237        .collect();
1238    let is_header: Vec<bool> = items
1239        .iter()
1240        .map(|it| reg(it).label == "page_header")
1241        .collect();
1242    let is_footer: Vec<bool> = items
1243        .iter()
1244        .map(|it| reg(it).label == "page_footer")
1245        .collect();
1246    let order =
1247        crate::reading_order::order_page(&boxes, cids, &is_header, &is_footer, page_w, page_h);
1248    *items = order.iter().map(|&i| items[i].clone()).collect();
1249}
1250
1251/// docling's assembly order of a page's clusters (`LayoutPostprocessor`'s
1252/// final `_sort_clusters(mode="id")`, #424): each region's rank when sorted by
1253/// its first source cell, then by top edge, then left edge; a region with no
1254/// cells sorts after every one that has some. docling numbers its page
1255/// elements (`cid`) in this order, and the reading-order predictor's same-row
1256/// rule pairs elements with consecutive numbers, so the ranks are what
1257/// [`order_with_containers`] hands the predictor.
1258///
1259/// A regular region's first cell is the smallest index among the cells it
1260/// claims. A table, picture or container has no cells of its own upstream
1261/// either — its cells are its *children's*: the regular clusters > 0.8 inside
1262/// it, and upstream every cell no regular cluster claimed is an orphan cluster
1263/// of its own, so a table's interior text (which no regular cluster claims)
1264/// reaches the table through those orphans. Here that is the cells > 0.8
1265/// inside the region plus the claimed cells of the regular regions > 0.8
1266/// inside it. Without the interior cells every table would sort last, and two
1267/// side-by-side tables would then be consecutive and row-linked — reading the
1268/// right table's caption ahead of the left column's headings (2206 page 8).
1269pub fn cluster_cids(regions: &[Region], cells: &[TextCell]) -> Vec<usize> {
1270    let owned = assign_cells(regions, cells);
1271    let first_cell: Vec<usize> = regions
1272        .iter()
1273        .enumerate()
1274        .map(|(i, r)| {
1275            if claims_cells(r) {
1276                return owned[i].iter().copied().min().unwrap_or(usize::MAX);
1277            }
1278            let interior = cells
1279                .iter()
1280                .enumerate()
1281                .filter(|(_, c)| {
1282                    !c.text.trim().is_empty()
1283                        && inter(r, c.l, c.t, c.r, c.b) / area(c.l, c.t, c.r, c.b).max(1.0) > 0.8
1284                })
1285                .map(|(ci, _)| ci)
1286                .min();
1287            let children = regions
1288                .iter()
1289                .enumerate()
1290                .filter(|(j, child)| {
1291                    *j != i && claims_cells(child) && {
1292                        let ca = area(child.l, child.t, child.r, child.b).max(1.0);
1293                        inter(r, child.l, child.t, child.r, child.b) / ca > 0.8
1294                    }
1295                })
1296                .filter_map(|(j, _)| owned[j].iter().copied().min())
1297                .min();
1298            interior
1299                .into_iter()
1300                .chain(children)
1301                .min()
1302                .unwrap_or(usize::MAX)
1303        })
1304        .collect();
1305    let mut by_source: Vec<usize> = (0..regions.len()).collect();
1306    // Stable, like Python's `sorted`: full ties keep the layout order.
1307    by_source.sort_by(|&a, &b| {
1308        first_cell[a]
1309            .cmp(&first_cell[b])
1310            .then(regions[a].t.total_cmp(&regions[b].t))
1311            .then(regions[a].l.total_cmp(&regions[b].l))
1312    });
1313    let mut cids = vec![0; regions.len()];
1314    for (rank, &i) in by_source.iter().enumerate() {
1315        cids[i] = rank;
1316    }
1317    cids
1318}
1319
1320/// Clean a region's assembled text: undo soft-hyphen line wraps, map curly
1321/// quotes and the ellipsis to ASCII (matching docling), and collapse runs of
1322/// whitespace. pdfium emits the line-wrap hyphen as U+0002 in this corpus
1323/// (U+00AD elsewhere), so `word\u{2} continuation` is one hyphenated word —
1324/// drop the hyphen + the joining space and merge (`com\u{2} pact` → `compact`,
1325/// `end-to\u{2} end` → `end-toend`), exactly as docling does.
1326///
1327/// Token spacing is otherwise left as the geometric join produced it. We do not
1328/// tighten punctuation spacing: docling preserves the PDF's own spaces (it keeps
1329/// `{ ahn }`, `Name 1 .`, `[ 9 ]`), and a geometric gap heuristic diverges from
1330/// it more than a plain single-space join does.
1331/// An ordered-list enumeration marker at the start of a list item: leading ASCII
1332/// digits followed by `.`, e.g. `1. Undo/Redo` → `(1, "Undo/Redo")`. Returns
1333/// `None` when the text doesn't start with `digits.`.
1334/// docling's `ListItemMarkerProcessor` bullet patterns
1335/// (`docling/models/postprocessing/list_marker_processor.py`), one glyph each.
1336const LIST_BULLET_MARKERS: &str = "\u{2022}\u{2023}\u{25E6}\u{2043}\u{204C}\u{204D}\u{2219}\u{25AA}\u{25AB}\u{25CF}\u{25CB}-*+•·‣⁃►▶▸➤➢✓✔✗✘";
1337
1338/// docling's numbered-marker patterns as byte-length scanners over the start
1339/// of the text, in its first-wins order (the compound ones first, as they are
1340/// the more specific). Each returns the marker's candidate lengths, longest
1341/// (greedy) first — the alternatives Python's regex would backtrack through
1342/// before the `\s(.+)` tail. Hand-rolled rather than `regex` because this file
1343/// is part of the `pdf-text` (wasm) build, where the `regex` crate is not.
1344/// `\d` is Unicode-aware in Python, hence `char::is_numeric`; the letters are
1345/// ASCII classes in both.
1346const LIST_NUMBERED_MARKERS: &[fn(&str) -> Vec<usize>] = &[
1347    // `\d+(?:\.\d+)+\.?` — 1.1  1.2.3  1.1.
1348    |s| {
1349        let mut i = digits(s, 0);
1350        if i == 0 {
1351            return Vec::new();
1352        }
1353        let mut groups = 0;
1354        while s[i..].starts_with('.') && digits(s, i + 1) > i + 1 {
1355            i = digits(s, i + 1);
1356            groups += 1;
1357        }
1358        if groups == 0 {
1359            return Vec::new();
1360        }
1361        if s[i..].starts_with('.') {
1362            vec![i + 1, i]
1363        } else {
1364            vec![i]
1365        }
1366    },
1367    // `\d+\.?[a-zA-Z]\.` — 9a. 3.a.
1368    |s| digits_dot_letter(s, 0, '.').into_iter().collect(),
1369    // `\d+\.?[a-zA-Z]\)` — 9a) 3.a)
1370    |s| digits_dot_letter(s, 0, ')').into_iter().collect(),
1371    // `\(\d+\.?[a-zA-Z]\)` — (9a) (3.a)
1372    |s| {
1373        if !s.starts_with('(') {
1374            return Vec::new();
1375        }
1376        digits_dot_letter(s, 1, ')').into_iter().collect()
1377    },
1378    // `\d+\.` — 1. 2. 3.
1379    |s| digits_then(s, 0, '.').into_iter().collect(),
1380    // `\d+\)` — 1) 2) 3)
1381    |s| digits_then(s, 0, ')').into_iter().collect(),
1382    // `\(\d+\)` — (1) (2) (3)
1383    |s| {
1384        if !s.starts_with('(') {
1385            return Vec::new();
1386        }
1387        digits_then(s, 1, ')').into_iter().collect()
1388    },
1389    // `\[\d+\]` — [1] [2] [3]
1390    |s| {
1391        if !s.starts_with('[') {
1392            return Vec::new();
1393        }
1394        digits_then(s, 1, ']').into_iter().collect()
1395    },
1396    // `[ivxlcdm]+\.` — i. ii. iii.
1397    |s| class_run_then(s, "ivxlcdm", '.').into_iter().collect(),
1398    // `[IVXLCDM]+\.` — I. II. III.
1399    |s| class_run_then(s, "IVXLCDM", '.').into_iter().collect(),
1400    // `[a-z]\.` / `[A-Z]\.` / `[a-z]\)` / `[A-Z]\)`
1401    |s| {
1402        letter_then(s, char::is_ascii_lowercase, '.')
1403            .into_iter()
1404            .collect()
1405    },
1406    |s| {
1407        letter_then(s, char::is_ascii_uppercase, '.')
1408            .into_iter()
1409            .collect()
1410    },
1411    |s| {
1412        letter_then(s, char::is_ascii_lowercase, ')')
1413            .into_iter()
1414            .collect()
1415    },
1416    |s| {
1417        letter_then(s, char::is_ascii_uppercase, ')')
1418            .into_iter()
1419            .collect()
1420    },
1421];
1422
1423/// Byte offset just past the run of `\d` characters starting at `from`
1424/// (`from` itself when there is none).
1425fn digits(s: &str, from: usize) -> usize {
1426    s[from..]
1427        .char_indices()
1428        .find(|(_, c)| !c.is_numeric())
1429        .map_or(s.len(), |(i, _)| from + i)
1430}
1431
1432/// `\d+<close>` from `from`: the length through `close`, if it matches.
1433fn digits_then(s: &str, from: usize, close: char) -> Option<usize> {
1434    let end = digits(s, from);
1435    (end > from && s[end..].starts_with(close)).then(|| end + close.len_utf8())
1436}
1437
1438/// `\d+\.?[a-zA-Z]<close>` from `from`.
1439fn digits_dot_letter(s: &str, from: usize, close: char) -> Option<usize> {
1440    let mut i = digits(s, from);
1441    if i == from {
1442        return None;
1443    }
1444    if s[i..].starts_with('.') {
1445        i += 1;
1446    }
1447    let letter = s[i..].chars().next().filter(char::is_ascii_alphabetic)?;
1448    i += letter.len_utf8();
1449    s[i..].starts_with(close).then(|| i + close.len_utf8())
1450}
1451
1452/// `[<class>]+<close>` at the start.
1453fn class_run_then(s: &str, class: &str, close: char) -> Option<usize> {
1454    let end = s
1455        .char_indices()
1456        .find(|(_, c)| !class.contains(*c))
1457        .map_or(s.len(), |(i, _)| i);
1458    (end > 0 && s[end..].starts_with(close)).then(|| end + close.len_utf8())
1459}
1460
1461/// `[<letter class>]<close>` at the start.
1462fn letter_then(s: &str, class: fn(&char) -> bool, close: char) -> Option<usize> {
1463    let letter = s.chars().next().filter(class)?;
1464    let i = letter.len_utf8();
1465    s[i..].starts_with(close).then(|| i + close.len_utf8())
1466}
1467
1468/// docling's `ListItemMarkerProcessor.process_list_item`: the item's original
1469/// text is matched against `^(marker)\s(.+)` (DOTALL) for the bullet patterns,
1470/// then the numbered ones in order; a hit splits it into
1471/// `(marker, text, enumerated)`. `\s` is one Unicode whitespace character and
1472/// `.+` everything after it, which must be non-empty.
1473fn split_list_marker(text: &str) -> Option<(&str, &str, bool)> {
1474    let tail_after = |len: usize| -> Option<&str> {
1475        let ws = text[len..].chars().next()?;
1476        if !ws.is_whitespace() {
1477            return None;
1478        }
1479        let rest = &text[len + ws.len_utf8()..];
1480        (!rest.is_empty()).then_some(rest)
1481    };
1482    let first = text.chars().next()?;
1483    if LIST_BULLET_MARKERS.contains(first) {
1484        if let Some(rest) = tail_after(first.len_utf8()) {
1485            return Some((&text[..first.len_utf8()], rest, false));
1486        }
1487    }
1488    for matcher in LIST_NUMBERED_MARKERS {
1489        for len in matcher(text) {
1490            if let Some(rest) = tail_after(len) {
1491                return Some((&text[..len], rest, true));
1492            }
1493        }
1494    }
1495    None
1496}
1497
1498/// A PDF `list_item` region as docling emits it: `ListItemMarkerProcessor`
1499/// splits the marker off (see [`split_list_marker`]), and docling-core's
1500/// Markdown list serializer (default `orig_list_item_marker_mode = AUTO`,
1501/// `ensure_valid_list_item_marker`) prints it as `N. text` for an `N.` marker
1502/// (`case_already_valid`: the marker verbatim); as `- text` for a bullet glyph
1503/// (no letter or digit in the marker: only the `-` the serializer adds); and
1504/// as `- a) text` / `- 1.2 text` / `- [3] text` for any other marker
1505/// (`case_auto`: the serializer's `-`, then the original marker) — spelled
1506/// here as a bullet item whose text carries the marker, the way the DOCX and
1507/// DOC backends already spell theirs. An item without a recognizable marker is
1508/// a plain bullet. The symbol-font bullets docling-parse filters out of its
1509/// cells (`•◦▪·*` glued to the text) are stripped before the match, as before.
1510fn list_item_node(text: &str, loc: [u16; 4], first_in_list: bool) -> Node {
1511    // docling's match runs on the text as docling-parse hands it over; the
1512    // glued symbol-font bullets it never sees are stripped only when the raw
1513    // text matches no marker (`•Text` → `Text`, but `• Text` → marker `•`).
1514    let stripped = text
1515        .trim_start_matches(['•', '◦', '▪', '·', '*'])
1516        .trim_start();
1517    let split = split_list_marker(text).or_else(|| split_list_marker(stripped));
1518    let bullet = |text: String, marker: &str| Node::ListItem {
1519        ordered: false,
1520        number: 0,
1521        first_in_list,
1522        text: md_escape(&text),
1523        level: 0,
1524        // docling keeps the marker as the DocLang list marker
1525        // (`<ldiv><marker>·</marker></ldiv>`); Markdown prints its own `-`.
1526        marker: Some(marker.to_string()),
1527        location: Some(loc),
1528        dclx: None,
1529        href: None,
1530        layer: None,
1531    };
1532    // docling-core's `case_already_valid` is a *full* `\d+\.` match: `3.a.`
1533    // and `1.2` are `case_auto` markers (`- 3.a. text`), not item numbers.
1534    let is_number_dot = |m: &str| {
1535        m.strip_suffix('.')
1536            .is_some_and(|d| !d.is_empty() && d.chars().all(char::is_numeric))
1537    };
1538    match split {
1539        Some((marker, body, true)) if is_number_dot(marker) => {
1540            let number = parse_ordered_marker(marker).map_or(0, |(n, _)| n);
1541            Node::ListItem {
1542                ordered: true,
1543                number,
1544                first_in_list,
1545                text: md_escape(body),
1546                level: 0,
1547                marker: Some(marker.to_string()),
1548                location: Some(loc),
1549                dclx: None,
1550                href: None,
1551                layer: None,
1552            }
1553        }
1554        // `case_auto`: a marker holding a letter or digit rides in the text.
1555        Some((marker, body, true)) => bullet(format!("{marker} {body}"), marker),
1556        Some((marker, body, false)) => bullet(body.to_string(), marker),
1557        None => bullet(stripped.to_string(), "·"),
1558    }
1559}
1560
1561fn parse_ordered_marker(s: &str) -> Option<(u64, String)> {
1562    let digits: String = s.chars().take_while(|c| c.is_ascii_digit()).collect();
1563    if digits.is_empty() {
1564        return None;
1565    }
1566    let rest = s[digits.len()..].strip_prefix('.')?;
1567    let number = digits.parse().ok()?;
1568    Some((number, rest.trim_start().to_string()))
1569}
1570
1571/// Escape markdown special characters the way docling-core's markdown serializer
1572/// does (`markdown.py` post_process): `_` → `\_`, then HTML-escape `&`, `<`, `>`
1573/// (quote=False, so quotes are left). Applied to prose (headings, list items,
1574/// paragraphs); code blocks, the formula placeholder, and table cells are left raw.
1575fn md_escape(text: &str) -> String {
1576    text.replace('_', "\\_")
1577        .replace('&', "&amp;")
1578        .replace('<', "&lt;")
1579        .replace('>', "&gt;")
1580}
1581
1582fn clean_text(text: &str) -> String {
1583    // Typographic-quote normalization follows docling-parse's sanitizer table
1584    // (`pdf_sanitators/constants.h`): every curly quote — single *and double* —
1585    // becomes the ASCII apostrophe `'`, and `‚` a comma. A `"` in docling's
1586    // output only ever comes from a literal `quotedbl` glyph, never from `“ ”`
1587    // (2206's `'text in the wild"` pairs a curly open with a literal-quote
1588    // close). This replaces an earlier Hangul-only special case that patched
1589    // one symptom of mapping `“ ”` to `"`.
1590    let replaced = text
1591        .replace("\u{2} ", "")
1592        .replace("\u{ad} ", "")
1593        .replace(['\u{2}', '\u{ad}'], "") // any stray wrap hyphens not at a join
1594        .replace(
1595            [
1596                '\u{2018}', '\u{2019}', '\u{201b}', '\u{201c}', '\u{201d}', '\u{201e}', '\u{201f}',
1597            ],
1598            "'",
1599        ) // ‘ ’ ‛ “ ” „ ‟ → '
1600        .replace('\u{201a}', ",") // ‚ → ,
1601        .replace(
1602            [
1603                '\u{2010}', '\u{2011}', '\u{2012}', '\u{2013}', '\u{2014}', '\u{2015}', '\u{2212}',
1604            ],
1605            "-",
1606        ) // hyphen/dash family → -
1607        .replace('\u{2044}', "/") // ⁄ fraction slash → /
1608        .replace('\u{2022}', "\u{b7}") // • → · (docling never emits •; inline CCS-concept separators)
1609        .replace('\u{2026}', "..."); // … → ...
1610                                     // The docling-parse sanitizer already placed the correct spacing (e.g.
1611                                     // justified double spaces); preserve internal runs of spaces, only
1612                                     // normalizing line breaks/tabs and trimming the ends.
1613    let out = replaced.replace(['\n', '\r', '\t'], " ").trim().to_string();
1614    fix_arabic_lam_alef(&out)
1615}
1616
1617/// pdfium decomposes the Arabic lam-alef ligature (لا / لإ / لأ / لآ) into its
1618/// glyph constituents in *visual* order — `alef-variant, lam` — but docling keeps
1619/// logical order, `lam, alef-variant`. Swap a mid-word `alef-variant + lam` back
1620/// to `lam + alef-variant`. "Mid-word" (the previous char is an Arabic letter)
1621/// distinguishes the ligature from the definite article `ال` (word-initial
1622/// `alef + lam`), which must stay. No-op for non-Arabic text.
1623fn fix_arabic_lam_alef(s: &str) -> String {
1624    let is_arabic_letter = |c: char| ('\u{0620}'..='\u{064A}').contains(&c);
1625    let chars: Vec<char> = s.chars().collect();
1626    if !chars.iter().any(|&c| is_arabic_letter(c)) {
1627        return s.to_string(); // no-op for non-Arabic text
1628    }
1629    // Pass 1: swap mid-word `alef-variant + lam` → `lam + alef-variant`. Only the
1630    // hamza/madda alef variants (إ أ آ) are safe: the definite article is always
1631    // plain `ا + ل`, so plain `alef + lam` is ambiguous (a legitimate `فعالة` vs a
1632    // reversed `لا` ligature look identical) — leaving plain alef alone avoids
1633    // corrupting legitimate words.
1634    let mut a: Vec<char> = Vec::with_capacity(chars.len());
1635    let mut i = 0;
1636    while i < chars.len() {
1637        let c = chars[i];
1638        if matches!(c, '\u{0622}' | '\u{0623}' | '\u{0625}')
1639            && chars.get(i + 1) == Some(&'\u{0644}')
1640            && i > 0
1641            && is_arabic_letter(chars[i - 1])
1642            // A preceding lam means this alef-variant is *already* the logical
1643            // `lam + alef` ligature; the following lam is the next syllable's
1644            // letter, not a reversed ligature — swapping it corrupts `لآل` → `للآ`
1645            // (e.g. التعلم الآلي → الآلي, not اللآي).
1646            && chars[i - 1] != '\u{0644}'
1647        {
1648            a.push('\u{0644}');
1649            a.push(c);
1650            i += 2;
1651            continue;
1652        }
1653        a.push(c);
1654        i += 1;
1655    }
1656    // Pass 2: insert a space at Arabic↔Latin boundaries (bidi script switch) that
1657    // pdfium runs together — docling separates the embedded Latin run (`وPython`
1658    // → `و Python`).
1659    let mut out: Vec<char> = Vec::with_capacity(a.len());
1660    for (j, &c) in a.iter().enumerate() {
1661        if j > 0 {
1662            let p = a[j - 1];
1663            if (is_arabic_letter(p) && c.is_ascii_alphabetic())
1664                || (p.is_ascii_alphabetic() && is_arabic_letter(c))
1665            {
1666                out.push(' ');
1667            }
1668        }
1669        out.push(c);
1670    }
1671    out.into_iter().collect()
1672}
1673
1674/// docling's `PageAssembleModel._match_hyperlink`: the URI whose link
1675/// annotations cover at least half of the region's box, or `None`. Coverage is
1676/// intersection-over-region-area, **accumulated per URI** — a URL that wraps
1677/// across lines carries several annotation rects that sum toward the same
1678/// target. Ties resolve to the first-seen URI (Python's `max` over dict
1679/// insertion order); the winner still needs `>= 0.5`
1680/// (`_HYPERLINK_COVERAGE_THRESHOLD`).
1681pub(crate) fn region_hyperlink(
1682    region: &Region,
1683    links: &[crate::pdfium_backend::LinkAnnot],
1684) -> Option<String> {
1685    if links.is_empty() {
1686        return None;
1687    }
1688    let area = (region.r - region.l).max(0.0) * (region.b - region.t).max(0.0);
1689    if area <= 0.0 {
1690        return None;
1691    }
1692    let mut coverage: Vec<(&str, f32)> = Vec::new();
1693    for link in links {
1694        let ix = (region.r.min(link.r) - region.l.max(link.l)).max(0.0);
1695        let iy = (region.b.min(link.b) - region.t.max(link.t)).max(0.0);
1696        let c = ix * iy / area;
1697        match coverage.iter_mut().find(|(uri, _)| *uri == link.uri) {
1698            Some((_, acc)) => *acc += c,
1699            None => coverage.push((&link.uri, c)),
1700        }
1701    }
1702    let mut best: Option<(&str, f32)> = None;
1703    for (uri, c) in coverage {
1704        // Strictly greater keeps the first-seen URI on ties, like Python's max.
1705        if best.is_none_or(|(_, bc)| c > bc) {
1706            best = Some((uri, c));
1707        }
1708    }
1709    let (uri, c) = best?;
1710    (c >= 0.5).then(|| normalize_uri(uri))
1711}
1712
1713/// The pydantic-`AnyUrl` normalization docling's hyperlink value passes
1714/// through on its way to the serializer: a URL with an authority but no path
1715/// gains a trailing `/` (`https://arxiv.org` → `https://arxiv.org/`). Other
1716/// AnyUrl canonicalizations (scheme/host lowercasing, percent-encoding) don't
1717/// occur in PDF link annotations in practice, so they are not reproduced.
1718fn normalize_uri(uri: &str) -> String {
1719    if let Some((_, rest)) = uri.split_once("://") {
1720        if !rest.is_empty() && !rest.contains(['/', '?', '#']) {
1721            return format!("{uri}/");
1722        }
1723    }
1724    uri.to_string()
1725}
1726
1727/// Resolve each page hyperlink to the visible text it covers, as `(anchor, uri)`
1728/// in reading order. The anchor is the cells whose centre falls in the link rect,
1729/// joined left-to-right and cleaned the same way prose is (so it matches the
1730/// serialized text), deduped against the immediately-preceding link so pdfium's
1731/// occasional duplicate annotation doesn't double-list. Empty anchors are dropped.
1732pub(crate) fn resolve_link_anchors(page: &PdfPage) -> Vec<(String, String)> {
1733    let mut out: Vec<(String, String)> = Vec::new();
1734    // Use per-word cells, not the line-merged `cells`: a link rect covers a few
1735    // words on a line, and a whole merged line cell would over-capture (its centre
1736    // lands in one link's rect, grabbing the entire line as that link's anchor).
1737    let words = if page.word_cells.is_empty() {
1738        &page.cells
1739    } else {
1740        &page.word_cells
1741    };
1742    for link in &page.links {
1743        // A cell participates when its centre row is inside the rect and it
1744        // overlaps the rect horizontally. A cell can be *wider* than the rect:
1745        // PDFs often draw a whole header line as one text run ("LinkedIn |
1746        // GitHub | Credly"), which docling-parse's word grouping keeps as one
1747        // cell even though each label carries its own link annotation —
1748        // centre-in-rect alone would hand the entire line to every link.
1749        // [`cell_text_in_rect`] clips such a cell to the tokens under the rect.
1750        let mut inside: Vec<(&TextCell, String)> = words
1751            .iter()
1752            .filter(|c| {
1753                let cy = (c.t + c.b) / 2.0;
1754                cy >= link.t && cy <= link.b && c.r.min(link.r) > c.l.max(link.l)
1755            })
1756            .filter_map(|c| {
1757                let text = cell_text_in_rect(c, link.l, link.r);
1758                (!text.is_empty()).then_some((c, text))
1759            })
1760            .collect();
1761        // Reading order: top band then left-to-right (link anchors are LTR).
1762        let band = inside
1763            .iter()
1764            .map(|(c, _)| (c.b - c.t).abs())
1765            .fold(0.0f32, f32::max)
1766            .max(1.0);
1767        inside.sort_by_key(|(c, _)| ((c.t / band).round() as i64, (c.l * 10.0) as i64));
1768        let anchor = clean_text(
1769            &inside
1770                .iter()
1771                .map(|(_, t)| t.trim())
1772                .filter(|t| !t.is_empty())
1773                .collect::<Vec<_>>()
1774                .join(" "),
1775        );
1776        if anchor.is_empty() {
1777            continue;
1778        }
1779        if out
1780            .last()
1781            .is_some_and(|(a, u)| a == &anchor && u == &link.uri)
1782        {
1783            continue;
1784        }
1785        out.push((anchor, link.uri.clone()));
1786    }
1787    out
1788}
1789
1790/// The part of a cell's text that lies under a link rect's x-range. A cell
1791/// fully inside the rect (by centre) returns its whole text. A wider cell is
1792/// split into whitespace tokens whose x-spans are estimated proportionally to
1793/// their character positions (kerning makes this approximate, so selection
1794/// snaps to whole tokens, never characters); tokens whose estimated centre
1795/// falls inside the rect are kept. Returns "" when nothing falls inside.
1796fn cell_text_in_rect(c: &TextCell, l: f32, r: f32) -> String {
1797    let cx = (c.l + c.r) / 2.0;
1798    if cx >= l && cx <= r && c.l >= l - (c.r - c.l) * 0.25 && c.r <= r + (c.r - c.l) * 0.25 {
1799        return c.text.trim().to_string();
1800    }
1801    let chars: Vec<char> = c.text.chars().collect();
1802    let n = chars.len();
1803    if n == 0 || c.r <= c.l {
1804        return String::new();
1805    }
1806    let per = (c.r - c.l) / n as f32;
1807    let mut out: Vec<String> = Vec::new();
1808    let mut token = String::new();
1809    let mut start = 0usize;
1810    // A trailing sentinel space flushes the last token.
1811    for (i, &ch) in chars.iter().enumerate().chain(std::iter::once((n, &' '))) {
1812        if ch.is_whitespace() {
1813            if !token.is_empty() {
1814                let mid = c.l + (start as f32 + (i - start) as f32 / 2.0) * per;
1815                if mid >= l && mid <= r {
1816                    out.push(std::mem::take(&mut token));
1817                } else {
1818                    token.clear();
1819                }
1820            }
1821        } else {
1822            if token.is_empty() {
1823                start = i;
1824            }
1825            token.push(ch);
1826        }
1827    }
1828    out.join(" ")
1829}
1830
1831/// Cells assigned to a region (best container), in reading order, joined.
1832fn region_text(region: &Region, cells: &[TextCell]) -> String {
1833    let inside: Vec<&TextCell> = cells
1834        .iter()
1835        .filter(|c| {
1836            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1837            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1838        })
1839        .collect();
1840    cells_text(inside)
1841}
1842
1843/// docling's exclusive cell assignment (`_assign_cells_to_clusters`): every
1844/// non-empty cell goes to the single best-overlapping *regular* region at
1845/// intersection-over-self > 0.2, and each region serializes exactly its
1846/// assigned cells. A cell under two overlapping boxes is emitted once (by the
1847/// better-covering one), and a cell only partially under its region — e.g.
1848/// normal_4pages' big section numeral, ~30 % inside the heading box — still
1849/// joins it (`## 들어가며 1`) instead of leaking as an orphan. Pictures and
1850/// wrappers never claim (docling walks regular clusters only); ties go to the
1851/// first region, like docling's strict `>` best-overlap scan.
1852pub fn region_texts_exclusive(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
1853    let owned = assign_cells(regions, cells);
1854    // Non-claimers (tables/wrappers/pictures) keep the inclusive > 0.5 text:
1855    // docling fills a special cluster's cells from its contained children, and
1856    // downstream table assembly gates on that text being non-empty.
1857    regions
1858        .iter()
1859        .zip(owned)
1860        .map(|(r, cs)| {
1861            if claims_cells(r) {
1862                cells_text(cs.iter().map(|&i| &cells[i]).collect())
1863            } else {
1864                region_text(r, cells)
1865            }
1866        })
1867        .collect()
1868}
1869
1870/// A *regular* region in docling's sense — one that claims cells. Pictures and
1871/// the wrappers (`table`, `document_index`, `form`, `key_value_region`) fill
1872/// their cells from contained children instead.
1873fn claims_cells(r: &Region) -> bool {
1874    r.label != "picture" && !is_wrapper(r.label)
1875}
1876
1877/// docling's `_assign_cells_to_clusters`: each non-empty cell's index goes to
1878/// the single best-overlapping regular region at intersection-over-self > 0.2
1879/// (ties to the first region, like docling's strict `>` scan). One entry per
1880/// region, in region order.
1881fn assign_cells(regions: &[Region], cells: &[TextCell]) -> Vec<Vec<usize>> {
1882    let mut owned: Vec<Vec<usize>> = vec![Vec::new(); regions.len()];
1883    for (ci, c) in cells.iter().enumerate() {
1884        if c.text.trim().is_empty() {
1885            continue;
1886        }
1887        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1888        let mut best: Option<(usize, f32)> = None;
1889        for (i, r) in regions.iter().enumerate() {
1890            if !claims_cells(r) {
1891                continue;
1892            }
1893            let ov = inter(r, c.l, c.t, c.r, c.b) / ca;
1894            if ov > 0.2 && best.is_none_or(|(_, b)| ov > b) {
1895                best = Some((i, ov));
1896            }
1897        }
1898        if let Some((i, _)) = best {
1899            owned[i].push(ci);
1900        }
1901    }
1902    owned
1903}
1904
1905/// docling's regular-cluster refinement after cell assignment
1906/// (`LayoutPostprocessor._process_regular_clusters`, #419), run once the page's
1907/// cells are final and before reading order:
1908///
1909/// 1. every regular region's box becomes the union of the cells it claimed
1910///    (`_adjust_cluster_bboxes` — a regular cluster's bbox *is* its cells'
1911///    bbox; a table's is the union with the model box, and pictures keep
1912///    theirs, so neither is touched here);
1913/// 2. a regular region that claimed no cell is dropped (`keep_empty_clusters`
1914///    is off; a `formula` is kept, as upstream keeps it);
1915/// 3. an orphan text region (`score == 0.0`, from [`add_orphan_regions`]) that
1916///    now sits > 0.8 inside another regular region's fitted box is folded into
1917///    it (`_remove_overlapping_clusters` at containment 0.8, the larger box
1918///    winning the group) — up to three rounds, like upstream's loop.
1919///
1920/// Why it matters: the layout model's box can end partway through a line. That
1921/// line fails the 0.2 claim and becomes an orphan — recoverable — but the
1922/// *model* box still overlaps the orphan's line by a few points, so the
1923/// reading-order graph, which links only strictly-above pairs, gets no edge
1924/// between them and may emit the next paragraph first, stranding the line
1925/// after the paragraph it belongs in (1540 of 6050 text blocks on the #419
1926/// book began mid-sentence). Fitted to its cells, the box ends on a line
1927/// boundary and the orphan slots in between; an orphan the fitted box
1928/// swallows joins the paragraph outright. Cell assignment is untouched: a
1929/// region's fitted box contains every cell it claimed, so
1930/// [`region_texts_exclusive`] hands it the same cells afterwards.
1931///
1932/// A page with no cells yet (a scan before OCR) is left alone: dropping every
1933/// text region for want of cells would be wrong, and the OCR paths call this
1934/// again once the cells exist.
1935pub fn fit_regions_to_cells(regions: &mut Vec<Region>, cells: &[TextCell]) {
1936    if !cells.iter().any(|c| !c.text.trim().is_empty()) {
1937        return;
1938    }
1939    for _ in 0..3 {
1940        let owned = assign_cells(regions, cells);
1941        let mut fitted: Vec<Region> = Vec::with_capacity(regions.len());
1942        for (r, own) in regions.iter().zip(&owned) {
1943            if !claims_cells(r) {
1944                fitted.push(r.clone());
1945                continue;
1946            }
1947            if own.is_empty() {
1948                if r.label == "formula" {
1949                    fitted.push(r.clone());
1950                }
1951                continue;
1952            }
1953            let mut f = r.clone();
1954            f.l = own
1955                .iter()
1956                .map(|&i| cells[i].l)
1957                .fold(f32::INFINITY, f32::min);
1958            f.t = own
1959                .iter()
1960                .map(|&i| cells[i].t)
1961                .fold(f32::INFINITY, f32::min);
1962            f.r = own
1963                .iter()
1964                .map(|&i| cells[i].r)
1965                .fold(f32::NEG_INFINITY, f32::max);
1966            f.b = own
1967                .iter()
1968                .map(|&i| cells[i].b)
1969                .fold(f32::NEG_INFINITY, f32::max);
1970            fitted.push(f);
1971        }
1972        let mut changed = fitted.len() != regions.len();
1973        // Fold orphans into the regular region whose fitted box holds them.
1974        let mut drop = vec![false; fitted.len()];
1975        for i in 0..fitted.len() {
1976            let o = &fitted[i];
1977            if !(o.score == 0.0 && o.label == "text") {
1978                continue;
1979            }
1980            let oa = area(o.l, o.t, o.r, o.b).max(1.0);
1981            let mut best: Option<(usize, f32)> = None;
1982            for (j, r) in fitted.iter().enumerate() {
1983                if j == i || drop[j] || r.score == 0.0 || !claims_cells(r) {
1984                    continue;
1985                }
1986                let ov = inter(r, o.l, o.t, o.r, o.b) / oa;
1987                if ov > 0.8 && best.is_none_or(|(_, b)| ov > b) {
1988                    best = Some((j, ov));
1989                }
1990            }
1991            if let Some((j, _)) = best {
1992                let (l, t, r, b) = (o.l, o.t, o.r, o.b);
1993                let host = &mut fitted[j];
1994                host.l = host.l.min(l);
1995                host.t = host.t.min(t);
1996                host.r = host.r.max(r);
1997                host.b = host.b.max(b);
1998                drop[i] = true;
1999                changed = true;
2000            }
2001        }
2002        let mut drop = drop.into_iter();
2003        fitted.retain(|_| !drop.next().expect("aligned"));
2004        *regions = fitted;
2005        if !changed {
2006            break;
2007        }
2008    }
2009}
2010
2011/// Join a prefiltered cell list into the region's text (docling's
2012/// `sanitize_text` over the sanitizer's cell order).
2013fn cells_text(inside: Vec<&TextCell>) -> String {
2014    // docling orders a cluster's cells by their docling-parse cell index
2015    // alone (`LayoutPostprocessor._sort_cells`: `sorted(cells, key=c.index)`)
2016    // — the sanitizer's output order, which our `cells` slice already is.
2017    // No geometric re-sort: normal_4pages' big section numerals paint
2018    // *after* their heading text, and docling's `## 들어가며 1` (numeral
2019    // last) only falls out of pure index order — a band sort dragged the
2020    // numeral to the front. The overlap-grouped line restore this replaced
2021    // measured strictly worse on the corpus (it fixed nothing the index
2022    // order broke, and broke the numerals).
2023    let joined = {
2024        // docling's `PageAssembleModel.sanitize_text`, ported verbatim over the
2025        // parse-index-ordered lines: append a separating space to a line —
2026        // unless it ends with `-`. A dash-ending line whose last word and the
2027        // next line's first word are both alphanumeric is a wrapped word: the
2028        // dash is dropped and the lines fuse (`platforms-` + `reflects` →
2029        // `platformsreflects`, `pp. 545-` + `561` → `545561`). Any other
2030        // dash-ending line — e.g. the *bare* `-` cell a superscript ORCID or an
2031        // inline `–` bullet splits off (its word list is empty, so the fuse
2032        // test fails) — keeps its dash and still takes no trailing space:
2033        // `[0000` `-` `0002` joins as docling's `[0000 -0002`, and the OTSL
2034        // list's `-` + `"C" cell -` + `a new table cell` collapses to
2035        // `-"C" cell a new table cell`. Our cells still carry the raw dash
2036        // family (docling-parse normalizes to `-` before this; clean_text does
2037        // it after), so the endswith test matches them all.
2038        let texts: Vec<&str> = inside
2039            .iter()
2040            .map(|c| c.text.trim())
2041            // Skip whitespace-only cells (a justified line's trailing space
2042            // glyph): an empty line would double the separator.
2043            .filter(|t| !t.is_empty())
2044            .collect();
2045        let last_word_alnum = |s: &str| {
2046            s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
2047                .rfind(|w| !w.is_empty())
2048                .is_some_and(|w| w.chars().all(char::is_alphanumeric))
2049        };
2050        let first_word_alnum = |s: &str| {
2051            s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
2052                .find(|w| !w.is_empty())
2053                .is_some_and(|w| w.chars().all(char::is_alphanumeric))
2054        };
2055        let mut out = String::new();
2056        for (i, t) in texts.iter().enumerate() {
2057            if i > 0 {
2058                let prev = texts[i - 1];
2059                let dashish = matches!(
2060                    prev.chars().last(),
2061                    Some(
2062                        '-' | '\u{2010}'
2063                            | '\u{2011}'
2064                            | '\u{2012}'
2065                            | '\u{2013}'
2066                            | '\u{2014}'
2067                            | '\u{2015}'
2068                            | '\u{2212}'
2069                    )
2070                );
2071                // docling#4052 (2.122): a dash only splits a word when it is
2072                // *attached* to one — the character before it is alphanumeric.
2073                // A dash that follows whitespace (a separator dash, a bullet
2074                // marker, a wrapped `-prefixed` token, the bare `-` cell an
2075                // ORCID splits off) is a literal character: it is kept and the
2076                // lines join with the ordinary space.
2077                let attached = prev.chars().rev().nth(1).is_some_and(char::is_alphanumeric);
2078                if dashish && attached {
2079                    if last_word_alnum(prev) && first_word_alnum(t) {
2080                        out.pop(); // wrapped word: fuse without the dash
2081                    }
2082                    // an attached dash never takes a separating space
2083                } else {
2084                    out.push(' ');
2085                }
2086            }
2087            out.push_str(t);
2088        }
2089        out
2090    };
2091    clean_text(&joined)
2092}
2093
2094/// Tighten the spaces pdfium leaves around tight punctuation in a code line
2095/// (`console .log` → `console.log`, `add (3 , 5)` → `add(3, 5)`), matching
2096/// docling-parse's source spacing.
2097fn tighten_code_punct(s: &str) -> String {
2098    s.replace(" .", ".")
2099        .replace(" ,", ",")
2100        .replace(" ;", ";")
2101        .replace(" )", ")")
2102        .replace(" (", "(")
2103}
2104
2105/// Assemble a **code** region's text with its line structure preserved.
2106///
2107/// Unlike [`region_text`] — which joins every cell with a single space, the right
2108/// thing for prose reflow — a code block's line breaks and indentation are
2109/// significant. The `code_cells` are already one physical source line each
2110/// (grouped space-glyph-only, so monospace runs keep their spacing), so this:
2111///
2112/// 1. groups the cells into vertical line bands and orders them top→bottom,
2113///    left→right;
2114/// 2. joins the lines with `\n` (rather than spaces), keeping the carriage
2115///    returns; and
2116/// 3. reconstructs each line's leading indentation from its left offset, in units
2117///    of the block's estimated monospace character width, so nesting survives.
2118///
2119/// Typography is normalized per line via [`clean_text`] (smart quotes, dashes,
2120/// ellipsis), which never merges lines. Returns an empty string if the region has
2121/// no code cells (the caller falls back to the prose text).
2122fn code_region_text(region: &Region, cells: &[TextCell]) -> String {
2123    let mut inside: Vec<&TextCell> = cells
2124        .iter()
2125        .filter(|c| {
2126            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2127            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
2128        })
2129        .filter(|c| !c.text.trim().is_empty())
2130        .collect();
2131    if inside.is_empty() {
2132        return String::new();
2133    }
2134
2135    // Quantize the top edge into ~line bands (like `region_text`), then order the
2136    // cells by band (top→bottom) and, within a band, by left edge.
2137    let band = inside
2138        .iter()
2139        .map(|c| (c.b - c.t).abs())
2140        .fold(0.0f32, f32::max)
2141        .max(1.0);
2142    let line_of = |c: &TextCell| (c.t / band).round() as i64;
2143    inside.sort_by_key(|c| (line_of(c), (c.l * 10.0) as i64));
2144
2145    // Estimate one monospace character's width (total ink width / total glyphs) to
2146    // convert a line's left offset into a count of leading spaces. Measured over
2147    // all lines so a single short line can't skew it.
2148    let (mut total_w, mut total_chars) = (0.0f32, 0usize);
2149    for c in &inside {
2150        let n = c.text.trim().chars().count();
2151        if n > 0 {
2152            total_w += (c.r - c.l).max(0.0);
2153            total_chars += n;
2154        }
2155    }
2156    let char_w = if total_chars > 0 {
2157        (total_w / total_chars as f32).max(1.0)
2158    } else {
2159        1.0
2160    };
2161    // The block's own left margin is the zero-indent baseline.
2162    let base_l = inside.iter().map(|c| c.l).fold(f32::INFINITY, f32::min);
2163
2164    let mut lines: Vec<String> = Vec::new();
2165    let mut cur: Option<i64> = None;
2166    for c in &inside {
2167        // Tighten pdfium's spaced punctuation per line (on the trimmed content, so
2168        // the reconstructed leading indentation is never nibbled).
2169        let text = tighten_code_punct(&clean_text(c.text.trim()));
2170        if Some(line_of(c)) == cur {
2171            // A second cell sharing this band (rare — e.g. split columns): keep it
2172            // on the same source line, separated by a space.
2173            if let Some(last) = lines.last_mut() {
2174                last.push(' ');
2175                last.push_str(&text);
2176            }
2177            continue;
2178        }
2179        let indent = ((c.l - base_l) / char_w).round().max(0.0) as usize;
2180        lines.push(format!("{}{}", " ".repeat(indent), text));
2181        cur = Some(line_of(c));
2182    }
2183    lines.join("\n")
2184}
2185
2186/// Reconstruct a table's grid geometrically from the text cells inside its
2187/// region: cluster cells into rows (by vertical centre) and columns (by clustered
2188/// left edges), then place each cell. A model-free stand-in for TableFormer that
2189/// recovers grid-aligned tables from the precise PDF text layer (it does not
2190/// resolve row/column spans).
2191pub fn reconstruct_table(region: &Region, cells: &[TextCell]) -> Vec<Vec<String>> {
2192    let mut inside: Vec<&TextCell> = cells
2193        .iter()
2194        .filter(|c| {
2195            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2196            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
2197        })
2198        .collect();
2199    if inside.is_empty() {
2200        return Vec::new();
2201    }
2202    inside.sort_by(|a, b| a.t.total_cmp(&b.t));
2203
2204    // Rows: consecutive cells whose vertical centre is within ~0.7 line height.
2205    let mut rows: Vec<(f32, Vec<&TextCell>)> = Vec::new();
2206    for c in &inside {
2207        let cyc = (c.t + c.b) / 2.0;
2208        let lh = (c.b - c.t).abs().max(1.0);
2209        if let Some((ryc, row)) = rows.last_mut() {
2210            if (cyc - *ryc).abs() < lh * 0.7 {
2211                row.push(c);
2212                continue;
2213            }
2214        }
2215        rows.push((cyc, vec![c]));
2216    }
2217
2218    // Columns: cluster left edges (merge those within a tolerance).
2219    let tol = {
2220        let mut hs: Vec<f32> = inside.iter().map(|c| (c.b - c.t).abs()).collect();
2221        hs.sort_by(f32::total_cmp);
2222        hs[hs.len() / 2].max(4.0) * 1.5
2223    };
2224    let mut lefts: Vec<f32> = inside.iter().map(|c| c.l).collect();
2225    lefts.sort_by(f32::total_cmp);
2226    let mut col_starts: Vec<f32> = Vec::new();
2227    for l in lefts {
2228        if col_starts.last().is_none_or(|&last| l - last > tol) {
2229            col_starts.push(l);
2230        }
2231    }
2232    let ncols = col_starts.len().max(1);
2233    let col_of = |l: f32| -> usize {
2234        col_starts
2235            .iter()
2236            .rposition(|&s| l + tol * 0.5 >= s)
2237            .unwrap_or(0)
2238            .min(ncols - 1)
2239    };
2240
2241    let mut grid = Vec::with_capacity(rows.len());
2242    for (_, mut row) in rows {
2243        row.sort_by(|a, b| a.l.total_cmp(&b.l));
2244        let mut cols = vec![String::new(); ncols];
2245        for c in row {
2246            let ci = col_of(c.l);
2247            // Strip the wrap-hyphen control char so it never lands in a cell.
2248            let t = c.text.trim().replace(['\u{2}', '\u{ad}'], "");
2249            if cols[ci].is_empty() {
2250                cols[ci] = t;
2251            } else {
2252                cols[ci].push(' ');
2253                cols[ci].push_str(&t);
2254            }
2255        }
2256        grid.push(cols);
2257    }
2258    grid
2259}
2260
2261/// Does the geometric reconstruction of a table look trustworthy enough to use
2262/// as-is, instead of paying for TableFormer?
2263///
2264/// [`reconstruct_table`] derives columns by clustering cell **left edges**. On a
2265/// clean grid that is exact, but when a column's entries are not left-aligned
2266/// (or the OCR boxes wobble) the clustering splits one real column into several,
2267/// and the result is a wide, mostly-empty grid — the "spurious empty columns"
2268/// failure TableFormer exists to fix.
2269///
2270/// Two symptoms separate the two cases, and both are properties of the grid
2271/// alone (no model needed):
2272/// * **density** — a real table is mostly full; a split-up one is mostly holes;
2273/// * **thin columns** — a column carrying at most one entry across several rows
2274///   is almost always a split artefact rather than a real column.
2275///
2276/// Deliberately conservative: it answers `true` only for grids that are plainly
2277/// well-formed, so the expensive path stays the default whenever there is doubt.
2278/// A caller that skips TableFormer on `true` trades no quality for the time.
2279pub fn geometric_table_is_reliable(rows: &[Vec<String>]) -> bool {
2280    let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
2281    // Fewer than two columns is not a grid this heuristic can vouch for: it is
2282    // exactly the shape a collapsed table takes, and TableFormer may recover
2283    // real structure from it.
2284    if rows.len() < 2 || ncols < 2 {
2285        return false;
2286    }
2287    let filled = |c: &String| !c.trim().is_empty();
2288    let total = rows.len() * ncols;
2289    let full = rows.iter().flatten().filter(|c| filled(c)).count();
2290    if (full as f32) < MIN_TABLE_FILL * total as f32 {
2291        return false;
2292    }
2293    // A column used by at most one row, when there are rows enough to tell.
2294    if rows.len() >= 3 {
2295        for ci in 0..ncols {
2296            let used = rows
2297                .iter()
2298                .filter(|r| r.get(ci).is_some_and(filled))
2299                .count();
2300            if used <= 1 {
2301                return false;
2302            }
2303        }
2304    }
2305    true
2306}
2307
2308/// Share of a geometric grid's cells that must carry text for it to be trusted
2309/// without TableFormer. Chosen well above the density a left-edge split
2310/// produces (those land nearer a third) and below what a genuine table with a
2311/// few blank cells reaches.
2312const MIN_TABLE_FILL: f32 = 0.6;
2313
2314/// The union bbox of the text cells assigned to a region (same >50%-overlap
2315/// rule as [`region_text`]), or `None` when no cell lands in it. docling's
2316/// LayoutPostprocessor shrinks a regular cluster's bbox to its cells, and the
2317/// enrichment crops are taken from that cell-tight box — cropping the raw
2318/// detector box instead hands the VLM surrounding chrome (e.g. the `Listing N:`
2319/// caption under a code block) that changes its output.
2320pub fn region_cell_bbox(region: &Region, cells: &[TextCell]) -> Option<[f32; 4]> {
2321    let mut bbox: Option<[f32; 4]> = None;
2322    for c in cells {
2323        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2324        if inter(region, c.l, c.t, c.r, c.b) / ca <= 0.5 {
2325            continue;
2326        }
2327        bbox = Some(match bbox {
2328            None => [c.l, c.t, c.r, c.b],
2329            Some([l, t, r, b]) => [l.min(c.l), t.min(c.t), r.max(c.r), b.max(c.b)],
2330        });
2331    }
2332    bbox
2333}
2334
2335/// One region's enrichment-model result, produced by the pipeline's opt-in
2336/// passes (issue #76) and applied during assembly.
2337#[derive(Debug, Clone)]
2338pub enum Enrichment {
2339    /// DocumentPictureClassifier predictions, descending confidence.
2340    PictureClasses(Vec<PictureClass>),
2341    /// CodeFormulaV2 output for a `code` region: the rewritten source text and
2342    /// the `<_language_>` prefix (when the model emitted one).
2343    Code {
2344        language: Option<String>,
2345        text: String,
2346    },
2347    /// CodeFormulaV2 output for a `formula` region: the decoded LaTeX.
2348    Formula { latex: String },
2349}
2350
2351/// Crop a region (page points, already expanded by the caller if needed) from
2352/// the rendered page image and resize it to `target_scale` pixels per point —
2353/// the enrichment-model equivalent of docling's
2354/// `page.get_image(scale=…, cropbox=…)`, sourced from the existing
2355/// [`crate::pdfium_backend::RENDER_SCALE`] render instead of a fresh pdfium
2356/// pass (the page bitmap is already the exact docling render at scale 2).
2357#[cfg(feature = "ml")]
2358pub fn crop_region_scaled(page: &PdfPage, bbox: [f32; 4], target_scale: f32) -> Option<RgbImage> {
2359    let s = page.scale;
2360    let [l, t, r, b] = bbox;
2361    let (iw, ih) = (page.image.width(), page.image.height());
2362    let x = (l * s).max(0.0) as u32;
2363    let y = (t * s).max(0.0) as u32;
2364    if x >= iw || y >= ih {
2365        return None;
2366    }
2367    let w = (((r - l.max(0.0)) * s) as u32).min(iw - x);
2368    let h = (((b - t.max(0.0)) * s) as u32).min(ih - y);
2369    if w == 0 || h == 0 {
2370        return None;
2371    }
2372    let crop = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
2373    // docling renders the crop at `target_scale` directly; from the scale-2
2374    // page render that is a resize to the same pixel geometry
2375    // (`round(width_points * scale)`, PIL's BICUBIC ≙ CatmullRom).
2376    let tw = ((w as f32 / s) * target_scale).round().max(1.0) as u32;
2377    let th = ((h as f32 / s) * target_scale).round().max(1.0) as u32;
2378    if (tw, th) == (w, h) {
2379        return Some(crop);
2380    }
2381    Some(image::imageops::resize(
2382        &crop,
2383        tw,
2384        th,
2385        image::imageops::FilterType::CatmullRom,
2386    ))
2387}
2388
2389/// Resample `img` (rendered at `from` px/pt, covering `w_pt`×`h_pt` points)
2390/// to `to` px/pt — docling's `round(points * scale)` pixel geometry, PIL's
2391/// BICUBIC ≙ CatmullRom. Unchanged when the geometry already matches.
2392#[cfg(feature = "ocr-prep")]
2393fn rescale(img: RgbImage, w_pt: f32, h_pt: f32, to: f32) -> RgbImage {
2394    let tw = (w_pt * to).round().max(1.0) as u32;
2395    let th = (h_pt * to).round().max(1.0) as u32;
2396    if (tw, th) == img.dimensions() {
2397        return img;
2398    }
2399    image::imageops::resize(&img, tw, th, image::imageops::FilterType::CatmullRom)
2400}
2401
2402/// Encode `img` as a PNG [`PictureImage`] rendered at `scale` px/pt.
2403#[cfg(feature = "ocr-prep")]
2404fn png_image(img: &RgbImage, scale: f32) -> Option<PictureImage> {
2405    let mut buf = std::io::Cursor::new(Vec::new());
2406    img.write_to(&mut buf, image::ImageFormat::Png).ok()?;
2407    Some(PictureImage {
2408        mimetype: "image/png".into(),
2409        width: img.width(),
2410        height: img.height(),
2411        data: buf.into_inner(),
2412        dpi: PictureImage::dpi_for_scale(scale),
2413    })
2414}
2415
2416/// The whole page render as docling's `PageItem.image` (#520): at `scale`
2417/// px/pt (`None` = the render's own), `None` when the page has no bitmap.
2418#[cfg(feature = "ocr-prep")]
2419pub fn page_image(page: &PdfPage, scale: Option<f32>) -> Option<PictureImage> {
2420    if page.image.width() == 0 || page.image.height() == 0 || page.scale <= 0.0 {
2421        return None;
2422    }
2423    let scale = scale.unwrap_or(page.scale);
2424    let img = rescale(page.image.clone(), page.width, page.height, scale);
2425    png_image(&img, scale)
2426}
2427
2428/// Crop a layout region from the rendered page image and encode it as PNG (the
2429/// figure bytes docling stores on a `PictureItem`). Region coordinates are page
2430/// points; the image is rendered at `page.scale` and resampled to `scale`
2431/// px/pt when one is given (docling's `images_scale`, #520). The image's `dpi`
2432/// is 72·scale (#519).
2433#[cfg(feature = "ocr-prep")]
2434fn crop_region(page: &PdfPage, region: &Region, scale: Option<f32>) -> Option<PictureImage> {
2435    let s = page.scale;
2436    let (iw, ih) = (page.image.width(), page.image.height());
2437    let x = (region.l * s).max(0.0) as u32;
2438    let y = (region.t * s).max(0.0) as u32;
2439    if x >= iw || y >= ih {
2440        return None;
2441    }
2442    let w = (((region.r - region.l) * s) as u32).min(iw - x);
2443    let h = (((region.b - region.t) * s) as u32).min(ih - y);
2444    if w == 0 || h == 0 {
2445        return None;
2446    }
2447    let sub = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
2448    match scale {
2449        Some(to) if (to - s).abs() > f32::EPSILON => {
2450            // The crop's own point extent (the pixel box, back in points), so
2451            // the resampled geometry is `round(points * scale)`.
2452            let img = rescale(sub, w as f32 / s, h as f32 / s, to);
2453            png_image(&img, to)
2454        }
2455        _ => png_image(&sub, s),
2456    }
2457}
2458
2459/// For each `picture` region, find the `caption` region closest below it (and
2460/// horizontally overlapping); docling pairs them and emits the caption first.
2461/// Each caption is claimed by at most one picture.
2462fn pair_captions(regions: &[Region]) -> Vec<Option<usize>> {
2463    let mut pairs = vec![None; regions.len()];
2464    let mut taken = vec![false; regions.len()];
2465    for (pi, p) in regions.iter().enumerate() {
2466        if p.label != "picture" {
2467            continue;
2468        }
2469        let mut best: Option<(usize, f32)> = None;
2470        for (ci, c) in regions.iter().enumerate() {
2471            if c.label != "caption" || taken[ci] {
2472                continue;
2473            }
2474            let line_h = (c.b - c.t).abs().max(1.0);
2475            let gap = c.t - p.b; // caption sits below the picture
2476            let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
2477            if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
2478                let dist = gap.abs();
2479                if best.is_none_or(|(_, bd)| dist < bd) {
2480                    best = Some((ci, dist));
2481                }
2482            }
2483        }
2484        if let Some((ci, _)) = best {
2485            pairs[pi] = Some(ci);
2486            taken[ci] = true;
2487        }
2488    }
2489    pairs
2490}
2491
2492/// Pair each `code` region with the `caption` region just **above** it (a
2493/// `Listing N:` label). docling renders the code block first, then its caption,
2494/// so the caption is consumed from its own (earlier) reading-order slot and
2495/// re-emitted after the code.
2496fn pair_code_captions(regions: &[Region]) -> Vec<Option<usize>> {
2497    let mut pairs = vec![None; regions.len()];
2498    let mut taken = vec![false; regions.len()];
2499    for (pi, p) in regions.iter().enumerate() {
2500        if p.label != "code" {
2501            continue;
2502        }
2503        let mut best: Option<(usize, f32)> = None;
2504        for (ci, c) in regions.iter().enumerate() {
2505            if c.label != "caption" || taken[ci] {
2506                continue;
2507            }
2508            let line_h = (c.b - c.t).abs().max(1.0);
2509            let gap = p.t - c.b; // caption sits above the code
2510            let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
2511            if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
2512                let dist = gap.abs();
2513                if best.is_none_or(|(_, bd)| dist < bd) {
2514                    best = Some((ci, dist));
2515                }
2516            }
2517        }
2518        if let Some((ci, _)) = best {
2519            pairs[pi] = Some(ci);
2520            taken[ci] = true;
2521        }
2522    }
2523    pairs
2524}
2525
2526/// Pair each `table`/`document_index` region with its `caption` (#265) the way
2527/// docling's `ReadingOrderPredictor._find_to_captions` does: by **reading-order
2528/// adjacency**, not geometry. A caption claims the media element
2529/// (table/picture/code) immediately next to it in the ordered region sequence,
2530/// and only when exactly one side holds one — a caption sandwiched between two
2531/// media elements stays unattached, and a text paragraph between caption and
2532/// table breaks the bond. This is what lets a flush-left "Table 3: …" label
2533/// bind a centered grid it doesn't horizontally overlap, while a caption in
2534/// the neighbouring column of a two-column page — geometrically close — never
2535/// pairs across the gutter. Runs after the picture and code pairings (the
2536/// picture/code arms of the same upstream matcher), so a caption they claimed
2537/// stays claimed. docling attaches these as `TableItem.captions` refs; the
2538/// paired caption is consumed from its own reading-order slot and rides on the
2539/// table node instead.
2540fn pair_table_captions(regions: &[Region], taken: &mut [bool]) -> Vec<Option<usize>> {
2541    let is_media = |label: &str| is_table_like(label) || matches!(label, "picture" | "code");
2542    let mut pairs: Vec<Option<usize>> = vec![None; regions.len()];
2543    for ci in 0..regions.len() {
2544        if regions[ci].label != "caption" || taken[ci] {
2545            continue;
2546        }
2547        // Furniture (headers/footers, form chrome) is not part of docling's
2548        // body-element sequence, so it neither bonds nor blocks.
2549        let prev = regions[..ci].iter().rposition(|r| !is_skipped(r.label));
2550        let next = regions[ci + 1..]
2551            .iter()
2552            .position(|r| !is_skipped(r.label))
2553            .map(|off| ci + 1 + off);
2554        let prev_media = prev.is_some_and(|j| is_media(regions[j].label));
2555        let next_media = next.is_some_and(|j| is_media(regions[j].label));
2556        let target = match (prev_media, next_media) {
2557            (true, false) => prev,
2558            (false, true) => next,
2559            // Ambiguous (media on both sides) or no media at all: leave the
2560            // caption in its own reading-order slot, as docling does.
2561            _ => None,
2562        };
2563        if let Some(ti) = target {
2564            // A first claim wins (a table with captions above *and* below
2565            // keeps the earlier one — docling's nearest-first tiebreak).
2566            if is_table_like(regions[ti].label) && pairs[ti].is_none() {
2567                pairs[ti] = Some(ci);
2568                taken[ci] = true;
2569            }
2570        }
2571    }
2572    pairs
2573}
2574
2575/// Assemble one page from its (already overlap-resolved) layout regions and
2576/// text cells.
2577/// Normalize a layout region (page points, top-left origin) to DocLang's 0–511
2578/// location grid: `clamp(round(512 · coord / page_dim), 0, 511)`, per axis,
2579/// order `[x0, y0, x1, y1]`. Mirrors docling_core's
2580/// `_doclang_utils._create_location_tokens_for_bbox` (resolution 512) so the
2581/// emitted `<location>` tokens line up with the Python groundtruth. Our heron
2582/// cluster boxes match docling's to within ~1 grid unit; the residual (mainly
2583/// the aspect-ratio-stretch vs letterbox preprocessing difference) is absorbed
2584/// by the conformance harness's geometry tolerance.
2585fn norm_loc(region: &Region, page_w: f32, page_h: f32) -> [u16; 4] {
2586    let q = |v: f32, dim: f32| -> u16 {
2587        if dim <= 0.0 {
2588            return 0;
2589        }
2590        let g = (512.0 * (v as f64) / (dim as f64)).round() as i64;
2591        g.clamp(0, 511) as u16
2592    };
2593    [
2594        q(region.l, page_w),
2595        q(region.t, page_h),
2596        q(region.r, page_w),
2597        q(region.b, page_h),
2598    ]
2599}
2600
2601/// Wrap a node in its layout provenance so the DocLang serializer emits the four
2602/// `<location>` tokens as the element's head (Markdown/JSON render `inner`
2603/// unchanged).
2604fn located(loc: [u16; 4], inner: Node) -> Node {
2605    Node::Located {
2606        location: loc,
2607        inner: Box::new(inner),
2608    }
2609}
2610
2611/// Stamp the real 1-based page number onto a page's leading marker (see
2612/// [`assemble_page`], which emits it with `page_no: 0` because only the
2613/// document-level collector knows the true index — `--pages` windows shift it).
2614pub fn stamp_page_no(nodes: &mut [Node], page_no: usize) {
2615    if let Some(Node::PageInfo { page_no: p, .. }) = nodes.first_mut() {
2616        *p = page_no;
2617    }
2618}
2619
2620/// A dense table grid plus its first-class cells (#240): `rows` is the text
2621/// grid every serializer renders (spans replicate their anchor's text);
2622/// `cells` are the docling-parity per-cell records (text, page-point bbox,
2623/// span rectangle, OTSL header roles). Produced by the TableFormer paths
2624/// (`tf_core`); lives in this always-compiled module so the pure-text (wasm
2625/// `pdf-text`) build sees the type.
2626#[derive(Clone, Debug)]
2627pub struct TableGrid {
2628    pub rows: Vec<Vec<String>>,
2629    pub cells: Vec<docling_core::TableCell>,
2630}
2631
2632/// docling's `_RICH_CELL_PICTURE_COVERAGE_THRESHOLD`.
2633const RICH_CELL_PICTURE_COVERAGE: f32 = 0.8;
2634
2635/// docling `ReadingOrderModel._match_table_pictures` (#3906, 2.118.1): every
2636/// picture ≥ 80 % inside a TableFormer-structured table on the page is matched
2637/// to the cell covering it, and returned per table as `cell index → pictures`.
2638/// A picture that pairs with a caption stays a standalone figure (upstream
2639/// would nest it and lose the caption; keeping the caption is the better
2640/// failure). Tables without first-class cells (geometric fallback) have no cell
2641/// boxes to match against and nest nothing.
2642fn match_table_pictures(
2643    regions: &[Region],
2644    table_rows: &[Option<TableGrid>],
2645    caption_for: &[Option<usize>],
2646) -> std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> {
2647    let mut out: std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> =
2648        std::collections::HashMap::new();
2649    for (p, pic) in regions.iter().enumerate() {
2650        if pic.label != "picture" || caption_for.get(p).is_some_and(Option::is_some) {
2651            continue;
2652        }
2653        let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2654        let mut best: Option<(f32, usize, usize)> = None; // (coverage, table, cell)
2655        for (t, tbl) in regions.iter().enumerate() {
2656            if !is_table_like(tbl.label) {
2657                continue;
2658            }
2659            let Some(grid) = table_rows.get(t).and_then(Option::as_ref) else {
2660                continue;
2661            };
2662            if inter(pic, tbl.l, tbl.t, tbl.r, tbl.b) / pa < RICH_CELL_PICTURE_COVERAGE {
2663                continue;
2664            }
2665            if let Some((cov, cell)) = match_picture_to_cell(pic, &grid.cells) {
2666                if best.is_none_or(|(b, _, _)| cov > b) {
2667                    best = Some((cov, t, cell));
2668                }
2669            }
2670        }
2671        if let Some((_, t, cell)) = best {
2672            let entry = out.entry(t).or_default();
2673            match entry.iter_mut().find(|(c, _)| *c == cell) {
2674                Some((_, pics)) => pics.push(p),
2675                None => entry.push((cell, vec![p])),
2676            }
2677        }
2678    }
2679    out
2680}
2681
2682/// docling `_match_picture_to_table_cell`: among the cells covering ≥ 80 % of
2683/// the picture, prefer the one at the picture's inferred grid position (the
2684/// row / column whose median cell center is nearest the picture's center —
2685/// cell boxes can overlap across logical rows and columns), else the best
2686/// coverage. Returns `(coverage, cell index)`.
2687fn match_picture_to_cell(pic: &Region, cells: &[docling_core::TableCell]) -> Option<(f32, usize)> {
2688    let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2689    let cover = |b: &[f32; 4]| inter(pic, b[0], b[1], b[2], b[3]) / pa;
2690    let eligible: Vec<(f32, usize)> = cells
2691        .iter()
2692        .enumerate()
2693        .filter_map(|(i, c)| {
2694            let b = c.bbox.as_ref()?;
2695            let cov = cover(b);
2696            (cov >= RICH_CELL_PICTURE_COVERAGE).then_some((cov, i))
2697        })
2698        .collect();
2699    if eligible.is_empty() {
2700        return None;
2701    }
2702    let mut row_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2703    let mut col_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2704    for c in cells {
2705        let Some(b) = c.bbox.as_ref() else { continue };
2706        for r in c.start_row..c.start_row + c.row_span {
2707            row_centers.entry(r).or_default().push((b[1] + b[3]) / 2.0);
2708        }
2709        for k in c.start_col..c.start_col + c.col_span {
2710            col_centers.entry(k).or_default().push((b[0] + b[2]) / 2.0);
2711        }
2712    }
2713    let median = |v: &mut Vec<f32>| -> f32 {
2714        v.sort_by(f32::total_cmp);
2715        let n = v.len();
2716        if n % 2 == 1 {
2717            v[n / 2]
2718        } else {
2719            (v[n / 2 - 1] + v[n / 2]) / 2.0
2720        }
2721    };
2722    let (px, py) = ((pic.l + pic.r) / 2.0, (pic.t + pic.b) / 2.0);
2723    let nearest = |centers: &mut std::collections::BTreeMap<usize, Vec<f32>>, target: f32| {
2724        centers
2725            .iter_mut()
2726            .map(|(&i, v)| (i, (median(v) - target).abs()))
2727            .min_by(|a, b| a.1.total_cmp(&b.1))
2728            .map(|(i, _)| i)
2729    };
2730    let row = nearest(&mut row_centers, py);
2731    let col = nearest(&mut col_centers, px);
2732    let logical: Vec<(f32, usize)> = eligible
2733        .iter()
2734        .copied()
2735        .filter(|&(_, i)| {
2736            let c = &cells[i];
2737            row.is_some_and(|r| c.start_row <= r && r < c.start_row + c.row_span)
2738                && col.is_some_and(|k| c.start_col <= k && k < c.start_col + c.col_span)
2739        })
2740        .collect();
2741    let pool = if logical.is_empty() {
2742        &eligible
2743    } else {
2744        &logical
2745    };
2746    // Python's `max` over `(coverage, cell_index, cell)` tuples: highest
2747    // coverage, ties to the higher index.
2748    pool.iter()
2749        .copied()
2750        .max_by(|a, b| a.0.total_cmp(&b.0).then(a.1.cmp(&b.1)))
2751}
2752
2753/// The DocLang structure overlay derived from first-class cells: span
2754/// continuations (`lcel`/`ucel`/`xcel`) and per-cell header roles, so the
2755/// PDF path's DCLX carries real spans instead of a flat grid.
2756fn structure_from_cells(
2757    cells: &[docling_core::TableCell],
2758    nrows: usize,
2759    ncols: usize,
2760) -> docling_core::TableStructure {
2761    let grid = || vec![vec![false; ncols]; nrows];
2762    let mut col_cont = grid();
2763    let mut row_cont = grid();
2764    let mut row_header = grid();
2765    let mut col_header = grid();
2766    for c in cells {
2767        for r in c.start_row..(c.start_row + c.row_span).min(nrows) {
2768            for k in c.start_col..(c.start_col + c.col_span).min(ncols) {
2769                col_cont[r][k] = k > c.start_col;
2770                row_cont[r][k] = r > c.start_row;
2771                row_header[r][k] = c.row_header;
2772                col_header[r][k] = c.column_header;
2773            }
2774        }
2775    }
2776    docling_core::TableStructure {
2777        header_row: Vec::new(),
2778        col_continuation: col_cont,
2779        row_continuation: row_cont,
2780        row_header,
2781        col_header,
2782    }
2783}
2784
2785pub fn assemble_page(
2786    page: &PdfPage,
2787    regions: Vec<Region>,
2788    table_rows: &[Option<TableGrid>],
2789    enrichments: &[Option<Enrichment>],
2790    // Picture-crop scale in px/pt (docling's `images_scale`, #520); `None`
2791    // keeps the page render's own scale.
2792    picture_scale: Option<f32>,
2793) -> (Vec<Node>, Vec<(String, String)>) {
2794    // Without pixels (the text-layer-only wasm build) no picture is cropped.
2795    #[cfg(not(feature = "ocr-prep"))]
2796    let _ = picture_scale;
2797    let mut nodes: Vec<Node> = Vec::new();
2798    // Every page opens with an invisible page marker carrying its size in
2799    // points — what the JSON export needs to build docling's `pages` map and
2800    // denormalize the 0–511 `<location>` grid into point bboxes (#171). The
2801    // page *number* is stamped by the document-level collector (which knows
2802    // the real 1-based index, `--pages` windows included); every serializer
2803    // except JSON skips the marker, so Markdown/DocLang stay byte-identical.
2804    nodes.push(Node::PageInfo {
2805        page_no: 0,
2806        width: page.width,
2807        height: page.height,
2808    });
2809    // Recover this page's hyperlinks (anchor-precise pairs for strict
2810    // Markdown; whole-item docling-parity links are baked below and their
2811    // pairs dropped from this list so strict output doesn't double-wrap).
2812    let mut links = resolve_link_anchors(page);
2813    // Pair each region with its precomputed TableFormer grid and enrichment
2814    // (indexed by original order) and order by reading order together, so they
2815    // stay aligned.
2816    // A picture's children (docling's `_set_cluster_children`: the regulars
2817    // > 80 % inside it) are not page elements — they leave the reading order
2818    // here and ride with their picture, to be written under it in the JSON.
2819    let parents = picture_parents(&regions);
2820    let mut kids: Vec<Vec<Region>> = vec![Vec::new(); regions.len()];
2821    let mut top: Vec<(usize, Region)> = Vec::with_capacity(regions.len());
2822    for (i, (r, parent)) in regions.into_iter().zip(parents).enumerate() {
2823        match parent {
2824            Some(p) => kids[p].push(r),
2825            None => top.push((i, r)),
2826        }
2827    }
2828    // docling's assembly order of the regions — what its reading-order
2829    // predictor knows as `cid` (#424) — before they are shuffled.
2830    let top_regions: Vec<Region> = top.iter().map(|(_, r)| r.clone()).collect();
2831    let cids = cluster_cids(&top_regions, &page.cells);
2832    type RegionItem = (Region, Option<TableGrid>, Option<Enrichment>, Vec<Region>);
2833    let mut items: Vec<RegionItem> = top
2834        .into_iter()
2835        .map(|(i, r)| {
2836            (
2837                r,
2838                table_rows.get(i).cloned().flatten(),
2839                enrichments.get(i).cloned().flatten(),
2840                std::mem::take(&mut kids[i]),
2841            )
2842        })
2843        .collect();
2844    order_with_containers(&mut items, &cids, page.width, page.height, |it| &it.0);
2845    // Float a margin page number to the front of reading order (docling parity:
2846    // right_to_left_02's bottom `11` is its first item). Stable, so everything
2847    // else keeps its order; no-op on pages without such a region.
2848    let page_h = page.height;
2849    items.sort_by_key(|(r, _, _, _)| !is_page_number(r, &page.cells, page_h));
2850    let table_rows: Vec<Option<TableGrid>> = items.iter().map(|(_, t, _, _)| t.clone()).collect();
2851    let enrichments: Vec<Option<Enrichment>> = items.iter().map(|(_, _, e, _)| e.clone()).collect();
2852    let mut picture_children: Vec<Vec<Region>> = items
2853        .iter_mut()
2854        .map(|it| std::mem::take(&mut it.3))
2855        .collect();
2856    let regions: Vec<Region> = items.into_iter().map(|(r, _, _, _)| r).collect();
2857    // Children in docling's `_sort_clusters(mode="id")` order: first source
2858    // cell, then top, then left.
2859    for kids in picture_children.iter_mut().filter(|k| k.len() > 1) {
2860        let rank = cluster_cids(kids, &page.cells);
2861        let mut ranked: Vec<(usize, Region)> = rank.into_iter().zip(kids.drain(..)).collect();
2862        ranked.sort_by_key(|(k, _)| *k);
2863        kids.extend(ranked.into_iter().map(|(_, r)| r));
2864    }
2865    // docling emits a figure's caption *before* the image marker. Pair each
2866    // picture with the caption region nearest below it and consume that caption,
2867    // so it isn't also emitted in its own (lower) reading-order position.
2868    let caption_for = pair_captions(&regions);
2869    let code_caption_for = pair_code_captions(&regions);
2870    let mut consumed = vec![false; regions.len()];
2871    for ci in caption_for.iter().flatten() {
2872        consumed[*ci] = true;
2873    }
2874    for ci in code_caption_for.iter().flatten() {
2875        consumed[*ci] = true;
2876    }
2877    // Table captions (#265) claim from what the picture/code pairings left.
2878    let mut caption_taken = consumed.clone();
2879    let table_caption_for = pair_table_captions(&regions, &mut caption_taken);
2880    for ci in table_caption_for.iter().flatten() {
2881        consumed[*ci] = true;
2882    }
2883    // Pictures inside a table become rich-cell content (docling#3906, 2.118.1):
2884    // the picture is nested in the cell it covers and not emitted standalone.
2885    let rich_cell_pictures = match_table_pictures(&regions, &table_rows, &caption_for);
2886    for (_, pics) in rich_cell_pictures.values().flatten() {
2887        for &p in pics {
2888            consumed[p] = true;
2889        }
2890    }
2891    // A code block's language label (`XML`, `C#`, …) is chrome, not content — the
2892    // detector emits it as its own region above the code; consume it.
2893    for (i, is_label) in code_language_labels(&regions, &page.cells)
2894        .into_iter()
2895        .enumerate()
2896    {
2897        if is_label {
2898            consumed[i] = true;
2899        }
2900    }
2901
2902    // docling `ReadingOrderPredictor.predict_merges`: join a text fragment with a
2903    // following text fragment strictly to its right (an author column that wraps
2904    // into the next, a paragraph continuing in the next column) into one block —
2905    // the intra-page half of docling's reading-order merges (cross-page/vertical
2906    // continuations stay with [`merge_continuations`]). Already-consumed regions
2907    // (paired captions, code labels) are excluded.
2908    // Exclusive docling cell assignment: computed once for the ordered region
2909    // list and reused for every serialization below, so a cell can never render
2910    // in two regions. The picture children take part (docling assigns cells to
2911    // every regular cluster before it nests any); their texts are split off.
2912    let with_children: Vec<Region> = regions
2913        .iter()
2914        .chain(picture_children.iter().flatten())
2915        .cloned()
2916        .collect();
2917    let mut region_texts: Vec<String> = region_texts_exclusive(&with_children, &page.cells);
2918    let mut kid_texts = region_texts.split_off(regions.len()).into_iter();
2919    let child_texts: Vec<Vec<String>> = picture_children
2920        .iter()
2921        .map(|k| kid_texts.by_ref().take(k.len()).collect())
2922        .collect();
2923    let is_text: Vec<bool> = regions
2924        .iter()
2925        .enumerate()
2926        .map(|(i, r)| r.label == "text" && !consumed[i])
2927        .collect();
2928    let is_skip: Vec<bool> = regions
2929        .iter()
2930        .enumerate()
2931        .map(|(i, r)| {
2932            consumed[i]
2933                || matches!(
2934                    r.label,
2935                    "page_header" | "page_footer" | "table" | "picture" | "caption" | "footnote"
2936                )
2937        })
2938        .collect();
2939    let boxes: Vec<(f32, f32, f32, f32)> = regions.iter().map(|r| (r.l, r.t, r.r, r.b)).collect();
2940    if docling_core::env::flag("DOCLING_RS_DEBUG_MERGES") {
2941        for (i, r) in regions.iter().enumerate() {
2942            eprintln!(
2943                "MRG {i:2} {} text={} skip={} [{:.0},{:.0},{:.0},{:.0}] {:?}",
2944                r.label,
2945                is_text[i],
2946                is_skip[i],
2947                r.l,
2948                r.t,
2949                r.r,
2950                r.b,
2951                region_texts[i].chars().take(40).collect::<String>()
2952            );
2953        }
2954    }
2955    let mut merge_suffix: Vec<String> = vec![String::new(); regions.len()];
2956    for (head, children) in
2957        crate::reading_order::predict_merges(&boxes, &region_texts, &is_text, &is_skip)
2958            .into_iter()
2959            .enumerate()
2960    {
2961        for c in children {
2962            let t = region_texts[c].trim();
2963            if !t.is_empty() {
2964                merge_suffix[head].push(' ');
2965                merge_suffix[head].push_str(t);
2966            }
2967            consumed[c] = true;
2968        }
2969    }
2970
2971    for (i, region) in regions.iter().enumerate() {
2972        if consumed[i] {
2973            continue;
2974        }
2975        // Page headers/footers: docling emits them as furniture blocks
2976        // (`<page_header>`/`<page_footer>` with a layer + location + text) at
2977        // their reading-order position, not as body — emit them, don't skip.
2978        if matches!(region.label, "page_header" | "page_footer") {
2979            let text = region_texts[i].clone();
2980            if !text.is_empty() {
2981                nodes.push(Node::PageFurniture {
2982                    footer: region.label == "page_footer",
2983                    location: norm_loc(region, page.width, page_h),
2984                    text: md_escape(&text),
2985                });
2986            }
2987            continue;
2988        }
2989        if is_skipped(region.label) {
2990            continue;
2991        }
2992        // Layout provenance for this region, normalized to docling's 0–511 grid.
2993        let loc = norm_loc(region, page.width, page_h);
2994        if region.label == "picture" {
2995            // The figure pixels are cropped from the page render for image export.
2996            // Captions are prose: markdown-escaped like a paragraph (the JSON
2997            // export unescapes back to the raw text, matching docling).
2998            let caption = caption_for[i]
2999                .map(|ci| md_escape(&region_texts[ci]))
3000                .filter(|t| !t.is_empty());
3001            let classification = match &enrichments[i] {
3002                Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
3003                _ => None,
3004            };
3005            // Without the page render (text-layer-only build) a picture keeps
3006            // its caption/classification but carries no cropped pixels.
3007            #[cfg(feature = "ocr-prep")]
3008            let image =
3009                crate::timing::timed("crop_region", || crop_region(page, region, picture_scale));
3010            #[cfg(not(feature = "ocr-prep"))]
3011            let image: Option<PictureImage> = None;
3012            nodes.push(located(
3013                loc,
3014                Node::Picture {
3015                    caption,
3016                    caption_href: None,
3017                    image,
3018                    classification,
3019                    // docling's layout pipeline parents a figure's caption to
3020                    // the picture itself (#390) — the one backend that does.
3021                    caption_parent: CaptionParent::Item,
3022                },
3023            ));
3024            let children: Vec<Node> = picture_children[i]
3025                .iter()
3026                .zip(&child_texts[i])
3027                .filter_map(|(r, text)| {
3028                    picture_child_node(r, text, norm_loc(r, page.width, page_h))
3029                })
3030                .collect();
3031            if !children.is_empty() {
3032                nodes.push(Node::PictureChildren(children));
3033            }
3034            continue;
3035        }
3036        let mut text = region_texts[i].clone();
3037        text.push_str(&merge_suffix[i]);
3038        if text.is_empty() {
3039            continue;
3040        }
3041        match region.label {
3042            // docling assembles checkboxes as TEXT_ELEM items (the region's
3043            // cells are the option label, e.g. right_to_left_03's بلی/خير)
3044            // and its Markdown serializer renders them as task-list lines
3045            // (`- [x] …`) — mirrored by [`Node::CheckboxItem`].
3046            "checkbox_selected" | "checkbox_unselected" => nodes.push(Node::CheckboxItem {
3047                checked: region.label == "checkbox_selected",
3048                text: md_escape(&text),
3049            }),
3050            // docling renders both the document title and section headers as
3051            // `##` (it never emits a top-level `#` for PDFs), so match that.
3052            "title" | "section_header" => nodes.push(located(
3053                loc,
3054                Node::Heading {
3055                    level: 2,
3056                    text: md_escape(&text),
3057                },
3058            )),
3059            // docling's `ListItemMarkerProcessor.process_list_item` runs on
3060            // every PDF list item: a leading bullet glyph or enumeration marker
3061            // followed by whitespace is split off into the item's `marker`, and
3062            // docling-core's Markdown then prints `- text` for a bullet, `N. text`
3063            // for an `N.` marker and `- a) text` for any other marker holding a
3064            // letter or digit (see [`list_item_node`]). The symbol-font bullets
3065            // docling-parse filters out of its cells are stripped first.
3066            "list_item" => nodes.push(list_item_node(&text, loc, false)),
3067            // TableFormer structure (cells + spans, text matched from word cells)
3068            // when available; otherwise geometric grid reconstruction; finally a
3069            // single cell.
3070            "table" | "document_index" => {
3071                // TableFormer grids carry first-class cells (#240: text +
3072                // page-point bbox + span rectangle + OTSL header roles) into
3073                // the public model, and the DocLang structure overlay derives
3074                // from them so DCLX emits real span/header tokens. The
3075                // geometric fallback has no per-cell records.
3076                let (mut rows, cells, structure) = match table_rows[i].clone() {
3077                    Some(grid) => {
3078                        let nrows = grid.rows.len();
3079                        let ncols = grid.rows.first().map_or(0, Vec::len);
3080                        let structure = structure_from_cells(&grid.cells, nrows, ncols);
3081                        (grid.rows, Some(grid.cells), Some(structure))
3082                    }
3083                    None => {
3084                        let rows = reconstruct_table(region, &page.cells);
3085                        let rows = if rows.iter().any(|r| r.len() > 1) {
3086                            rows
3087                        } else {
3088                            vec![vec![text.clone()]]
3089                        };
3090                        (rows, None, None)
3091                    }
3092                };
3093                // The paired caption (#265) rides on the table — docling's
3094                // TableItem.captions ref; Markdown prints it above the grid,
3095                // the JSON export emits the $ref, DocLang the <caption>.
3096                let caption = table_caption_for[i]
3097                    .map(|ci| md_escape(&region_texts[ci]))
3098                    .filter(|t| !t.is_empty());
3099                // Rich cells (docling#3906): the covering cell's blocks are its
3100                // text followed by the nested picture(s). docling's Markdown
3101                // renders a `RichTableCell` through the serializer — the
3102                // group's children joined by blank lines, newlines flattened
3103                // to spaces — so the flat `rows` text becomes
3104                // `text  <!-- image -->`; the first-class `cells` (the JSON
3105                // `table_cells` / `grid`) keep the plain text, as upstream.
3106                let mut cell_blocks: Option<Vec<Vec<Vec<Node>>>> = None;
3107                if let (Some(by_cell), Some(fc)) = (rich_cell_pictures.get(&i), cells.as_ref()) {
3108                    let nrows = rows.len();
3109                    let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
3110                    let mut blocks = vec![vec![Vec::<Node>::new(); ncols]; nrows];
3111                    for (cell_idx, pics) in by_cell {
3112                        let cell = &fc[*cell_idx];
3113                        let (r, c) = (cell.start_row, cell.start_col);
3114                        if r >= nrows || c >= ncols {
3115                            continue;
3116                        }
3117                        let mut parts: Vec<String> = Vec::new();
3118                        let mut cell_nodes: Vec<Node> = Vec::new();
3119                        if !cell.text.trim().is_empty() {
3120                            parts.push(cell.text.clone());
3121                            cell_nodes.push(Node::Paragraph {
3122                                text: cell.text.clone(),
3123                            });
3124                        }
3125                        for &p in pics {
3126                            parts.push("<!-- image -->".to_string());
3127                            let classification = match &enrichments[p] {
3128                                Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
3129                                _ => None,
3130                            };
3131                            #[cfg(feature = "ocr-prep")]
3132                            let image = crop_region(page, &regions[p], picture_scale);
3133                            #[cfg(not(feature = "ocr-prep"))]
3134                            let image: Option<PictureImage> = None;
3135                            cell_nodes.push(located(
3136                                norm_loc(&regions[p], page.width, page_h),
3137                                Node::Picture {
3138                                    caption: None,
3139                                    caption_href: None,
3140                                    image,
3141                                    classification,
3142                                    caption_parent: Default::default(),
3143                                },
3144                            ));
3145                        }
3146                        let rendered = parts.join("  ");
3147                        for row in rows.iter_mut().skip(r).take(cell.row_span) {
3148                            for slot in row.iter_mut().skip(c).take(cell.col_span) {
3149                                *slot = rendered.clone();
3150                            }
3151                        }
3152                        blocks[r][c] = cell_nodes;
3153                    }
3154                    cell_blocks = Some(blocks);
3155                }
3156                nodes.push(located(
3157                    loc,
3158                    Node::Table(Table {
3159                        rows,
3160                        location: None,
3161                        structure,
3162                        cell_blocks,
3163                        cells,
3164                        caption,
3165                        // As for pictures: the caption is the table's child.
3166                        caption_parent: CaptionParent::Item,
3167                    }),
3168                ));
3169            }
3170            // With formula enrichment the CodeFormula model decodes the region
3171            // to LaTeX; otherwise docling emits a placeholder comment rather
3172            // than the (garbled) raw glyph text.
3173            "formula" => match &enrichments[i] {
3174                Some(Enrichment::Formula { latex }) => nodes.push(Node::Formula {
3175                    latex: latex.clone(),
3176                    orig: text.clone(),
3177                    location: Some(loc),
3178                }),
3179                _ => nodes.push(Node::Paragraph {
3180                    text: "<!-- formula-not-decoded -->".into(),
3181                }),
3182            },
3183            // Code blocks: use the space-glyph-only grouping (monospace keeps its
3184            // source spacing) and emit a fenced block, preserving the line breaks
3185            // and indentation of the source (unlike prose, which reflows). pdfium
3186            // still inserts spaces around tight punctuation (`console .log`,
3187            // `add (3 , 5)`); tighten them to match docling-parse's source spacing.
3188            "code" => {
3189                // `code_region_text` preserves line breaks/indentation and tightens
3190                // each line itself; the fallback prose `text` is tightened here.
3191                let code = code_region_text(region, &page.code_cells);
3192                let code = if code.is_empty() {
3193                    tighten_code_punct(&text)
3194                } else {
3195                    code
3196                };
3197                // With code enrichment the CodeFormula model rewrites the block
3198                // (and names its language); `orig` keeps the raw extraction in
3199                // docling's shape — its parser has no line-preserving code
3200                // path, so its `orig` is the same code with the lines joined
3201                // by single spaces (indentation collapsed).
3202                // docling's parser has no line-preserving code path — its code
3203                // items carry the lines joined by single spaces. That flat
3204                // form is what every byte-conformance surface serializes
3205                // (legacy Markdown, JSON, DocLang); the line-preserving
3206                // extraction rides in `pretty` for strict Markdown only.
3207                let flat = code
3208                    .lines()
3209                    .map(str::trim)
3210                    .filter(|l| !l.is_empty())
3211                    .collect::<Vec<_>>()
3212                    .join(" ");
3213                let node = match &enrichments[i] {
3214                    Some(Enrichment::Code {
3215                        language,
3216                        text: enriched,
3217                    }) => Node::Code {
3218                        language: language.clone(),
3219                        text: enriched.clone(),
3220                        orig: Some(flat),
3221                        pretty: None,
3222                    },
3223                    _ => Node::Code {
3224                        language: None,
3225                        text: flat,
3226                        orig: None,
3227                        pretty: Some(code),
3228                    },
3229                };
3230                nodes.push(located(loc, node));
3231                // docling emits the `Listing N:` caption after the code block.
3232                if let Some(ci) = code_caption_for[i] {
3233                    let cap = md_escape(&region_texts[ci]);
3234                    if !cap.is_empty() {
3235                        nodes.push(Node::Paragraph { text: cap });
3236                    }
3237                }
3238            }
3239            // text, caption, footnote → paragraph
3240            _ => {
3241                // docling parity (`PageAssembleModel._match_hyperlink`): when
3242                // link annotations cover ≥ half of the region's box, the
3243                // hyperlink attaches to the item and the legacy Markdown
3244                // serializer wraps its full text — 2206.01062's footnote URLs
3245                // render as `[1 https://…](https://…)`. Sparse in-paragraph
3246                // citation links stay below the 0.5 coverage threshold and
3247                // remain plain text, exactly like docling.
3248                //
3249                // Scope: **footnote regions only.** Upstream's page_assemble
3250                // matches every TEXT_ELEM label, but published docling
3251                // observably carries the hyperlink into the document only for
3252                // footnote items — in both committed groundtruth generations
3253                // (docling-JSON and Markdown, independent runs) the fully
3254                // covered plain-text DOI line of 2206.01062 page 1 has
3255                // `hyperlink: None` while the equally covered footnotes carry
3256                // theirs. The corpus is the conformance reference, so match
3257                // the observed behavior; widen the label set if a future
3258                // groundtruth refresh starts linking plain text too.
3259                let escaped = md_escape(&text);
3260                let hyperlink = (region.label == "footnote")
3261                    .then(|| region_hyperlink(region, &page.links))
3262                    .flatten();
3263                let text = match hyperlink {
3264                    Some(uri) => {
3265                        // The strict-mode anchor pairs this item covers are
3266                        // superseded by the baked whole-item link.
3267                        links.retain(|(anchor, href)| {
3268                            !(href == &uri && region_texts[i].contains(anchor.as_str()))
3269                        });
3270                        format!("[{escaped}]({uri})")
3271                    }
3272                    None => escaped,
3273                };
3274                nodes.push(located(loc, Node::Paragraph { text }))
3275            }
3276        }
3277    }
3278    // A `/Rotate`-normalized scanned page (see `pdfium_backend`) was assembled
3279    // in upright space; rotate the finished geometry back so locations and the
3280    // page size are display-space, like docling and every viewer report them.
3281    if page.rotation != 0 {
3282        rotate_nodes_to_display(&mut nodes, page.rotation);
3283    }
3284    (nodes, links)
3285}
3286
3287/// One child of a picture as docling's `ReadingOrderModel._add_child_elements`
3288/// writes it under the `PictureItem`: a heading for a `section_header` /
3289/// `title` (upstream remaps title to section header), a list item for a
3290/// `list_item`, a furniture-layer text for a page header/footer, a caption,
3291/// otherwise a text item. `None` for a child that claimed no text.
3292fn picture_child_node(region: &Region, text: &str, loc: [u16; 4]) -> Option<Node> {
3293    if text.is_empty() {
3294        return None;
3295    }
3296    Some(match region.label {
3297        "title" | "section_header" => located(
3298            loc,
3299            Node::Heading {
3300                level: 2,
3301                text: md_escape(text),
3302            },
3303        ),
3304        // docling-core's `add_list_item` under a non-list parent opens a
3305        // list group per item, so every child item starts its own list;
3306        // `_add_child_elements` runs the marker processor on it too.
3307        "list_item" => list_item_node(text, loc, true),
3308        "page_header" | "page_footer" => Node::PageFurniture {
3309            footer: region.label == "page_footer",
3310            location: loc,
3311            text: md_escape(text),
3312        },
3313        "caption" => located(
3314            loc,
3315            Node::Caption {
3316                text: md_escape(text),
3317                href: None,
3318            },
3319        ),
3320        _ => located(
3321            loc,
3322            Node::Paragraph {
3323                text: md_escape(text),
3324            },
3325        ),
3326    })
3327}
3328
3329/// Rotate one 0–511 location bbox 90° clockwise on the grid (top-left origin):
3330/// `(x, y) → (511 - y, x)`.
3331fn rot_loc_cw(l: [u16; 4]) -> [u16; 4] {
3332    [511 - l[3], l[0], 511 - l[1], l[2]]
3333}
3334
3335/// Map upright-space geometry back to display space for a page whose `/Rotate`
3336/// was normalized away before inference: every `<location>` rotates `rot`°
3337/// clockwise on the 0–511 grid (the grid is per-axis normalized, so no page
3338/// dims are needed), and the `PageInfo` size returns to the display box. Node
3339/// text and order are untouched — reading order was decided upright, which is
3340/// the whole point.
3341fn rotate_nodes_to_display(nodes: &mut [Node], rot: u16) {
3342    let quarter_turns = (rot / 90) as usize;
3343    let rot_loc = |l: &mut [u16; 4]| {
3344        for _ in 0..quarter_turns {
3345            *l = rot_loc_cw(*l);
3346        }
3347    };
3348    fn walk(node: &mut Node, rot_loc: &impl Fn(&mut [u16; 4]), swap_dims: bool) {
3349        match node {
3350            Node::PageInfo { width, height, .. } => {
3351                if swap_dims {
3352                    std::mem::swap(width, height);
3353                }
3354            }
3355            Node::Located { location, inner } => {
3356                rot_loc(location);
3357                walk(inner, rot_loc, swap_dims);
3358            }
3359            Node::Furniture { inner, .. } => walk(inner, rot_loc, swap_dims),
3360            Node::Group { children, .. } | Node::PictureChildren(children) => {
3361                for c in children {
3362                    walk(c, rot_loc, swap_dims);
3363                }
3364            }
3365            Node::ListItem { location, .. }
3366            | Node::Formula { location, .. }
3367            | Node::Chart { location, .. } => {
3368                if let Some(l) = location {
3369                    rot_loc(l);
3370                }
3371            }
3372            Node::PageFurniture { location, .. } => rot_loc(location),
3373            Node::Table(t) => {
3374                if let Some(l) = &mut t.location {
3375                    rot_loc(l);
3376                }
3377            }
3378            _ => {}
3379        }
3380    }
3381    let swap_dims = quarter_turns % 2 == 1;
3382    for node in nodes {
3383        walk(node, &rot_loc, swap_dims);
3384    }
3385}
3386
3387/// Merge paragraph fragments split across a column or page break. docling joins a
3388/// paragraph whose previous fragment ends mid-sentence (a letter, not sentence
3389/// punctuation) with a lowercase continuation: `…definition of` + `lists in…` →
3390/// `…definition of lists in…`. The fragments are consecutive paragraphs, or
3391/// separated only by figure(s) the text wraps around: a column whose body flows
3392/// past a figure resumes below it (`…The wing type that is` ⟶[figure]⟶ `the most
3393/// common…`), and docling emits the whole paragraph before the figure. A heading,
3394/// table, or list between them ends the paragraph (no merge).
3395/// A paragraph that is really a figure/table caption (`Fig. 1. …`, `Table 2 …`).
3396/// Used to skip an unpaired caption when stitching a paragraph that wraps around
3397/// a figure.
3398fn looks_like_caption(text: &str) -> bool {
3399    let head: String = text.trim_start().chars().take(14).collect();
3400    (head.starts_with("Fig") || head.starts_with("Table"))
3401        && head.contains(|c: char| c.is_ascii_digit())
3402}
3403
3404/// A paragraph fragment is "open" — i.e. it might continue into the next
3405/// paragraph — when it ends mid-word (a letter) or with a wrap hyphen/dash.
3406/// docling joins `vocab-` + `ulary` → `vocab- ulary`.
3407fn paragraph_is_open(text: &str) -> bool {
3408    // docling's merge head test (`.+([a-z,\-\u00AD])\s*`): at least two chars,
3409    // ending in an ASCII lowercase letter, a comma, a hyphen, or a soft
3410    // hyphen. The comma matters: 2206's "…In phase four," resumes across the
3411    // page break. Uppercase/non-Latin endings do not merge, exactly as
3412    // upstream (the dash family is already `-` here — clean_text normalized).
3413    let t = text.trim_end();
3414    t.chars().count() >= 2
3415        && t.chars()
3416            .next_back()
3417            .is_some_and(|c| matches!(c, 'a'..='z' | ',' | '-' | '\u{ad}'))
3418}
3419
3420/// The paragraph text inside a node, looking through a [`Node::Located`]
3421/// provenance wrapper (PDF body paragraphs are wrapped since they carry a
3422/// `<location>`). Returns `None` for non-paragraph nodes.
3423fn as_paragraph(n: &Node) -> Option<&str> {
3424    match n {
3425        Node::Paragraph { text } => Some(text),
3426        Node::Located { inner, .. } => match inner.as_ref() {
3427            Node::Paragraph { text } => Some(text),
3428            _ => None,
3429        },
3430        _ => None,
3431    }
3432}
3433
3434/// Whether a node is a picture, looking through a [`Node::Located`] wrapper.
3435fn is_picture_node(n: &Node) -> bool {
3436    match n {
3437        Node::Picture { .. } => true,
3438        Node::Located { inner, .. } => matches!(inner.as_ref(), Node::Picture { .. }),
3439        _ => false,
3440    }
3441}
3442
3443/// A node a forward paragraph merge looks straight past: a figure or *table*
3444/// the text wraps around, or a page header/footer that falls between the two
3445/// fragments of a paragraph continuing across a page break (docling's merge
3446/// skip-labels: page_header, page_footer, table, picture, caption, footnote —
3447/// 2206's "…In phase four," resumes after a full caption+table+figure block).
3448fn is_merge_trailer(n: &Node) -> bool {
3449    is_picture_node(n)
3450        || matches!(
3451            n,
3452            Node::PageFurniture { .. }
3453                | Node::PageInfo { .. }
3454                | Node::Table(_)
3455                | Node::PictureChildren(_)
3456        )
3457        || matches!(n, Node::Located { inner, .. } if matches!(inner.as_ref(), Node::Table(_)))
3458        || as_paragraph(n).is_some_and(looks_like_caption)
3459}
3460
3461/// Rebuild node `i` as a paragraph with `text`, preserving its `<location>`
3462/// wrapper (and thus provenance) if it had one.
3463fn reparagraph(node: &Node, text: String) -> Node {
3464    match node {
3465        Node::Located { location, .. } => located(*location, Node::Paragraph { text }),
3466        _ => Node::Paragraph { text },
3467    }
3468}
3469
3470pub(crate) fn merge_continuations(nodes: &mut Vec<Node>) {
3471    let mut i = 0;
3472    while i + 1 < nodes.len() {
3473        let Some(a) = as_paragraph(&nodes[i]) else {
3474            i += 1;
3475            continue;
3476        };
3477        // A figure/table caption is a self-contained unit; body text resuming
3478        // after a figure is the continuation case, not the caption itself. Never
3479        // stitch *from* a caption — otherwise a caption that ends in a lone glyph
3480        // (`Fig. 5. … PubTabNet. μ`) would swallow a following stray figure label
3481        // (a standalone `μ`) into `… μ μ`.
3482        if looks_like_caption(a) {
3483            i += 1;
3484            continue;
3485        }
3486        if !paragraph_is_open(a) {
3487            i += 1;
3488            continue;
3489        }
3490        // The continuation is the next paragraph, looking past any figures the
3491        // text wraps around — and a figure/table caption that was emitted as its
3492        // own paragraph (an above-the-figure caption that didn't pair), since the
3493        // body text resumes after the whole figure+caption block.
3494        let mut j = i + 1;
3495        while nodes.get(j).is_some_and(is_merge_trailer) {
3496            j += 1;
3497        }
3498        // docling's continuation regex allows either case, but its merge runs
3499        // over the pre-assembly element stream; at node level an uppercase
3500        // start is overwhelmingly a new sentence/heading fragment (allowing it
3501        // swallowed 2305's formula blocks and redp's chapter openers), so the
3502        // continuation stays lowercase-start here.
3503        let cont = nodes.get(j).and_then(as_paragraph).is_some_and(|b| {
3504            b.trim_start()
3505                .chars()
3506                .next()
3507                .is_some_and(char::is_lowercase)
3508        });
3509        if cont {
3510            let a = as_paragraph(&nodes[i]).unwrap().trim_end().to_string();
3511            let b = as_paragraph(&nodes[j]).unwrap().trim_start().to_string();
3512            // A soft hyphen -- or a hard hyphen followed by a lowercase
3513            // continuation (guaranteed lowercase by the `cont` gate above) --
3514            // is a word split across the break: strip it and join without a
3515            // space, docling#3888 ("vocab-" + "ulary" -> "vocabulary");
3516            // docling's older serializer kept the artifact ("vocab- ulary").
3517            // Everything else joins with the space, as before.
3518            let merged = match a.strip_suffix('\u{ad}').or_else(|| a.strip_suffix('-')) {
3519                Some(stem) => format!("{stem}{b}"),
3520                None => format!("{a} {b}"),
3521            };
3522            // Keep node i's provenance wrapper; docling's merged paragraph keeps
3523            // the first fragment's geometry as its primary location.
3524            nodes[i] = reparagraph(&nodes[i], merged);
3525            nodes.remove(j);
3526            // Re-check i: the merged paragraph may continue further.
3527        } else {
3528            i += 1;
3529        }
3530    }
3531}
3532
3533/// How many leading nodes of `nodes` are safe to flush now — i.e. cannot be
3534/// rewritten by a future [`merge_continuations`] once more pages are appended.
3535///
3536/// A forward merge can only start from an "open" paragraph (ends mid-word) and
3537/// only reaches across trailing pictures and figure/table captions. So we scan
3538/// from the end past those skippable trailers: if the first non-skippable node is
3539/// an open paragraph, it (and the trailers after it) must be held; anything else —
3540/// a closed paragraph, a heading, a table, a list — blocks any forward merge, so
3541/// the whole buffer is safe to flush.
3542fn hold_start(nodes: &[Node]) -> usize {
3543    for k in (0..nodes.len()).rev() {
3544        // Skippable trailers (figures, page furniture, captions): a forward merge
3545        // looks straight past them.
3546        if is_merge_trailer(&nodes[k]) {
3547            continue;
3548        }
3549        match as_paragraph(&nodes[k]) {
3550            // An open body paragraph might still pull a continuation off the next
3551            // page — hold from here to the end.
3552            Some(text) if paragraph_is_open(text) => return k,
3553            // A closed paragraph, heading, table, list, etc. ends the paragraph:
3554            // nothing after it can merge backwards across it. Flush everything.
3555            _ => return nodes.len(),
3556        }
3557    }
3558    // Only skippable trailers (or empty) and no open paragraph to anchor a merge.
3559    nodes.len()
3560}
3561
3562/// Streaming counterpart of [`merge_continuations`]: feed per-page node batches in
3563/// document order and get back the prefix that is final (its cross-page merges are
3564/// resolved and no future page can change it), holding back only the small tail
3565/// that might still merge into the next page. Concatenating every flushed batch
3566/// (then [`finish`](Self::finish)) yields exactly the same nodes as running
3567/// [`merge_continuations`] once over the whole document.
3568pub(crate) struct StreamAssembler {
3569    pending: Vec<Node>,
3570}
3571
3572impl StreamAssembler {
3573    pub(crate) fn new() -> Self {
3574        Self {
3575            pending: Vec::new(),
3576        }
3577    }
3578
3579    /// Append one page's nodes, resolve merges within the buffer, and return the
3580    /// now-final prefix to emit (possibly empty).
3581    pub(crate) fn push(&mut self, mut nodes: Vec<Node>) -> Vec<Node> {
3582        self.pending.append(&mut nodes);
3583        merge_continuations(&mut self.pending);
3584        let cut = hold_start(&self.pending);
3585        let tail = self.pending.split_off(cut);
3586        std::mem::replace(&mut self.pending, tail)
3587    }
3588
3589    /// Flush whatever is left after the last page (the held tail is final once no
3590    /// more pages can follow).
3591    pub(crate) fn finish(self) -> Vec<Node> {
3592        self.pending
3593    }
3594}
3595
3596#[cfg(test)]
3597mod tests {
3598    use super::{cells_text, clean_text, merge_overlapping_regulars, reclaim_heading_body_footers};
3599
3600    /// docling drops a picture covering > 90 % of the page (its labels then
3601    /// read out as text); a dominant-but-not-full figure and any other label
3602    /// stay whatever their size.
3603    #[test]
3604    fn full_page_pictures_are_dropped_like_docling() {
3605        use super::drop_full_page_pictures;
3606        use crate::layout::Region;
3607        let region = |label: &'static str, l, t, r, b| Region {
3608            label,
3609            score: 0.99,
3610            l,
3611            t,
3612            r,
3613            b,
3614        };
3615        let mut regions = vec![
3616            region("picture", 0.0, 0.5, 478.9, 241.8),
3617            region("picture", 10.0, 10.0, 400.0, 200.0),
3618            region("table", 0.0, 0.0, 480.0, 243.0),
3619            region("text", 5.0, 5.0, 100.0, 20.0),
3620        ];
3621        drop_full_page_pictures(&mut regions, 480.75, 243.75);
3622        let labels: Vec<_> = regions.iter().map(|r| (r.label, r.l)).collect();
3623        assert_eq!(
3624            labels,
3625            vec![("picture", 10.0), ("table", 0.0), ("text", 5.0)]
3626        );
3627    }
3628    use super::{code_region_text, merge_continuations, resolve_link_anchors, StreamAssembler};
3629    use crate::layout::Region;
3630    use crate::pdfium_backend::{LinkAnnot, PdfPage, TextCell};
3631    use docling_core::Node;
3632
3633    /// The int8-layout guard's coverage metric: cells under detections count,
3634    /// cells outside don't, whitespace cells are ignored, and a cell-less page
3635    /// reads as fully covered (nothing to rescue).
3636    #[test]
3637    fn layout_cell_coverage_counts_claimed_text_cells() {
3638        let cell = |text: &str, l: f32, t: f32| TextCell {
3639            text: text.into(),
3640            l,
3641            t,
3642            r: l + 40.0,
3643            b: t + 10.0,
3644        };
3645        let region = Region {
3646            label: "text",
3647            score: 0.9,
3648            l: 0.0,
3649            t: 0.0,
3650            r: 100.0,
3651            b: 50.0,
3652        };
3653        let cells = vec![
3654            cell("inside", 10.0, 10.0),
3655            cell("also inside", 10.0, 30.0),
3656            cell("outside", 10.0, 200.0),
3657            cell("   ", 10.0, 210.0), // whitespace: not counted at all
3658        ];
3659        let cov = super::layout_cell_coverage(std::slice::from_ref(&region), &cells);
3660        assert!((cov - 2.0 / 3.0).abs() < 1e-6, "got {cov}");
3661        assert_eq!(super::layout_cell_coverage(&[], &[]), 1.0);
3662        assert_eq!(super::layout_cell_coverage(&[], &cells), 0.0);
3663    }
3664
3665    /// #165: a picture no longer claims cells at 0.2 intersection-over-self.
3666    /// A line straddling the figure border (≤80 % contained) becomes an orphan
3667    /// region and is emitted as page text — before the fix its cells were
3668    /// silently erased. A line fully inside the picture is the picture's child
3669    /// (docling's `_set_cluster_children`): it survives the containment drop,
3670    /// leaves the page's reading order, and is written only under the picture
3671    /// in the JSON — never in the Markdown, like docling's picture serializer.
3672    #[test]
3673    fn border_straddlers_are_page_text_picture_interior_is_a_picture_child() {
3674        let pic = Region {
3675            label: "picture",
3676            score: 0.9,
3677            l: 0.0,
3678            t: 0.0,
3679            r: 100.0,
3680            b: 100.0,
3681        };
3682        // ~35 % of this cell overlaps the picture (l=90..120 of 0..100): above
3683        // the old 0.2 claim (was swallowed), below full containment (survives).
3684        let straddler = TextCell {
3685            text: "axis label".into(),
3686            l: 90.0,
3687            t: 40.0,
3688            r: 120.0,
3689            b: 48.0,
3690        };
3691        let interior = TextCell {
3692            text: "in-figure callout".into(),
3693            l: 10.0,
3694            t: 10.0,
3695            r: 60.0,
3696            b: 18.0,
3697        };
3698        let cells = vec![straddler, interior];
3699        let mut regions = vec![pic];
3700        super::add_orphan_regions(&mut regions, &cells);
3701        super::drop_contained_regulars(&mut regions);
3702        assert_eq!(
3703            regions.iter().filter(|r| r.label == "text").count(),
3704            2,
3705            "both unclaimed lines become orphans, and a picture swallows neither"
3706        );
3707        let parents = super::picture_parents(&regions);
3708        let parent_of = |l: f32| {
3709            regions
3710                .iter()
3711                .zip(&parents)
3712                .find(|(r, _)| r.label == "text" && r.l == l)
3713                .and_then(|(_, p)| *p)
3714        };
3715        assert_eq!(
3716            parent_of(10.0),
3717            Some(0),
3718            "the callout is the picture's child"
3719        );
3720        assert_eq!(parent_of(90.0), None, "the straddler is a page element");
3721
3722        let page = PdfPage::from_cells(200.0, 200.0, 2.0, cells);
3723        let n = regions.len();
3724        let (nodes, _) = super::assemble_page(&page, regions, &vec![None; n], &vec![None; n], None);
3725        let children: Vec<&Node> = nodes
3726            .iter()
3727            .filter_map(|n| match n {
3728                Node::PictureChildren(c) => Some(c),
3729                _ => None,
3730            })
3731            .flatten()
3732            .collect();
3733        assert!(
3734            matches!(children.as_slice(), [Node::Located { inner, .. }]
3735                if matches!(inner.as_ref(), Node::Paragraph { text } if text == "in-figure callout")),
3736            "{children:?}"
3737        );
3738        let mut doc = docling_core::DoclingDocument::new("t");
3739        doc.nodes = nodes;
3740        let md = doc.export_to_markdown();
3741        assert!(md.contains("axis label"), "{md}");
3742        assert!(!md.contains("in-figure callout"), "{md}");
3743        let json = doc.export_to_json_value();
3744        let pic = &json["pictures"][0];
3745        let child = pic["children"][0]["$ref"].as_str().expect("a child ref");
3746        let idx: usize = child.rsplit('/').next().unwrap().parse().unwrap();
3747        assert_eq!(json["texts"][idx]["text"], "in-figure callout");
3748        assert_eq!(json["texts"][idx]["parent"]["$ref"], "#/pictures/0");
3749        assert_eq!(json["texts"][idx]["content_layer"], "body");
3750    }
3751
3752    /// docling#3906's concern, pinned on our side: a picture detected fully
3753    /// inside a table region must survive the containment drop (upstream now
3754    /// attaches it to the table's cell; we keep it as a body sibling — either
3755    /// way it must not vanish). The text region inside the same table is the
3756    /// control: regulars are the ones the drop swallows.
3757    #[test]
3758    fn picture_inside_a_table_region_survives_the_containment_drop() {
3759        let mut regions = vec![
3760            region("table", 0.9, 0.0, 0.0, 200.0, 200.0),
3761            region("picture", 0.9, 20.0, 20.0, 120.0, 120.0),
3762            region("text", 0.9, 20.0, 140.0, 180.0, 180.0),
3763        ];
3764        super::drop_contained_regulars(&mut regions);
3765        let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
3766        assert_eq!(
3767            labels,
3768            ["table", "picture"],
3769            "the in-table picture stays; the in-table regular is the special's child"
3770        );
3771    }
3772
3773    /// Table–caption pairing (#265) is reading-order adjacency, docling's
3774    /// `_find_to_captions`: a caption binds the table directly next to it in
3775    /// the region sequence — above-caption and below-caption both work, and
3776    /// geometry is irrelevant (a same-page caption in the other column of a
3777    /// two-column layout is *not* adjacent, however close its box is). A
3778    /// caption with media on both sides, or separated from the table by a
3779    /// text paragraph, stays unattached.
3780    #[test]
3781    fn table_captions_pair_by_reading_order_adjacency() {
3782        // caption → table (above-caption), then table → caption (below-caption),
3783        // then a caption fenced off by a paragraph, then one between two tables.
3784        let regions = vec![
3785            region("text", 0.9, 0.0, 0.0, 100.0, 10.0), // 0 body text
3786            region("caption", 0.9, 0.0, 12.0, 60.0, 20.0), // 1 above-caption
3787            region("table", 0.9, 20.0, 22.0, 90.0, 60.0), // 2 ← pairs with 1
3788            region("table", 0.9, 0.0, 70.0, 100.0, 110.0), // 3 ← pairs with 4
3789            region("caption", 0.9, 0.0, 112.0, 60.0, 120.0), // 4 below-caption
3790            region("text", 0.9, 0.0, 130.0, 100.0, 140.0), // 5 body text
3791            region("caption", 0.9, 0.0, 142.0, 60.0, 150.0), // 6 fenced by 5/7
3792            region("text", 0.9, 0.0, 152.0, 100.0, 162.0), // 7 body text
3793            region("table", 0.9, 0.0, 170.0, 100.0, 200.0), // 8 unpaired
3794            region("caption", 0.9, 0.0, 202.0, 60.0, 210.0), // 9 ambiguous
3795            region("table", 0.9, 0.0, 212.0, 100.0, 240.0), // 10 unpaired
3796        ];
3797        let mut taken = vec![false; regions.len()];
3798        let pairs = super::pair_table_captions(&regions, &mut taken);
3799        assert_eq!(pairs[2], Some(1), "caption directly above its table pairs");
3800        assert_eq!(pairs[3], Some(4), "caption directly below its table pairs");
3801        assert_eq!(
3802            pairs[8], None,
3803            "a text paragraph between caption and table breaks the bond"
3804        );
3805        assert_eq!(
3806            pairs[10], None,
3807            "a caption between two tables is ambiguous and stays loose"
3808        );
3809        assert!(taken[1] && taken[4] && !taken[6] && !taken[9]);
3810    }
3811
3812    /// A colored terms-and-conditions panel detected as `picture` demotes into
3813    /// per-paragraph `text` regions (the blank line between C.7 and C.8 splits
3814    /// them); a chart whose only text is a few narrow axis labels keeps its
3815    /// crop untouched.
3816    #[test]
3817    fn text_panels_demote_to_paragraphs_but_charts_keep_their_crop() {
3818        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3819            text: text.to_string(),
3820            l,
3821            t,
3822            r,
3823            b,
3824        };
3825        let panel = Region {
3826            label: "picture",
3827            score: 0.9,
3828            l: 0.0,
3829            t: 0.0,
3830            r: 100.0,
3831            b: 100.0,
3832        };
3833        // Three tight lines, a blank-line gap, two more: two paragraphs.
3834        let cells = vec![
3835            cell(
3836                "C.7. Wenn Sie diesen Vertrag widerrufen,",
3837                5.0,
3838                10.0,
3839                95.0,
3840                18.0,
3841            ),
3842            cell(
3843                "haben wir Ihnen alle Zahlungen, die wir",
3844                5.0,
3845                20.0,
3846                95.0,
3847                28.0,
3848            ),
3849            cell(
3850                "von Ihnen erhalten haben, zurückzuzahlen.",
3851                5.0,
3852                30.0,
3853                90.0,
3854                38.0,
3855            ),
3856            cell(
3857                "C.8. Wir können die Rückzahlung verweigern,",
3858                5.0,
3859                52.0,
3860                95.0,
3861                60.0,
3862            ),
3863            cell(
3864                "bis wir die Waren wieder zurückerhalten haben.",
3865                5.0,
3866                62.0,
3867                92.0,
3868                70.0,
3869            ),
3870        ];
3871        let mut regions = vec![panel.clone()];
3872        super::recover_text_panels(&mut regions, &cells);
3873        assert_eq!(
3874            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3875            ["text", "text"],
3876            "dense panel must demote into one text region per paragraph"
3877        );
3878        assert!(regions[0].b < regions[1].t, "paragraphs split at the gap");
3879        // Sparse narrow labels (a chart): picture survives.
3880        let labels = vec![
3881            cell("0", 5.0, 90.0, 8.0, 95.0),
3882            cell("50", 5.0, 50.0, 10.0, 55.0),
3883            cell("100", 5.0, 10.0, 12.0, 15.0),
3884            cell("t, s", 45.0, 96.0, 55.0, 100.0),
3885        ];
3886        let mut regions = vec![panel];
3887        super::recover_text_panels(&mut regions, &labels);
3888        assert_eq!(
3889            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3890            ["picture"]
3891        );
3892    }
3893
3894    /// An uncaptioned chart on a scanned page whose title, axis labels, and
3895    /// OCR boxes over the plot area are dense and wide enough to pass the
3896    /// coverage/width gates still keeps its crop: its line heights are ragged
3897    /// (title face vs tick labels vs bar-area OCR), failing the uniform-leading
3898    /// gate — a real text panel is set with constant leading (#173).
3899    #[test]
3900    fn dense_titled_chart_keeps_its_crop() {
3901        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3902            text: text.to_string(),
3903            l,
3904            t,
3905            r,
3906            b,
3907        };
3908        let chart = Region {
3909            label: "picture",
3910            score: 0.9,
3911            l: 0.0,
3912            t: 0.0,
3913            r: 100.0,
3914            b: 100.0,
3915        };
3916        // Five wide lines at wildly different heights: a 12-pt title, 20-pt OCR
3917        // boxes over the bars, 4–5-pt tick/axis labels. Coverage and median
3918        // width both clear the panel thresholds.
3919        let cells = vec![
3920            cell("Underground Water Storage", 10.0, 5.0, 90.0, 17.0),
3921            cell("aquifer recharge zone", 15.0, 30.0, 75.0, 50.0),
3922            cell("confined | unconfined | perched", 12.0, 55.0, 80.0, 59.0),
3923            cell("saturated thickness", 8.0, 70.0, 60.0, 90.0),
3924            cell("distance from well, km", 20.0, 92.0, 85.0, 97.0),
3925        ];
3926        let mut regions = vec![chart];
3927        super::recover_text_panels(&mut regions, &cells);
3928        assert_eq!(
3929            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3930            ["picture"],
3931            "ragged line heights mark a figure, not a text panel"
3932        );
3933    }
3934
3935    /// docling serializes a cluster's cells in docling-parse index order
3936    /// (`_sort_cells`) and joins them with `PageAssembleModel.sanitize_text`:
3937    /// a space after every line except one ending in `-`, which either fuses a
3938    /// wrapped word (alnum on both sides — dash dropped) or glues verbatim (a
3939    /// bare `-` cell: `[0000` `-` `0002` → `[0000 -0002`, the 2305 ORCID line;
3940    /// `-` + `"C" cell -` + `a new table cell` → `-"C" cell a new table cell`,
3941    /// its OTSL list). Verified against the corpus: pure index order beats any
3942    /// geometric re-sort (normal_4pages' heading numerals paint after their
3943    /// text and belong last: `## 들어가며 1`).
3944    #[test]
3945    fn cells_join_in_index_order_with_sanitize_text_rules() {
3946        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3947            text: text.to_string(),
3948            l,
3949            t,
3950            r,
3951            b,
3952        };
3953        let region = Region {
3954            label: "text",
3955            score: 1.0,
3956            l: 0.0,
3957            t: 95.0,
3958            r: 200.0,
3959            b: 130.0,
3960        };
3961        // ORCID superscript: a bare dash cell is a *detached* dash — kept, and
3962        // since docling#4052 (2.122) it joins with the ordinary space on both
3963        // sides (`[0000 -0002 -6960]` before that fix).
3964        let orcid = vec![
3965            cell("[0000", 10.0, 100.0, 30.0, 110.0),
3966            cell("−", 30.0, 100.0, 34.0, 110.0),
3967            cell("0002", 34.0, 100.0, 50.0, 110.0),
3968            cell("−", 50.0, 100.0, 54.0, 110.0),
3969            cell("6960]", 54.0, 100.0, 70.0, 110.0),
3970        ];
3971        assert_eq!(super::region_text(&region, &orcid), "[0000 - 0002 - 6960]");
3972        // Wrapped word: dash dropped, lines fused (both boundary words alnum).
3973        let wrapped = vec![
3974            cell("platforms-", 10.0, 100.0, 60.0, 110.0),
3975            cell("reflects the design", 10.0, 112.0, 90.0, 122.0),
3976        ];
3977        assert_eq!(
3978            super::region_text(&region, &wrapped),
3979            "platformsreflects the design"
3980        );
3981        // Dash-ending lines that are *detached* dashes (a bare bullet cell, a
3982        // `cell -` separator): the dash stays and the lines join with a space
3983        // — docling#4052; before it they glued (`-"C" cell a new table cell`,
3984        // 2305's OTSL list bullets).
3985        let otsl = vec![
3986            cell("–", 10.0, 100.0, 14.0, 110.0),
3987            cell("\"C\" cell -", 16.0, 100.0, 60.0, 110.0),
3988            cell("a new table cell", 10.0, 112.0, 80.0, 122.0),
3989        ];
3990        assert_eq!(
3991            super::region_text(&region, &otsl),
3992            "- \"C\" cell - a new table cell"
3993        );
3994        // Index order is authoritative — no geometric re-sort.
3995        let numeral = vec![
3996            cell("들어가며", 30.0, 100.0, 80.0, 110.0),
3997            cell("1", 10.0, 98.0, 25.0, 112.0), // big numeral painted last
3998        ];
3999        assert_eq!(super::region_text(&region, &numeral), "들어가며 1");
4000    }
4001
4002    /// The geometric-reliability gate, on the two shapes it has to tell apart.
4003    #[test]
4004    fn geometric_reliability_rejects_split_column_grids() {
4005        let g = |rows: &[&[&str]]| -> Vec<Vec<String>> {
4006            rows.iter()
4007                .map(|r| r.iter().map(|c| c.to_string()).collect())
4008                .collect()
4009        };
4010        // A genuine grid: dense, every column carrying entries. Nothing for
4011        // TableFormer to improve, so geometry is used as-is.
4012        assert!(super::geometric_table_is_reliable(&g(&[
4013            &["Datum", "Leistung", "Anzahl", "Kosten"],
4014            &["04.07", "Internet", "1", "40.30"],
4015            &["04.07", "Telefon", "2", "8.06"],
4016        ])));
4017        // The left-edge split artefact (the shape a scanned invoice produced):
4018        // one real label column plus values scattered across three sparse ones.
4019        assert!(!super::geometric_table_is_reliable(&g(&[
4020            &["www.magenta.at/faq", "", "", ""],
4021            &["Serviceteam", "", "", ""],
4022            &["Telefon", "0676/2000", "", ""],
4023            &["Kundennummer", "", "", "1.21699482"],
4024            &["Rechnungsnummer", "", "922769430725", ""],
4025            &["Rechnungsdatum", "", "", "04.07.2025"],
4026        ])));
4027        // A column only one row ever uses is a split artefact even when the
4028        // grid is otherwise dense.
4029        assert!(!super::geometric_table_is_reliable(&g(&[
4030            &["a", "b", ""],
4031            &["c", "d", ""],
4032            &["e", "f", "g"],
4033        ])));
4034        // Degenerate shapes are never vouched for — TableFormer may recover
4035        // structure a collapsed reconstruction lost.
4036        assert!(!super::geometric_table_is_reliable(&g(&[&[
4037            "only one column"
4038        ]])));
4039        assert!(!super::geometric_table_is_reliable(&[]));
4040    }
4041
4042    /// A `picture` region is cropped out of the rendered page, whatever built
4043    /// that page. The browser pipeline (#157) has no pdfium but does hand over
4044    /// the rasterized bitmap through `from_cells_with_image`, so it must get
4045    /// the same figure bytes the native path does — that is what makes
4046    /// `images = "embedded"` inline real pixels instead of a placeholder.
4047    #[cfg(feature = "ocr-prep")]
4048    #[test]
4049    fn picture_regions_are_cropped_from_a_host_supplied_page_image() {
4050        let mut img = image::RgbImage::new(200, 200);
4051        // Paint the figure area so the crop is distinguishable from the page.
4052        for y in 100..160 {
4053            for x in 20..120 {
4054                img.put_pixel(x, y, image::Rgb([255, 0, 0]));
4055            }
4056        }
4057        // scale 2.0: the region is in page points, the bitmap in pixels.
4058        let page = PdfPage::from_cells_with_image(100.0, 100.0, 2.0, Vec::new(), img);
4059        let region = Region {
4060            label: "picture",
4061            score: 0.9,
4062            l: 10.0,
4063            t: 50.0,
4064            r: 60.0,
4065            b: 80.0,
4066        };
4067        let (nodes, _) = super::assemble_page(&page, vec![region], &[None], &[None], None);
4068        // Layout-derived nodes carry provenance, so the picture arrives wrapped.
4069        let image = nodes
4070            .iter()
4071            .find_map(|n| match n {
4072                Node::Located { inner, .. } => match &**inner {
4073                    Node::Picture { image, .. } => image.as_ref(),
4074                    _ => None,
4075                },
4076                Node::Picture { image, .. } => image.as_ref(),
4077                _ => None,
4078            })
4079            .expect("a picture node with cropped pixels");
4080        assert_eq!(image.mimetype, "image/png");
4081        assert_eq!((image.width, image.height), (100, 60), "region × scale");
4082        assert!(!image.data.is_empty(), "PNG bytes were encoded");
4083    }
4084
4085    #[test]
4086    fn link_anchors_split_a_shared_word_cell_between_adjacent_links() {
4087        // A common header layout: one text run holds several pipe-separated
4088        // labels, each carrying its own link annotation. Every link must get
4089        // its own label as the anchor (and the "|" separators must belong to
4090        // none), not the whole run.
4091        let annot = |l: f32, r: f32, uri: &str| LinkAnnot {
4092            l,
4093            t: 100.0,
4094            r,
4095            b: 114.0,
4096            uri: uri.into(),
4097        };
4098        let page = PdfPage {
4099            width: 600.0,
4100            height: 800.0,
4101            scale: 2.0,
4102            cells: Vec::new(),
4103            code_cells: Vec::new(),
4104            // "LinkedIn | GitHub | Credly" = 26 chars over x 100..360.
4105            word_cells: vec![cell(
4106                "LinkedIn | GitHub | Credly",
4107                100.0,
4108                100.0,
4109                360.0,
4110                114.0,
4111            )],
4112            image: image::RgbImage::new(1, 1),
4113            image_layout: None,
4114            links: vec![
4115                annot(100.0, 180.0, "https://l"),
4116                annot(200.0, 260.0, "https://g"),
4117                annot(290.0, 360.0, "https://c"),
4118            ],
4119            rotation: 0,
4120        };
4121        assert_eq!(
4122            resolve_link_anchors(&page),
4123            vec![
4124                ("LinkedIn".to_string(), "https://l".to_string()),
4125                ("GitHub".to_string(), "https://g".to_string()),
4126                ("Credly".to_string(), "https://c".to_string()),
4127            ]
4128        );
4129    }
4130
4131    /// A one-line code cell at `[l, r] × [t, b]` (top-left coords).
4132    fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
4133        TextCell {
4134            text: text.into(),
4135            l,
4136            t,
4137            r,
4138            b,
4139        }
4140    }
4141
4142    /// OCR-path grouping (docling's `_remove_overlapping_clusters("regular")`):
4143    /// the low-score paragraph box RT-DETR draws over its own high-score line
4144    /// boxes collapses to one region — the group's union, with the survivor's
4145    /// label and score — so region-scoped OCR reads each line once. Regions
4146    /// that merely sit near each other, and specials, are untouched.
4147    #[test]
4148    fn merge_overlapping_regulars_collapses_a_block_over_its_lines() {
4149        let mut regions = vec![
4150            region("text", 0.84, 60.0, 186.0, 270.0, 198.0),
4151            region("text", 0.80, 60.0, 160.0, 294.0, 172.0),
4152            region("text", 0.79, 59.0, 107.0, 272.0, 119.0),
4153            // The paragraph box, lower score, containing all three lines.
4154            region("text", 0.52, 59.0, 107.0, 295.0, 200.0),
4155            // Elsewhere on the page: stays as is.
4156            region("section_header", 0.77, 60.0, 71.0, 253.0, 86.0),
4157            // A picture the block overlaps is not a regular — never grouped.
4158            region("picture", 0.9, 50.0, 100.0, 300.0, 210.0),
4159        ];
4160        merge_overlapping_regulars(&mut regions);
4161        assert_eq!(regions.len(), 3, "{regions:?}");
4162        let block = regions
4163            .iter()
4164            .find(|r| r.label == "text")
4165            .expect("one text");
4166        // docling keeps the largest passing candidate unless a rival is both
4167        // comparable in size and > 0.05 more confident; the 16× larger block
4168        // passes, and a smaller line never replaces a larger current best.
4169        // Either way the survivor spans the whole group.
4170        assert_eq!(
4171            (block.l, block.t, block.r, block.b),
4172            (59.0, 107.0, 295.0, 200.0)
4173        );
4174        assert!(regions.iter().any(|r| r.label == "section_header"));
4175        assert!(regions.iter().any(|r| r.label == "picture"));
4176    }
4177
4178    /// The pairwise rules, each in the arrangement where it decides the
4179    /// outcome: docling seeds the survivor with the group's first passing
4180    /// cluster and a later one replaces it only when larger *and* within
4181    /// 0.05 confidence, so a rule that merely lets a cluster pass matters
4182    /// exactly when that cluster comes first — a same-sized list item ahead
4183    /// of a far more confident text box, a code box ahead of the text it
4184    /// contains. Without the rule either would be rejected outright (similar
4185    /// size, rival > 0.05 more confident) and the text box would win.
4186    #[test]
4187    fn merge_overlapping_regulars_follows_the_preference_rules() {
4188        let mut regions = vec![
4189            region("list_item", 0.6, 0.0, 0.0, 102.0, 20.0),
4190            region("text", 0.9, 0.0, 0.0, 100.0, 20.0),
4191        ];
4192        merge_overlapping_regulars(&mut regions);
4193        assert_eq!(regions.len(), 1);
4194        assert_eq!(regions[0].label, "list_item");
4195
4196        let mut regions = vec![
4197            region("code", 0.6, 0.0, 0.0, 100.0, 100.0),
4198            region("text", 0.9, 2.0, 2.0, 98.0, 98.0),
4199        ];
4200        merge_overlapping_regulars(&mut regions);
4201        assert_eq!(regions.len(), 1);
4202        assert_eq!(regions[0].label, "code");
4203
4204        // No rule applies: a near-identical rival that is > 0.05 more
4205        // confident rejects the candidate whatever the order.
4206        for order in [[0.9, 0.6], [0.6, 0.9]] {
4207            let mut regions = vec![
4208                region("text", order[0], 0.0, 0.0, 100.0, 20.0),
4209                region("text", order[1], 0.0, 0.0, 105.0, 21.0),
4210            ];
4211            merge_overlapping_regulars(&mut regions);
4212            assert_eq!(regions.len(), 1);
4213            assert_eq!(regions[0].score, 0.9, "the confident twin wins");
4214            assert_eq!(
4215                (regions[0].r, regions[0].b),
4216                (105.0, 21.0),
4217                "on the union box"
4218            );
4219        }
4220
4221        // Side by side (no containment, IoU 0): nothing to merge.
4222        let mut regions = vec![
4223            region("text", 0.9, 0.0, 0.0, 100.0, 20.0),
4224            region("text", 0.9, 0.0, 22.0, 100.0, 42.0),
4225        ];
4226        merge_overlapping_regulars(&mut regions);
4227        assert_eq!(regions.len(), 2);
4228    }
4229
4230    #[test]
4231    fn footer_under_a_body_less_heading_is_its_text() {
4232        // The reporting CV's last page: `## Languages` at t=773.7..783.4 and
4233        // its one-line body at 798.5..808.7, labelled page_footer (A4, 595 pt).
4234        let mut regions = vec![
4235            region("list_item", 0.9, 66.0, 754.0, 353.0, 765.0),
4236            region("section_header", 0.94, 43.0, 773.7, 89.5, 783.4),
4237            region("page_footer", 0.88, 43.0, 798.5, 527.8, 808.7),
4238        ];
4239        reclaim_heading_body_footers(&mut regions, 595.28);
4240        assert_eq!(regions[2].label, "text");
4241        assert_eq!(regions[1].label, "section_header");
4242
4243        // A heading with its own paragraph and a running footer below: kept.
4244        let mut regions = vec![
4245            region("section_header", 0.9, 43.0, 700.0, 120.0, 710.0),
4246            region("text", 0.9, 43.0, 714.0, 520.0, 780.0),
4247            region("page_footer", 0.9, 43.0, 798.0, 520.0, 808.0),
4248        ];
4249        reclaim_heading_body_footers(&mut regions, 595.28);
4250        assert_eq!(regions[2].label, "page_footer");
4251
4252        // A page number under a trailing heading is too narrow to be a body.
4253        let mut regions = vec![
4254            region("section_header", 0.9, 43.0, 773.0, 120.0, 783.0),
4255            region("page_footer", 0.9, 280.0, 798.0, 300.0, 808.0),
4256        ];
4257        reclaim_heading_body_footers(&mut regions, 595.28);
4258        assert_eq!(regions[1].label, "page_footer");
4259
4260        // Too far below the heading (a real footer after a heading that ends
4261        // the page): kept.
4262        let mut regions = vec![
4263            region("section_header", 0.9, 43.0, 700.0, 120.0, 710.0),
4264            region("page_footer", 0.9, 43.0, 798.0, 520.0, 808.0),
4265        ];
4266        reclaim_heading_body_footers(&mut regions, 595.28);
4267        assert_eq!(regions[1].label, "page_footer");
4268    }
4269
4270    fn region(label: &'static str, score: f32, l: f32, t: f32, r: f32, b: f32) -> Region {
4271        Region {
4272            label,
4273            score,
4274            l,
4275            t,
4276            r,
4277            b,
4278        }
4279    }
4280
4281    #[test]
4282    fn resolve_collapses_nested_code_keeping_the_larger_box() {
4283        // A tight high-score `code` box and a taller lower-score near-duplicate that
4284        // contains it must collapse to one — the *larger* box, so every cell stays
4285        // covered and nothing leaks out as orphan text.
4286        let tight = region("code", 0.95, 78.0, 292.0, 300.0, 330.0);
4287        let wide = region("code", 0.66, 63.0, 260.0, 320.0, 346.0);
4288        let kept = super::resolve(vec![tight, wide]);
4289        assert_eq!(kept.len(), 1, "nested code boxes must collapse to one");
4290        assert!(
4291            kept[0].l == 63.0 && kept[0].b == 346.0,
4292            "the larger containing box is kept"
4293        );
4294    }
4295
4296    #[test]
4297    fn resolve_keeps_distinct_and_differently_typed_regions() {
4298        // A text box fully inside a lower-score *table* must NOT be collapsed (the
4299        // code dedup is code-only), and two separate code blocks stay separate.
4300        let text = region("text", 0.95, 90.0, 210.0, 200.0, 230.0);
4301        let table = region("table", 0.60, 80.0, 200.0, 400.0, 500.0);
4302        assert_eq!(super::resolve(vec![text, table]).len(), 2);
4303
4304        let code_a = region("code", 0.9, 78.0, 100.0, 300.0, 140.0);
4305        let code_b = region("code", 0.9, 78.0, 300.0, 300.0, 360.0); // far below, no overlap
4306        assert_eq!(super::resolve(vec![code_a, code_b]).len(), 2);
4307    }
4308
4309    /// A two-column glossary page came out as three column
4310    /// tables *and* one low-score whole-page table over them. docling's wrapper
4311    /// `_remove_overlapping_clusters` keeps one table per overlapping group
4312    /// (here the whole-page one: > 2× every rival's area and ≤ 0.2 less
4313    /// confident than the running best); `greedy` alone kept all four and
4314    /// emitted every cell twice.
4315    #[test]
4316    fn resolve_keeps_one_table_per_nested_group() {
4317        let kept = super::resolve(vec![
4318            region("table", 0.71, 26.0, 203.0, 183.0, 558.0),
4319            region("table", 0.67, 26.0, 55.0, 183.0, 196.0),
4320            region("table", 0.66, 196.0, 56.0, 354.0, 561.0),
4321            region("table", 0.53, 25.0, 53.0, 354.0, 561.0),
4322        ]);
4323        assert_eq!(kept.len(), 1, "one survivor per overlapping group");
4324        assert_eq!((kept[0].l, kept[0].b), (25.0, 561.0));
4325        // Side-by-side tables that don't overlap stay separate.
4326        let kept = super::resolve(vec![
4327            region("table", 0.9, 26.0, 55.0, 183.0, 558.0),
4328            region("table", 0.9, 196.0, 56.0, 354.0, 561.0),
4329        ]);
4330        assert_eq!(kept.len(), 2);
4331    }
4332
4333    /// A dense data table detected as both picture (0.80) and table (0.62) on one box.
4334    /// The picture is ≥ 0.1 more confident, so `_handle_cross_type_overlaps`
4335    /// keeps both, and the dense table text passed the text-panel gates: the
4336    /// demoted paragraph repeated every cell the table grid renders. A
4337    /// paragraph > 80 % inside a surviving table is the table's child and is
4338    /// not emitted; a panel with no table under it still demotes.
4339    #[test]
4340    fn text_panel_over_a_table_does_not_repeat_its_cells() {
4341        let lines = |t0: f32| -> Vec<TextCell> {
4342            (0..4)
4343                .map(|i| {
4344                    let t = t0 + 10.0 * i as f32;
4345                    cell("1 2 3 4 5 6 7 8 9 10 11 12", 5.0, t, 95.0, t + 8.0)
4346                })
4347                .collect()
4348        };
4349        let mut cells = lines(0.0);
4350        cells.extend(lines(200.0));
4351        let mut regions = vec![
4352            region("picture", 0.80, 0.0, 0.0, 100.0, 45.0),
4353            region("table", 0.62, 0.0, 0.0, 100.0, 45.0),
4354            region("picture", 0.80, 0.0, 200.0, 100.0, 245.0),
4355        ];
4356        super::recover_text_panels(&mut regions, &cells);
4357        assert_eq!(
4358            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
4359            ["table", "text"]
4360        );
4361        assert_eq!(regions[1].t, 200.0, "the table-free panel still demotes");
4362    }
4363
4364    /// docling's `ListItemMarkerProcessor` + docling-core's Markdown list rules
4365    /// (see `list_item_node`): a bullet glyph or `-` is split off and the
4366    /// serializer's own `-` printed (2305's OTSL list reads `- "C" cell`, not
4367    /// `- - "C" cell`); `N.` is an ordered item; any other marker with a letter
4368    /// or digit stays in the text behind the bullet; no marker → plain bullet.
4369    #[test]
4370    fn list_item_markers_split_like_docling() {
4371        let text_of = |n: &Node| match n {
4372            Node::ListItem {
4373                ordered,
4374                number,
4375                text,
4376                marker,
4377                ..
4378            } => (*ordered, *number, text.clone(), marker.clone()),
4379            other => panic!("{other:?}"),
4380        };
4381        let loc = [0, 0, 100, 10];
4382        assert_eq!(
4383            text_of(&super::list_item_node(
4384                "- \"C\" cell - a new table cell",
4385                loc,
4386                false
4387            )),
4388            (
4389                false,
4390                0,
4391                "\"C\" cell - a new table cell".into(),
4392                Some("-".into())
4393            )
4394        );
4395        assert_eq!(
4396            text_of(&super::list_item_node("• Bullet text", loc, false)),
4397            (false, 0, "Bullet text".into(), Some("•".into()))
4398        );
4399        assert_eq!(
4400            text_of(&super::list_item_node("3. Third step", loc, false)),
4401            (true, 3, "Third step".into(), Some("3.".into()))
4402        );
4403        assert_eq!(
4404            text_of(&super::list_item_node("a) Option", loc, false)),
4405            (false, 0, "a) Option".into(), Some("a)".into()))
4406        );
4407        assert_eq!(
4408            text_of(&super::list_item_node("1.2 Nested outline", loc, false)),
4409            (false, 0, "1.2 Nested outline".into(), Some("1.2".into()))
4410        );
4411        // 2203's `3.a. If all IOU scores…`: a compound marker is not an item
4412        // number — docling prints `- 3.a. If all…`.
4413        assert_eq!(
4414            text_of(&super::list_item_node("3.a. If all IOU scores", loc, false)),
4415            (
4416                false,
4417                0,
4418                "3.a. If all IOU scores".into(),
4419                Some("3.a.".into())
4420            )
4421        );
4422        // A glued symbol-font bullet is stripped, a spaced one is the marker.
4423        assert_eq!(
4424            text_of(&super::list_item_node("•Glued", loc, false)),
4425            (false, 0, "Glued".into(), Some("·".into()))
4426        );
4427        // No whitespace after the glyph → not a marker (docling's `\s` is required).
4428        assert_eq!(
4429            text_of(&super::list_item_node("-5 degrees", loc, false)),
4430            (false, 0, "-5 degrees".into(), Some("·".into()))
4431        );
4432        // The remaining numbered shapes, first-wins like docling's list.
4433        for (input, marker, body) in [
4434            ("1.2.3. Deep", "1.2.3.", "Deep"),
4435            ("9a) Nine-a", "9a)", "Nine-a"),
4436            ("(3.a) Paren", "(3.a)", "Paren"),
4437            ("12) Twelve", "12)", "Twelve"),
4438            ("(4) Four", "(4)", "Four"),
4439            ("[7] Seven", "[7]", "Seven"),
4440            ("iv. Roman", "iv.", "Roman"),
4441            ("IX. Roman", "IX.", "Roman"),
4442            ("b. Letter", "b.", "Letter"),
4443            ("B) Letter", "B)", "Letter"),
4444        ] {
4445            assert_eq!(
4446                super::split_list_marker(input),
4447                Some((marker, body, true)),
4448                "{input}"
4449            );
4450        }
4451        // A `1.2.` whose optional dot would eat the separator backtracks like
4452        // Python's regex; a marker with nothing after the whitespace is none.
4453        assert_eq!(
4454            super::split_list_marker("1.2.\tx"),
4455            Some(("1.2.", "x", true))
4456        );
4457        assert_eq!(super::split_list_marker("1. "), None);
4458        assert_eq!(super::split_list_marker("• "), None);
4459        assert_eq!(
4460            text_of(&super::list_item_node("Plain item", loc, false)),
4461            (false, 0, "Plain item".into(), Some("·".into()))
4462        );
4463    }
4464
4465    #[test]
4466    fn code_language_label_above_code_is_detected() {
4467        // A bare "XML" token directly above a code box is a language label; a real
4468        // heading above the same code is not; a language word with no code below is
4469        // left alone.
4470        let label = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
4471        let code = region("code", 0.7, 77.0, 552.0, 290.0, 640.0);
4472        let heading = region("section_header", 0.9, 76.0, 500.0, 260.0, 512.0);
4473        let cells = vec![
4474            cell("XML", 78.0, 541.0, 94.0, 548.0),       // inside `label`
4475            cell("Overview", 78.0, 501.0, 250.0, 511.0), // inside `heading`
4476        ];
4477        let drop = super::code_language_labels(&[label, code, heading], &cells);
4478        assert_eq!(drop, vec![true, false, false], "only the label is consumed");
4479
4480        // Same label with no code region present → not consumed.
4481        let label2 = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
4482        let only = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
4483        assert_eq!(super::code_language_labels(&[label2], &only), vec![false]);
4484
4485        // A label swallowed into the top of a wider code box (negative gap) is still
4486        // recognized.
4487        let inside_lbl = region("text", 0.9, 76.0, 540.0, 96.0, 549.0);
4488        let wide_code = region("code", 0.7, 63.0, 531.0, 320.0, 654.0);
4489        let cells2 = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
4490        assert_eq!(
4491            super::code_language_labels(&[inside_lbl, wide_code], &cells2),
4492            vec![true, false]
4493        );
4494
4495        assert!(super::is_code_language("XML") && super::is_code_language("c#"));
4496        assert!(!super::is_code_language("Configure") && !super::is_code_language("XML schema"));
4497    }
4498
4499    #[test]
4500    fn code_region_text_keeps_lines_and_indentation() {
4501        // Three source lines; each glyph is 6 units wide (width / chars = 6), so the
4502        // `int X;` line indented to x=22 is (22-10)/6 = 2 spaces in.
4503        let region = Region {
4504            label: "code",
4505            score: 1.0,
4506            l: 0.0,
4507            t: -5.0,
4508            r: 100.0,
4509            b: 40.0,
4510        };
4511        let cells = vec![
4512            cell("struct P {", 10.0, 0.0, 70.0, 10.0),
4513            cell("int X;", 22.0, 12.0, 58.0, 22.0),
4514            cell("}", 10.0, 24.0, 16.0, 34.0),
4515        ];
4516        assert_eq!(code_region_text(&region, &cells), "struct P {\n  int X;\n}");
4517    }
4518
4519    #[test]
4520    fn code_region_text_tightens_punctuation_without_eating_indentation() {
4521        // A fluent `.Foo()` line at x=22 (2 chars in). Per-line tightening must not
4522        // consume the leading indent space by matching " ." across it.
4523        let region = Region {
4524            label: "code",
4525            score: 1.0,
4526            l: 0.0,
4527            t: -5.0,
4528            r: 100.0,
4529            b: 40.0,
4530        };
4531        let cells = vec![
4532            cell("builder", 10.0, 0.0, 52.0, 10.0),
4533            // pdfium spaced the call: ".Foo (x)" tightens to ".Foo(x)", still 2-indented.
4534            cell(".Foo (x)", 22.0, 12.0, 70.0, 22.0),
4535        ];
4536        assert_eq!(code_region_text(&region, &cells), "builder\n  .Foo(x)");
4537    }
4538
4539    #[test]
4540    fn code_region_text_orders_out_of_order_cells_and_ignores_blank_lines() {
4541        let region = Region {
4542            label: "code",
4543            score: 1.0,
4544            l: 0.0,
4545            t: -5.0,
4546            r: 100.0,
4547            b: 60.0,
4548        };
4549        // Fed bottom-up and with a whitespace-only cell; output is top-down, no blank.
4550        let cells = vec![
4551            cell("b();", 10.0, 24.0, 34.0, 34.0),
4552            cell("   ", 10.0, 12.0, 20.0, 22.0),
4553            cell("a();", 10.0, 0.0, 34.0, 10.0),
4554        ];
4555        assert_eq!(code_region_text(&region, &cells), "a();\nb();");
4556        // No code cells → empty, so the caller falls back to the prose text.
4557        assert_eq!(code_region_text(&region, &[]), "");
4558    }
4559
4560    fn para(text: &str) -> Node {
4561        Node::Paragraph { text: text.into() }
4562    }
4563
4564    /// Run a node sequence through [`StreamAssembler`] with the given page splits
4565    /// and assert the flushed result equals one-shot [`merge_continuations`].
4566    fn assert_stream_eq(nodes: &[Node], splits: &[usize]) {
4567        let mut want = nodes.to_vec();
4568        merge_continuations(&mut want);
4569
4570        let mut asm = StreamAssembler::new();
4571        let mut got = Vec::new();
4572        let mut start = 0;
4573        for &end in splits {
4574            got.extend(asm.push(nodes[start..end].to_vec()));
4575            start = end;
4576        }
4577        got.extend(asm.push(nodes[start..].to_vec()));
4578        got.extend(asm.finish());
4579        assert_eq!(got, want, "stream assembly diverged (splits={splits:?})");
4580    }
4581
4582    #[test]
4583    fn stream_assembler_matches_merge_continuations() {
4584        // Open fragment + lowercase continuation split across a page boundary.
4585        let cross = [para("the definition of"), para("lists in scope")];
4586        assert_stream_eq(&cross, &[1]);
4587        assert_stream_eq(&cross, &[]);
4588
4589        // Continuation that wraps around a figure (+ its caption) on the boundary.
4590        let wrap = [
4591            para("the wing type that is"),
4592            Node::Picture {
4593                caption: None,
4594                caption_href: None,
4595                image: None,
4596                classification: None,
4597                caption_parent: Default::default(),
4598            },
4599            para("Fig. 1. a diagram"),
4600            para("the most common kind"),
4601        ];
4602        for splits in [&[][..], &[1][..], &[2][..], &[3][..], &[1, 3][..]] {
4603            assert_stream_eq(&wrap, splits);
4604        }
4605
4606        // A heading between fragments blocks the merge (must still flush correctly).
4607        let blocked = [
4608            para("ends mid word and"),
4609            Node::Heading {
4610                level: 2,
4611                text: "New Section".into(),
4612            },
4613            para("more body here"),
4614        ];
4615        for splits in [&[][..], &[1][..], &[2][..]] {
4616            assert_stream_eq(&blocked, splits);
4617        }
4618
4619        // A chain across three pages: each page is one open lowercase fragment.
4620        let chain = [
4621            para("alpha beta"),
4622            para("gamma delta"),
4623            para("epsilon zeta"),
4624        ];
4625        assert_stream_eq(&chain, &[1, 2]);
4626    }
4627
4628    #[test]
4629    fn clean_text_dehyphenates_and_normalizes_typography() {
4630        // U+0002 line-wrap hyphen + the join space → merged word (like docling).
4631        assert_eq!(clean_text("com\u{2} pact"), "compact");
4632        assert_eq!(clean_text("end-to\u{2} end deep"), "end-toend deep");
4633        // A stray wrap hyphen (no following join) is dropped.
4634        assert_eq!(clean_text("word\u{2}"), "word");
4635        // Typographic punctuation → ASCII: every curly quote becomes `'`
4636        // (docling-parse's sanitizer table), a literal `"` stays.
4637        assert_eq!(
4638            clean_text("Graph\u{2019}s \u{201c}x\u{201d} \"y\""),
4639            "Graph's 'x' \"y\""
4640        );
4641        assert_eq!(clean_text("a\u{2026}"), "a...");
4642        // The docling-parse sanitizer's internal spacing is preserved as
4643        // placed; line breaks/tabs normalize to a space, ends trim.
4644        assert_eq!(clean_text("a   b\nc"), "a   b c");
4645    }
4646
4647    /// docling#4064: a form's children are emitted together where the form
4648    /// sits in the top-level order, not interleaved with surrounding text.
4649    #[test]
4650    fn form_children_stay_together_in_reading_order() {
4651        let reg = |label: &'static str, l: f32, t: f32, r: f32, b: f32| Region {
4652            label,
4653            score: 0.9,
4654            l,
4655            t,
4656            r,
4657            b,
4658        };
4659        // Page: intro text, then a form spanning the left column with two
4660        // fields and a table inside, while a right-column paragraph sits
4661        // level with the form's first field (it would otherwise be read
4662        // between the form's children).
4663        let mut items = vec![
4664            reg("text", 50.0, 50.0, 550.0, 70.0),    // 0 intro
4665            reg("form", 50.0, 100.0, 300.0, 400.0),  // 1 container
4666            reg("text", 60.0, 110.0, 290.0, 130.0),  // 2 field A (child)
4667            reg("text", 320.0, 110.0, 550.0, 130.0), // 3 right column paragraph
4668            reg("table", 60.0, 150.0, 290.0, 300.0), // 4 table (child)
4669            reg("text", 60.0, 320.0, 290.0, 340.0),  // 5 field B (child)
4670            reg("text", 50.0, 450.0, 550.0, 470.0),  // 6 outro
4671        ];
4672        let cids = super::cluster_cids(&items, &[]);
4673        super::order_with_containers(&mut items, &cids, 600.0, 800.0, |r| r);
4674        let order: Vec<(&str, f32)> = items.iter().map(|r| (r.label, r.t)).collect();
4675        // The form block (container, then its children top-down) is one unit.
4676        let form_pos = order.iter().position(|(l, _)| *l == "form").unwrap();
4677        assert_eq!(
4678            &order[form_pos..form_pos + 4],
4679            &[
4680                ("form", 100.0),
4681                ("text", 110.0),
4682                ("table", 150.0),
4683                ("text", 320.0)
4684            ]
4685        );
4686        assert_eq!(order[0], ("text", 50.0));
4687        assert_eq!(order[order.len() - 1], ("text", 450.0));
4688        // Without a container the plain order interleaves by geometry.
4689        let mut flat: Vec<Region> = items
4690            .iter()
4691            .filter(|r| r.label != "form")
4692            .cloned()
4693            .collect();
4694        let cids = super::cluster_cids(&flat, &[]);
4695        super::order_regions(&mut flat, &cids, 600.0, 800.0, |r| r);
4696        assert_ne!(
4697            flat.iter().map(|r| r.t).collect::<Vec<_>>(),
4698            order
4699                .iter()
4700                .filter(|(l, _)| *l != "form")
4701                .map(|(_, t)| *t)
4702                .collect::<Vec<_>>()
4703        );
4704    }
4705
4706    /// docling#3906: a picture inside a table lands in the covering cell,
4707    /// chosen by the picture's inferred grid position when cell boxes overlap.
4708    #[test]
4709    fn picture_matches_the_cell_at_its_grid_position() {
4710        let cell = |r: usize, c: usize, bbox: [f32; 4]| docling_core::TableCell {
4711            text: format!("r{r}c{c}"),
4712            bbox: Some(bbox),
4713            start_row: r,
4714            start_col: c,
4715            row_span: 1,
4716            col_span: 1,
4717            column_header: false,
4718            row_header: false,
4719            row_section: false,
4720        };
4721        // 2×2 grid; the (1,0) cell box is generous and also covers the picture.
4722        let cells = vec![
4723            cell(0, 0, [0.0, 0.0, 100.0, 50.0]),
4724            cell(0, 1, [100.0, 0.0, 200.0, 50.0]),
4725            cell(1, 0, [0.0, 50.0, 100.0, 100.0]),
4726            cell(1, 1, [100.0, 50.0, 200.0, 100.0]),
4727        ];
4728        let pic = Region {
4729            label: "picture",
4730            score: 0.9,
4731            l: 110.0,
4732            t: 60.0,
4733            r: 190.0,
4734            b: 95.0,
4735        };
4736        assert_eq!(super::match_picture_to_cell(&pic, &cells), Some((1.0, 3)));
4737        // A picture only half inside any cell is not nested.
4738        let straddling = Region {
4739            label: "picture",
4740            score: 0.9,
4741            l: 60.0,
4742            t: 60.0,
4743            r: 160.0,
4744            b: 95.0,
4745        };
4746        assert_eq!(super::match_picture_to_cell(&straddling, &cells), None);
4747    }
4748
4749    /// docling#4052 (2.122): a line-final dash fuses the wrapped word only
4750    /// when attached to it; a detached dash is a literal and the lines join
4751    /// with a space.
4752    #[test]
4753    fn line_final_hyphen_fuses_only_when_attached_to_a_word() {
4754        let line = |text: &str, t: f32| TextCell {
4755            text: text.to_string(),
4756            l: 0.0,
4757            t,
4758            r: 100.0,
4759            b: t + 10.0,
4760        };
4761        // `algo-` / `rithms`: attached hyphen, alnum on both sides → fused.
4762        assert_eq!(
4763            cells_text(vec![&line("algo-", 0.0), &line("rithms", 12.0)]),
4764            "algorithms"
4765        );
4766        // `pp. 545-` / `561`: attached, digits count as alnum → `545561` (upstream).
4767        assert_eq!(
4768            cells_text(vec![&line("pp. 545-", 0.0), &line("561", 12.0)]),
4769            "pp. 545561"
4770        );
4771        // A dash after whitespace — a separator or a lone `-` cell — is kept and
4772        // the lines take the ordinary joining space.
4773        assert_eq!(
4774            cells_text(vec![&line("range -", 0.0), &line("wide", 12.0)]),
4775            "range - wide"
4776        );
4777        assert_eq!(
4778            cells_text(vec![&line("-", 0.0), &line("item", 12.0)]),
4779            "- item"
4780        );
4781        // Attached but the next line opens with no word (`x-` / `...`): dash
4782        // kept and, as before, no separating space.
4783        assert_eq!(
4784            cells_text(vec![&line("x-", 0.0), &line("...", 12.0)]),
4785            "x-..."
4786        );
4787    }
4788
4789    #[test]
4790    fn lam_alef_only_swaps_a_genuinely_reversed_ligature() {
4791        // A mid-word `alef-variant + lam` is pdfium's reversed lam-alef ligature and
4792        // is swapped back to logical `lam + alef-variant` (`ب أ ل` → `ب ل أ`).
4793        assert_eq!(
4794            clean_text("\u{0628}\u{0623}\u{0644}"),
4795            "\u{0628}\u{0644}\u{0623}"
4796        );
4797        // But when the alef-variant is *already* preceded by a lam it is the logical
4798        // ligature `لآ`; the following lam is the next syllable's letter and must not
4799        // move. `التعلم الآلي` must stay `الآلي`, not become `اللآي`.
4800        assert_eq!(
4801            clean_text("\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"),
4802            "\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"
4803        );
4804    }
4805
4806    /// The #419 page, in points: three layout boxes over one paragraph, two of
4807    /// them ending partway through a line. The sliced lines miss the 0.2 claim
4808    /// and become orphans; the third model box starts above the second orphan,
4809    /// so unfitted the reading order emits that box first and strands the line.
4810    fn sliced_paragraph() -> (Vec<Region>, Vec<TextCell>) {
4811        let line = |text: &str, t: f32, r: f32| cell(text, 60.0, t, r, t + 11.0);
4812        let cells = vec![
4813            line("The mission of this series is to improve", 135.0, 458.0),
4814            line("The books in this series are technical,", 147.0, 458.0),
4815            line("substantial. The authors are", 159.0, 458.0),
4816            line("highly experienced craftsmen and", 171.5, 458.0), // sliced: 1.5/11 under box A
4817            line("actually works in practice, as opposed", 185.0, 458.0),
4818            line("about what the author has done, not", 197.0, 458.0),
4819            line("about programming, there will be lots", 210.5, 458.0), // sliced: 1.5/11 under box B
4820            line("will be lots of case studies from real", 223.0, 206.0), // C's line
4821        ];
4822        let regions = vec![
4823            region("text", 0.9, 60.0, 132.0, 458.0, 173.0), // A: three lines + a sliver of the 4th
4824            region("text", 0.9, 60.0, 184.0, 458.0, 212.0), // B: two lines + a sliver of the 7th
4825            region("text", 0.9, 60.0, 216.0, 206.0, 227.0), // C: last line, box opening 5.5pt too early
4826        ];
4827        (regions, cells)
4828    }
4829
4830    fn ordered_texts(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
4831        let mut items: Vec<Region> = regions.to_vec();
4832        let cids = super::cluster_cids(&items, cells);
4833        super::order_regions(&mut items, &cids, 500.0, 700.0, |r| r);
4834        super::region_texts_exclusive(&items, cells)
4835            .into_iter()
4836            .map(|t| t.chars().take(9).collect())
4837            .collect()
4838    }
4839
4840    /// #419: fitted to its cells, a model box that cut a line in half no longer
4841    /// overlaps the orphan that line became, so the orphan orders where it
4842    /// reads; unfitted, the same page strands the line after the paragraph.
4843    #[test]
4844    fn fitting_boxes_to_cells_puts_a_sliced_line_back_in_order() {
4845        let (mut regions, cells) = sliced_paragraph();
4846        super::add_orphan_regions(&mut regions, &cells);
4847        assert_eq!(regions.len(), 5, "two orphan lines");
4848        // The defect, for the record: C (top 216) is not strictly below the
4849        // orphan at 210.5–221.5, so the graph orders C first.
4850        assert_eq!(
4851            ordered_texts(&regions, &cells).last().map(String::as_str),
4852            Some("about pro")
4853        );
4854
4855        super::fit_regions_to_cells(&mut regions, &cells);
4856        assert_eq!(regions.len(), 5);
4857        // A ends on its last claimed line, C starts on its only one.
4858        assert_eq!((regions[0].t, regions[0].b), (135.0, 170.0));
4859        assert_eq!((regions[2].t, regions[2].b), (223.0, 234.0));
4860        assert_eq!(
4861            ordered_texts(&regions, &cells),
4862            [
4863                "The missi",
4864                "highly ex",
4865                "actually ",
4866                "about pro",
4867                "will be l"
4868            ]
4869        );
4870    }
4871
4872    /// An orphan the fitted paragraph box surrounds (a short middle line the
4873    /// narrow model box missed while claiming the lines around it) is folded
4874    /// into the paragraph; an empty regular box goes away, a formula stays, a
4875    /// picture is never refitted, and a page with no cells is left untouched.
4876    #[test]
4877    fn fitting_folds_surrounded_orphans_and_drops_empty_regulars() {
4878        let wide = |text: &str, t: f32| cell(text, 60.0, t, 400.0, t + 11.0);
4879        let cells = vec![
4880            wide("first line of the paragraph", 100.0),
4881            cell("stray", 250.0, 112.0, 400.0, 123.0), // clear of the narrow box
4882            wide("third line of the paragraph", 124.0),
4883        ];
4884        let mut regions = vec![
4885            // Narrow box: claims the wide lines at 0.41, misses the short one.
4886            region("text", 0.9, 60.0, 98.0, 200.0, 136.0),
4887            region("section_header", 0.8, 60.0, 300.0, 200.0, 320.0), // no cells
4888            region("formula", 0.8, 60.0, 340.0, 200.0, 360.0),        // no cells, kept
4889            region("picture", 0.8, 0.0, 400.0, 500.0, 600.0),
4890        ];
4891        super::add_orphan_regions(&mut regions, &cells);
4892        assert_eq!(regions.len(), 5, "the short line became an orphan");
4893        super::fit_regions_to_cells(&mut regions, &cells);
4894        let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
4895        assert_eq!(labels, ["text", "formula", "picture"]);
4896        let para = &regions[0];
4897        assert_eq!(
4898            (para.l, para.t, para.r, para.b),
4899            (60.0, 100.0, 400.0, 135.0)
4900        );
4901        assert_eq!(
4902            super::region_texts_exclusive(&regions, &cells)[0],
4903            "first line of the paragraph stray third line of the paragraph"
4904        );
4905        assert_eq!(
4906            (regions[2].t, regions[2].b),
4907            (400.0, 600.0),
4908            "picture untouched"
4909        );
4910
4911        let mut untouched = vec![region("text", 0.9, 0.0, 0.0, 10.0, 10.0)];
4912        super::fit_regions_to_cells(&mut untouched, &[]);
4913        assert_eq!(untouched.len(), 1, "no cells yet: nothing dropped");
4914    }
4915}