Skip to main content

docling_pdf/
assemble.rs

1//! Layout-driven assembly: map detected [`Region`]s + text cells to a
2//! [`DoclingDocument`], mirroring docling's page-assembly + reading-order.
3//!
4//! Overlapping detections are resolved greedily by score, each text cell is
5//! assigned to its best-containing region, regions are ordered in reading order
6//! (two-column aware), and each becomes a typed node by its layout label.
7
8use docling_core::{CaptionParent, Node, PictureClass, PictureImage, Table};
9#[cfg(feature = "ml")]
10use image::RgbImage;
11
12use crate::layout::Region;
13use crate::pdfium_backend::{PdfPage, TextCell};
14
15fn area(l: f32, t: f32, r: f32, b: f32) -> f32 {
16    ((r - l).max(0.0)) * ((b - t).max(0.0))
17}
18
19/// Intersection area of two boxes.
20fn inter(a: &Region, l: f32, t: f32, r: f32, b: f32) -> f32 {
21    let il = a.l.max(l);
22    let it = a.t.max(t);
23    let ir = a.r.min(r);
24    let ib = a.b.min(b);
25    area(il, it, ir, ib)
26}
27
28/// Wrapper (structured-region) labels, ported from docling
29/// `LayoutPostprocessor.WRAPPER_TYPES`: a region that *contains* other regions
30/// and renders as a structured block (a table / table-of-contents index), not as
31/// its own flat text.
32fn is_wrapper(label: &str) -> bool {
33    matches!(
34        label,
35        "table" | "document_index" | "form" | "key_value_region"
36    )
37}
38
39/// Labels docling's table-structure (TableFormer) model runs on and that render
40/// as a Markdown table: a plain `table` and a `document_index` (a table of
41/// contents), which docling assembles as a `TableItem` too.
42pub fn is_table_like(label: &str) -> bool {
43    matches!(label, "table" | "document_index")
44}
45
46/// Greedily keep regions by descending score, dropping a region that is mostly
47/// covered by an already-kept one (RT-DETR emits overlapping duplicates).
48fn greedy(mut regions: Vec<Region>) -> Vec<Region> {
49    regions.sort_by(|a, b| b.score.total_cmp(&a.score));
50    let mut kept: Vec<Region> = Vec::new();
51    for r in regions {
52        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
53        let covered = kept.iter().any(|k| {
54            let i = inter(&r, k.l, k.t, k.r, k.b);
55            let ka = area(k.l, k.t, k.r, k.b).max(1.0);
56            // drop if most of r is inside k, or they strongly mutually overlap
57            i / ra > 0.7 || i / (ra + ka - i) > 0.5
58        });
59        if !covered {
60            kept.push(r);
61        }
62    }
63    kept
64}
65
66/// Resolve overlapping RT-DETR detections, ported from the bucket structure of
67/// docling's `LayoutPostprocessor`: regular, picture and wrapper clusters live in
68/// **separate** spatial indexes and are de-overlapped independently, so a
69/// high-score picture never suppresses a lower-score table or table-of-contents
70/// index (the redp5110 TOC that was otherwise replaced by a picture box). A
71/// cross-type pass first drops a picture that nearly coincides with a table
72/// (`_handle_cross_type_overlaps`), keeping the structured table.
73/// docling's `_remove_overlapping_clusters("picture")`: same-label picture
74/// detections whose boxes heavily overlap (IoU > 0.8, or either box > 80 %
75/// contained in the other) form one group, and a single survivor is kept per
76/// group. Survivor selection ports `_should_prefer_cluster` /
77/// `_select_best_cluster_from_group` with the picture params
78/// (`area_threshold` 2.0, `conf_threshold` 0.3): a candidate is rejected only
79/// when a rival is both comparable in size (candidate ≤ 2× its area) and
80/// clearly more confident (> 0.3); among the survivors the *larger* box wins
81/// unless it is > 0.3 less confident. Net effect on the corpus: a figure the
82/// detector proposes both whole and as its sub-panels (2206's four-thumbnail
83/// Figure 1) collapses to the whole-figure box, exactly like docling.
84pub(crate) fn dedup_pictures(regions: &mut Vec<Region>) {
85    remove_overlapping_specials(regions, |l| l == "picture", 2.0, 0.3);
86}
87
88/// docling's `_remove_overlapping_clusters` for one special bucket (the
89/// regions whose label satisfies `in_bucket`), with that bucket's
90/// `OVERLAP_PARAMS`: picture (2.0, 0.3) — see [`dedup_pictures`] — or wrapper
91/// (2.0, 0.2) for the table bucket in [`resolve`]. Regions outside the bucket
92/// are untouched; a bucket member overlapping nothing is always kept.
93fn remove_overlapping_specials(
94    regions: &mut Vec<Region>,
95    in_bucket: impl Fn(&str) -> bool,
96    area_threshold: f32,
97    conf_threshold: f32,
98) {
99    let idx: Vec<usize> = (0..regions.len())
100        .filter(|&i| in_bucket(regions[i].label))
101        .collect();
102    if idx.len() < 2 {
103        return;
104    }
105    // Union-find over the bucket.
106    let mut parent: Vec<usize> = (0..idx.len()).collect();
107    fn find(parent: &mut [usize], i: usize) -> usize {
108        let mut root = i;
109        while parent[root] != root {
110            root = parent[root];
111        }
112        let mut cur = i;
113        while parent[cur] != root {
114            let next = parent[cur];
115            parent[cur] = root;
116            cur = next;
117        }
118        root
119    }
120    let boxed = |r: &Region| (r.l, r.t, r.r, r.b);
121    for a in 0..idx.len() {
122        for b in (a + 1)..idx.len() {
123            let (ra, rb) = (&regions[idx[a]], &regions[idx[b]]);
124            let (al, at, ar, ab_) = boxed(ra);
125            let (bl, bt, br, bb) = boxed(rb);
126            let ix = (ar.min(br) - al.max(bl)).max(0.0);
127            let iy = (ab_.min(bb) - at.max(bt)).max(0.0);
128            let inter = ix * iy;
129            let aa = area(al, at, ar, ab_).max(f32::EPSILON);
130            let ba = area(bl, bt, br, bb).max(f32::EPSILON);
131            let iou = inter / (aa + ba - inter).max(f32::EPSILON);
132            if iou > 0.8 || inter / aa > 0.8 || inter / ba > 0.8 {
133                let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
134                if pa != pb {
135                    parent[pa] = pb;
136                }
137            }
138        }
139    }
140    // Per group, run docling's pairwise preference + larger-wins selection.
141    let mut groups: std::collections::HashMap<usize, Vec<usize>> = std::collections::HashMap::new();
142    for i in 0..idx.len() {
143        let root = find(&mut parent, i);
144        groups.entry(root).or_default().push(i);
145    }
146    let mut drop = vec![false; regions.len()];
147    for group in groups.values() {
148        if group.len() < 2 {
149            continue;
150        }
151        let area_of = |i: usize| {
152            let r = &regions[idx[i]];
153            area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
154        };
155        let mut best: Option<usize> = None;
156        for &cand in group {
157            let passes = group.iter().all(|&other| {
158                if other == cand {
159                    return true;
160                }
161                let area_ratio = area_of(cand) / area_of(other);
162                let conf_diff = regions[idx[other]].score - regions[idx[cand]].score;
163                !(area_ratio <= area_threshold && conf_diff > conf_threshold)
164            });
165            if passes {
166                best = Some(match best {
167                    None => cand,
168                    Some(cur) => {
169                        if area_of(cand) > area_of(cur)
170                            && regions[idx[cur]].score - regions[idx[cand]].score <= conf_threshold
171                        {
172                            cand
173                        } else {
174                            cur
175                        }
176                    }
177                });
178            }
179        }
180        // Every candidate rejected can't happen with docling's rule (rejection
181        // needs a strictly better rival); guard with highest score anyway.
182        let keep = best.unwrap_or_else(|| {
183            *group
184                .iter()
185                .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(&regions[idx[b]].score))
186                .expect("non-empty group")
187        });
188        for &i in group {
189            if i != keep {
190                drop[idx[i]] = true;
191            }
192        }
193    }
194    let mut keep_iter = drop.into_iter();
195    regions.retain(|_| !keep_iter.next().expect("aligned"));
196}
197
198/// docling's `_remove_overlapping_clusters("regular")` for what [`greedy`]
199/// leaves standing, run on OCR'd pages before the region-scoped OCR (found
200/// with #471's synthetic scan: every line of the page came out twice).
201///
202/// `greedy` keeps regions by descending score and drops a candidate mostly
203/// inside an already-kept one — so a *lower*-score block that contains
204/// several higher-score line boxes (RT-DETR's favourite reading of a sparse
205/// scanned page: every line, plus the paragraph) survives alongside them.
206/// On a digital page that is harmless: [`fit_regions_to_cells`] hands each
207/// text cell to one owner and drops the regions left empty. On a scanned
208/// page the cells don't exist yet — they come from OCR of *each* region's
209/// crop — so the block and its lines were each recognized, and the block,
210/// which then owned both cell sets, read every line twice.
211///
212/// Upstream never has this problem because its OCR runs over the bitmap
213/// before layout postprocessing and each cell is assigned once; the
214/// postprocessor's regular pass then groups clusters that overlap (IoU > 0.8,
215/// or either > 80 % contained in the other) with a union-find, keeps one
216/// survivor per group via `_should_prefer_cluster` /
217/// `_select_best_cluster_from_group` (`area_threshold` 1.3, `conf_threshold`
218/// 0.05; a LIST_ITEM beats a same-sized TEXT, a CODE box beats what it
219/// contains) and merges the losers' cells into it, whose box is then fitted
220/// to those cells. The equivalent for region-scoped OCR: one survivor per
221/// group, keeping its label and score, with the group's **union** box so the
222/// single crop still covers every merged line. Pictures and wrappers are not
223/// regulars and are left alone ([`dedup_pictures`], [`resolve`]).
224pub(crate) fn merge_overlapping_regulars(regions: &mut Vec<Region>) {
225    let idx: Vec<usize> = (0..regions.len())
226        .filter(|&i| regions[i].label != "picture" && !is_wrapper(regions[i].label))
227        .collect();
228    if idx.len() < 2 {
229        return;
230    }
231    let mut parent: Vec<usize> = (0..idx.len()).collect();
232    fn find(parent: &mut [usize], i: usize) -> usize {
233        let mut root = i;
234        while parent[root] != root {
235            root = parent[root];
236        }
237        let mut cur = i;
238        while parent[cur] != root {
239            let next = parent[cur];
240            parent[cur] = root;
241            cur = next;
242        }
243        root
244    }
245    for a in 0..idx.len() {
246        for b in (a + 1)..idx.len() {
247            let (ra, rb) = (&regions[idx[a]], &regions[idx[b]]);
248            let i = inter(ra, rb.l, rb.t, rb.r, rb.b);
249            let aa = area(ra.l, ra.t, ra.r, ra.b).max(f32::EPSILON);
250            let ba = area(rb.l, rb.t, rb.r, rb.b).max(f32::EPSILON);
251            if i / (aa + ba - i).max(f32::EPSILON) > 0.8 || i / aa > 0.8 || i / ba > 0.8 {
252                let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
253                if pa != pb {
254                    parent[pa] = pb;
255                }
256            }
257        }
258    }
259    let mut groups: std::collections::BTreeMap<usize, Vec<usize>> =
260        std::collections::BTreeMap::new();
261    for i in 0..idx.len() {
262        let root = find(&mut parent, i);
263        groups.entry(root).or_default().push(i);
264    }
265    const AREA_THRESHOLD: f32 = 1.3;
266    const CONF_THRESHOLD: f32 = 0.05;
267    let area_of = |i: usize| {
268        let r = &regions[idx[i]];
269        area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
270    };
271    // `_should_prefer_cluster(candidate, other)` with the regular params.
272    let prefer = |cand: usize, other: usize| -> bool {
273        let (c, o) = (&regions[idx[cand]], &regions[idx[other]]);
274        let area_ratio = area_of(cand) / area_of(other);
275        if c.label == "list_item" && o.label == "text" && (1.0 - area_ratio).abs() < 0.2 {
276            return true;
277        }
278        if c.label == "code" && inter(o, c.l, c.t, c.r, c.b) / area_of(other) > 0.8 {
279            return true;
280        }
281        !(area_ratio <= AREA_THRESHOLD && o.score - c.score > CONF_THRESHOLD)
282    };
283    let mut drop = vec![false; regions.len()];
284    let mut unions: Vec<(usize, (f32, f32, f32, f32))> = Vec::new();
285    for group in groups.values() {
286        if group.len() < 2 {
287            continue;
288        }
289        let mut best: Option<usize> = None;
290        for &cand in group {
291            if group
292                .iter()
293                .all(|&other| other == cand || prefer(cand, other))
294            {
295                best = Some(match best {
296                    None => cand,
297                    Some(cur)
298                        if area_of(cand) > area_of(cur)
299                            && regions[idx[cur]].score - regions[idx[cand]].score
300                                <= CONF_THRESHOLD =>
301                    {
302                        cand
303                    }
304                    Some(cur) => cur,
305                });
306            }
307        }
308        // docling falls back to the group's first cluster; the highest score
309        // is the deterministic equivalent for a set with no insertion order.
310        let keep = best.unwrap_or_else(|| {
311            *group
312                .iter()
313                .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(&regions[idx[b]].score))
314                .expect("non-empty group")
315        });
316        let mut u = (
317            f32::INFINITY,
318            f32::INFINITY,
319            f32::NEG_INFINITY,
320            f32::NEG_INFINITY,
321        );
322        for &i in group {
323            let r = &regions[idx[i]];
324            u = (u.0.min(r.l), u.1.min(r.t), u.2.max(r.r), u.3.max(r.b));
325            if i != keep {
326                drop[idx[i]] = true;
327            }
328        }
329        unions.push((idx[keep], u));
330    }
331    for (i, (l, t, r, b)) in unions {
332        let k = &mut regions[i];
333        (k.l, k.t, k.r, k.b) = (l, t, r, b);
334    }
335    let mut keep_iter = drop.into_iter();
336    regions.retain(|_| !keep_iter.next().expect("aligned"));
337}
338
339/// `intersection_over_union` of two regions.
340fn iou(a: &Region, b: &Region) -> f32 {
341    let i = inter(a, b.l, b.t, b.r, b.b);
342    let u = area(a.l, a.t, a.r, a.b) + area(b.l, b.t, b.r, b.b) - i;
343    if u > 0.0 {
344        i / u
345    } else {
346        0.0
347    }
348}
349
350/// docling's `_resolve_coincident_pairs` (#4059, 2.122): for every (loser,
351/// winner) pair at a near-identical box (IoU > 0.8) whose confidences are
352/// within 0.1 (`loser.score - winner.score < 0.1`), the loser label is dropped
353/// so the label with the richer downstream semantic survives. Nothing else —
354/// containment, area — is considered; a clearly more confident loser stays.
355fn coincident_losers(regions: &[Region], losers: &[usize], winners: &[usize]) -> Vec<usize> {
356    let mut out = Vec::new();
357    for &li in losers {
358        for &wi in winners {
359            if iou(&regions[li], &regions[wi]) > 0.8 && regions[li].score - regions[wi].score < 0.1
360            {
361                out.push(li);
362                break;
363            }
364        }
365    }
366    out
367}
368
369/// docling's `_handle_cross_type_overlaps` (2.122/2.123 shape): the layout
370/// model can emit one grounded region under several labels, and the picture /
371/// table / container buckets are de-overlapped independently, so such a region
372/// survives twice. Elect a winner for the near-identical pairs:
373///
374/// | pair                                  | loser     | winner              |
375/// |---------------------------------------|-----------|---------------------|
376/// | TABLE vs DOCUMENT_INDEX               | table     | document_index      |
377/// | PICTURE vs TABLE / DOCUMENT_INDEX     | picture   | the table-like      |
378/// | FORM / KEY_VALUE_REGION vs *surviving* TABLE / DOCUMENT_INDEX / PICTURE | container | structured element |
379///
380/// IoU (not containment) so a genuine small figure inside a large table region
381/// is not removed; the confidence tolerance keeps a clearly more confident
382/// loser (an earlier port dropped every coincident picture regardless).
383fn handle_cross_type_overlaps(regions: Vec<Region>) -> Vec<Region> {
384    let by = |pred: &dyn Fn(&str) -> bool| -> Vec<usize> {
385        (0..regions.len())
386            .filter(|&i| pred(regions[i].label))
387            .collect()
388    };
389    let tables = by(&|l| l == "table");
390    let doc_indices = by(&|l| l == "document_index");
391    let pictures = by(&|l| l == "picture");
392    let containers = by(&|l| matches!(l, "form" | "key_value_region"));
393    let mut drop = vec![false; regions.len()];
394    for i in coincident_losers(&regions, &tables, &doc_indices) {
395        drop[i] = true;
396    }
397    let table_like: Vec<usize> = tables.iter().chain(&doc_indices).copied().collect();
398    for i in coincident_losers(&regions, &pictures, &table_like) {
399        drop[i] = true;
400    }
401    let structured: Vec<usize> = table_like
402        .iter()
403        .chain(&pictures)
404        .copied()
405        .filter(|&i| !drop[i])
406        .collect();
407    for i in coincident_losers(&regions, &containers, &structured) {
408        drop[i] = true;
409    }
410    let mut drop = drop.into_iter();
411    let mut regions = regions;
412    regions.retain(|_| !drop.next().expect("aligned"));
413    regions
414}
415
416pub fn resolve(regions: Vec<Region>) -> Vec<Region> {
417    let regions = handle_cross_type_overlaps(regions);
418    // De-overlap each bucket on its own.
419    let pictures = greedy(
420        regions
421            .iter()
422            .filter(|r| r.label == "picture")
423            .cloned()
424            .collect(),
425    );
426    // Tables and containers are separate buckets since docling 2.123
427    // (`TABLE_TYPES` vs `CONTAINER_TYPES`, docling#4064): a form drawn around a
428    // table no longer competes with it for survival — the table nests inside
429    // the container instead (`order_with_containers`).
430    let mut tables = greedy(
431        regions
432            .iter()
433            .filter(|r| is_table_like(r.label))
434            .cloned()
435            .collect(),
436    );
437    // `greedy` only drops a table mostly inside a *more* confident one, so a
438    // low-score whole-page table proposed over the column tables it contains
439    // (a two-column glossary page: 0.53 over 0.71/0.67/0.66) survived next to them and every
440    // cell was emitted twice. docling's `_remove_overlapping_clusters(tables,
441    // "wrapper")` groups tables whose boxes overlap (IoU > 0.8, or either one
442    // > 80 % inside the other) and keeps one per group: run it on what
443    // `greedy` leaves, like `merge_overlapping_regulars` does for regulars.
444    remove_overlapping_specials(&mut tables, |_| true, 2.0, 0.2);
445    let containers = greedy(
446        regions
447            .iter()
448            .filter(|r| matches!(r.label, "form" | "key_value_region"))
449            .cloned()
450            .collect(),
451    );
452    let mut kept = greedy(
453        regions
454            .iter()
455            .filter(|r| r.label != "picture" && !is_wrapper(r.label))
456            .cloned()
457            .collect(),
458    );
459    dedup_nested_code(&mut kept);
460    kept.extend(pictures);
461    kept.extend(tables);
462    kept.extend(containers);
463    kept
464}
465
466/// Drop a regular region that is >80% contained in a surviving special region we
467/// render **as a single unit** — a table/table-of-contents index — ported from
468/// docling's "Remove regular clusters that are included in wrappers" step: the
469/// special absorbs it as a child (a table cell), so it must not also be emitted
470/// as its own paragraph/list-item. This stops the survey list-items from
471/// appearing both inside the detected table and again as bullets
472/// (`table_mislabeled_as_picture`).
473///
474/// `picture` regions are **not** in the swallow set: docling keeps a picture's
475/// contained clusters as the `PictureItem`'s *children* in the document JSON
476/// (`_set_cluster_children`, `ReadingOrderModel._add_child_elements`), while
477/// its `MarkdownPictureSerializer` prints only the caption and the image — the
478/// children never reach the Markdown (verified against the corpus groundtruth:
479/// `amt_handbook`'s in-figure callout labels are absent). So the regulars
480/// inside a picture stay in the region list, claim their cells and are fitted
481/// like any regular, and [`assemble_page`] lifts them out of the page's
482/// reading order into a [`Node::PictureChildren`] after their picture (see
483/// [`picture_parents`]). A line only partially under a figure box (straddling
484/// its border, ≤80 % contained) is no picture's child and is emitted, as
485/// docling emits it (#165).
486///
487/// `form` / `key_value_region` wrappers are deliberately **excluded**: this
488/// pipeline does not render them as a structured block (they are skipped), so
489/// their textual content comes precisely from the contained regular regions —
490/// A `page_footer` that is really the body of the last heading on the page
491/// becomes `text`.
492///
493/// The layout model labels the bottom margin by position as much as by
494/// content: a one-line paragraph that happens to sit where a running footer
495/// would — a CV's `Languages` line under its `## Languages` heading, the last
496/// entry of a section that runs to the page edge — comes back as
497/// `page_footer` (0.88 against 0.49 for `text` on the reporting file, with
498/// the fp32 model and either renderer alike), and Markdown drops furniture,
499/// so the document loses real content while the JSON keeps it under the
500/// wrong label. docling fails the same way. Deliberate deviation, gated so a
501/// real footer keeps its label: the nearest `section_header` above the footer
502/// must overlap it horizontally, end within 2.5 footer heights of its top,
503/// and be the **last body element** of its column — no other non-furniture
504/// region starts at or below the heading's bottom edge over the same span —
505/// and the footer must be a line of text, not a page number: at least 40 %
506/// of the page width. A running footer never follows a heading that has no
507/// body of its own, so the combination is the misread heading body.
508pub fn reclaim_heading_body_footers(regions: &mut [Region], page_w: f32) {
509    let n = regions.len();
510    let overlap_x = |a: &Region, b: &Region| a.r.min(b.r) - a.l.max(b.l) > 0.0;
511    for fi in 0..n {
512        let f = regions[fi].clone();
513        if f.label != "page_footer" || f.r - f.l < 0.4 * page_w {
514            continue;
515        }
516        let fh = (f.b - f.t).max(1.0);
517        // The nearest heading above the footer, over the footer's span.
518        let heading = (0..n)
519            .filter(|&j| {
520                let h = &regions[j];
521                j != fi && h.label == "section_header" && h.b <= f.t + 0.5 * fh && overlap_x(h, &f)
522            })
523            .min_by(|&a, &b| regions[b].b.total_cmp(&regions[a].b));
524        let Some(hi) = heading else {
525            continue;
526        };
527        let h = regions[hi].clone();
528        if f.t - h.b > 2.5 * fh {
529            continue;
530        }
531        // The heading must have no body of its own: nothing but the footer
532        // starts at or below its bottom edge over the heading's or footer's
533        // span (a heading whose paragraph follows is not this case, and a
534        // heading with the footer far below it was filtered above).
535        let has_body = (0..n).any(|j| {
536            let r = &regions[j];
537            j != fi
538                && j != hi
539                && !matches!(r.label, "page_footer" | "page_header")
540                && r.t >= h.b - 0.5 * fh
541                && (overlap_x(r, &h) || overlap_x(r, &f))
542        });
543        if has_body {
544            continue;
545        }
546        regions[fi].label = "text";
547    }
548}
549
550/// dropping those would erase the page (e.g. `right_to_left_03`'s form-heavy
551/// pages). Runs *after* [`drop_false_pictures`] so a phantom picture can't
552/// swallow real text on its way out.
553pub fn drop_contained_regulars(regions: &mut Vec<Region>) {
554    let specials: Vec<(f32, f32, f32, f32)> = regions
555        .iter()
556        .filter(|r| is_table_like(r.label))
557        .map(|r| (r.l, r.t, r.r, r.b))
558        .collect();
559    if specials.is_empty() {
560        return;
561    }
562    regions.retain(|r| {
563        if r.label == "picture" || is_wrapper(r.label) {
564            return true;
565        }
566        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
567        !specials
568            .iter()
569            .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.8)
570    });
571}
572
573/// docling's `_set_cluster_children` for pictures: a regular region (one that
574/// claims cells) whose box is > 80 % inside a picture's box
575/// (`intersection_over_self`) is that picture's child — not a page element
576/// (it leaves the reading order and the layout score), but an item under the
577/// `PictureItem` in the JSON. Returns, per region, the index of its parent
578/// picture: the smallest containing one when pictures nest (upstream would
579/// attach it to each).
580pub fn picture_parents(regions: &[Region]) -> Vec<Option<usize>> {
581    regions
582        .iter()
583        .map(|r| {
584            if !claims_cells(r) {
585                return None;
586            }
587            let ra = area(r.l, r.t, r.r, r.b).max(1.0);
588            regions
589                .iter()
590                .enumerate()
591                .filter(|(_, p)| p.label == "picture" && inter(r, p.l, p.t, p.r, p.b) / ra > 0.8)
592                .min_by(|(_, a), (_, b)| {
593                    area(a.l, a.t, a.r, a.b).total_cmp(&area(b.l, b.t, b.r, b.b))
594                })
595                .map(|(i, _)| i)
596        })
597        .collect()
598}
599
600/// True for a bare, single-token source-code language label (`XML`, `C#`, `JSON`,
601/// `bash`, …) — the little header the docs render above a code block. Matched
602/// case-insensitively; anything with whitespace or longer than a token is out.
603fn is_code_language(t: &str) -> bool {
604    let t = t.trim();
605    if t.is_empty() || t.chars().any(char::is_whitespace) || t.chars().count() > 12 {
606        return false;
607    }
608    const LANGS: &[&str] = &[
609        "xml",
610        "html",
611        "xhtml",
612        "json",
613        "jsonc",
614        "yaml",
615        "yml",
616        "toml",
617        "ini",
618        "c#",
619        "csharp",
620        "f#",
621        "fsharp",
622        "vb",
623        "c",
624        "c++",
625        "cpp",
626        "java",
627        "kotlin",
628        "scala",
629        "go",
630        "golang",
631        "rust",
632        "swift",
633        "javascript",
634        "js",
635        "typescript",
636        "ts",
637        "jsx",
638        "tsx",
639        "python",
640        "py",
641        "ruby",
642        "rb",
643        "php",
644        "perl",
645        "lua",
646        "r",
647        "dart",
648        "bash",
649        "sh",
650        "shell",
651        "powershell",
652        "zsh",
653        "batch",
654        "cmd",
655        "sql",
656        "tsql",
657        "plsql",
658        "graphql",
659        "dockerfile",
660        "makefile",
661        "css",
662        "scss",
663        "sass",
664        "less",
665        "markdown",
666        "md",
667        "tex",
668        "latex",
669        "diff",
670        "proto",
671        "razor",
672        "cshtml",
673        "xaml",
674        "aspx",
675        "http",
676    ];
677    let lower = t.to_ascii_lowercase();
678    LANGS.contains(&lower.as_str())
679}
680
681/// Mark the region indices that are a code block's **language label** — a bare
682/// `XML`/`C#`/… token sitting directly above a `code` region — so they are consumed
683/// rather than emitted as their own stray paragraph/heading. The label may also be
684/// captured inside a wider code box (rendered as the fence's first line); dropping
685/// the standalone copy just removes the duplicate.
686fn code_language_labels(regions: &[Region], cells: &[TextCell]) -> Vec<bool> {
687    let mut drop = vec![false; regions.len()];
688    for (i, r) in regions.iter().enumerate() {
689        if matches!(r.label, "code" | "picture" | "table") {
690            continue;
691        }
692        if !is_code_language(&region_text(r, cells)) {
693            continue;
694        }
695        // The label sits just above the code (a blank line's gap) or is swallowed
696        // into the top of a wider code box; either way it is that block's label.
697        // The window is generous because the label's own font is small, so a
698        // one-line gap is several times its height.
699        let line_h = (r.b - r.t).abs().max(1.0);
700        let window = (line_h * 4.0).max(28.0);
701        let labels_code = regions.iter().enumerate().any(|(j, c)| {
702            if j == i || c.label != "code" {
703                return false;
704            }
705            let gap = c.t - r.b; // >0 when the code is below the label
706            let h_overlap = (r.r.min(c.r) - r.l.max(c.l)).max(0.0);
707            gap > -line_h * 3.0 && gap < window && h_overlap > 0.0
708        });
709        if labels_code {
710            drop[i] = true;
711        }
712    }
713    drop
714}
715
716/// Collapse `code` regions where one is nested inside another, keeping the larger.
717///
718/// RT-DETR sometimes emits a tight code box *and* a wider near-duplicate that also
719/// captures the block's language label (`XML`, `C#`, …). When the tight box scores
720/// higher it is kept first, and the wider container — not "mostly inside" the tight
721/// box — survives [`resolve`]'s greedy pass, so the block is emitted twice. Keeping
722/// the **larger** box (rather than dropping it) collapses the pair without leaking
723/// the container's extra cells back out as orphan text, since the larger box still
724/// covers every cell. Restricted to `code` so genuinely distinct nested regions of
725/// other kinds are untouched.
726fn dedup_nested_code(kept: &mut Vec<Region>) {
727    let mut drop = vec![false; kept.len()];
728    for i in 0..kept.len() {
729        if kept[i].label != "code" {
730            continue;
731        }
732        let ai = area(kept[i].l, kept[i].t, kept[i].r, kept[i].b).max(1.0);
733        for j in 0..kept.len() {
734            if i == j || drop[j] || kept[j].label != "code" {
735                continue;
736            }
737            let aj = area(kept[j].l, kept[j].t, kept[j].r, kept[j].b).max(1.0);
738            // Drop i when it is mostly inside a strictly larger code box j.
739            let overlap = inter(&kept[i], kept[j].l, kept[j].t, kept[j].r, kept[j].b);
740            if aj > ai && overlap / ai > 0.7 {
741                drop[i] = true;
742                break;
743            }
744        }
745    }
746    let mut keep = drop.iter();
747    kept.retain(|_| !*keep.next().unwrap());
748}
749
750/// Fraction of the page's non-empty text cells that some detected region
751/// claims (>0.2 intersection-over-self, docling's assignment rule). 1.0 for a
752/// page without text cells.
753///
754/// The int8-layout guard keys off this: a dense digital page whose detections
755/// cover almost none of its text is the signature of quantized confidences
756/// flipping under the 0.5 label thresholds on this CPU's kernels — not of a
757/// genuinely empty layout — and is worth re-running on the fp32 graph.
758pub fn layout_cell_coverage(regions: &[Region], cells: &[TextCell]) -> f32 {
759    let mut total = 0usize;
760    let mut covered = 0usize;
761    for c in cells {
762        if c.text.trim().is_empty() {
763            continue;
764        }
765        total += 1;
766        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
767        if regions
768            .iter()
769            .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
770        {
771            covered += 1;
772        }
773    }
774    if total == 0 {
775        1.0
776    } else {
777        covered as f32 / total as f32
778    }
779}
780
781/// Append `text` regions for cells the layout left uncovered ("orphan cells"),
782/// the way docling's `LayoutPostprocessor` does (`create_orphan_clusters`): any
783/// non-empty cell that no kept region covers (>50% of the cell's area) becomes a
784/// text region of its own, so text the detector missed (a stray `.`, a small
785/// label) is still emitted instead of silently dropped. Adjacent orphan cells on a
786/// line are merged so a missed paragraph doesn't shatter into one block per line.
787pub fn add_orphan_regions(regions: &mut Vec<Region>, cells: &[TextCell]) {
788    // docling assigns each cell to its single best-overlapping cluster at
789    // intersection-over-self > 0.2 and serializes exactly the assigned cells —
790    // and since [`region_texts_exclusive`] now emits under that very rule, the
791    // claim test here matches it: any cell over 0.2 will actually render in
792    // its best region, everything else becomes an orphan. Completeness by
793    // construction, with no (0.2, 0.5] hole (the old > 0.5 serializer needed
794    // the claim test raised to > 0.5 to keep right_to_left_03's `20300` from
795    // vanishing; the exclusive port closes that structurally).
796    //
797    // Only *regular* clusters claim cells: docling's `_find_unassigned_cells`
798    // walks `regular_clusters` alone, so a cell under a `picture` or a wrapper
799    // (`table`/`document_index`/`form`/`key_value_region`) that no regular
800    // cluster covers still becomes an orphan text cluster (#165). The orphans
801    // that end up *fully* inside the special are re-dropped by
802    // [`drop_contained_regulars`] (docling's Markdown drops them the same way
803    // — a picture's children never reach its `MarkdownPictureSerializer`
804    // output, a table's text renders through the reconstructed grid). The
805    // observable fix is the border-straddlers: a line only partially under a
806    // figure box used to lose its cells to the picture's 0.2 claim and vanish
807    // — now it forms an orphan region and is emitted, as docling does.
808    let assigned = |c: &TextCell| {
809        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
810        regions
811            .iter()
812            .filter(|r| r.label != "picture" && !is_wrapper(r.label))
813            .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
814    };
815    // Collect orphan cells (non-empty, unassigned), in page order.
816    let mut orphans: Vec<&TextCell> = cells
817        .iter()
818        .filter(|c| !c.text.trim().is_empty() && !assigned(c))
819        .collect();
820    if orphans.is_empty() {
821        return;
822    }
823    orphans.sort_by(|a, b| a.t.total_cmp(&b.t).then(a.l.total_cmp(&b.l)));
824    // Merge cells that sit on the same line and nearly touch into one region, so a
825    // dropped multi-word line stays one block (docling's refinement merges these).
826    let mut merged: Vec<Region> = Vec::new();
827    for c in orphans {
828        let h = (c.b - c.t).abs().max(1.0);
829        if let Some(last) = merged.last_mut() {
830            let same_line = (last.t - c.t).abs() < h * 0.5;
831            let touching = c.l <= last.r + h && c.l >= last.l - h;
832            if same_line && touching {
833                last.l = last.l.min(c.l);
834                last.r = last.r.max(c.r);
835                last.t = last.t.min(c.t);
836                last.b = last.b.max(c.b);
837                continue;
838            }
839        }
840        merged.push(Region {
841            label: "text",
842            score: 0.0,
843            l: c.l,
844            t: c.t,
845            r: c.r,
846            b: c.b,
847        });
848    }
849    regions.extend(merged);
850}
851
852/// Demote a `picture` region that is really a **text panel** — a paragraph block
853/// the layout model boxed as a figure because it is typeset on a colored
854/// background (terms-and-conditions callouts, quote boxes) — into ordinary
855/// `text` regions, one per paragraph, so its words are read instead of shipped
856/// as pixels. docling loses this text the same way (cells assigned to a picture
857/// cluster are never serialized); this is a deliberate improvement, not parity.
858///
859/// The gate is conservative so a genuine figure keeps its crop: the region must
860/// contain at least three text lines whose median width spans most of the panel
861/// (axis labels and chat bubbles are narrow and varied) and whose cells cover a
862/// substantial fraction of its area (a photo or chart with sparse labels does
863/// not). Paragraph boundaries are re-derived from the line pitch: a vertical gap
864/// clearly larger than the panel's own leading starts a new `text` region, so
865/// the panel doesn't collapse into one giant paragraph.
866///
867/// Works on any cell source — the digital text layer or OCR lines recognized
868/// from the picture crop — so the native and browser paths, with or without
869/// force-OCR, demote identically.
870pub fn recover_text_panels(regions: &mut Vec<Region>, cells: &[TextCell]) {
871    // A *captioned* picture is a genuine figure whatever it contains — the
872    // corpus is full of document screenshots ("Figure 3: …" above a page
873    // image) that are exactly as dense and wide as a text panel. Only an
874    // uncaptioned picture is a demotion candidate.
875    let captioned: Vec<bool> = regions
876        .iter()
877        .map(|r| {
878            r.label == "picture"
879                && regions.iter().any(|c| {
880                    c.label == "caption" && c.r.min(r.r) - c.l.max(r.l) > 0.0 && {
881                        let gap = if c.t >= r.b {
882                            c.t - r.b
883                        } else if r.t >= c.b {
884                            r.t - c.b
885                        } else {
886                            f32::MAX // vertically overlapping: not a caption
887                        };
888                        gap <= 25.0
889                    }
890                })
891        })
892        .collect();
893    let mut out: Vec<Region> = Vec::with_capacity(regions.len());
894    // Synthesized paragraphs and the demoted panels' boxes are kept separate
895    // from `out` until the end: the dedup filter below must not confuse a
896    // paragraph we just built with a pre-existing region inside the panel.
897    let mut demoted_paras: Vec<Region> = Vec::new();
898    let mut demoted_boxes: Vec<(f32, f32, f32, f32)> = Vec::new();
899    for (i, r) in regions.drain(..).enumerate() {
900        if r.label != "picture" || captioned[i] {
901            out.push(r);
902            continue;
903        }
904        let inside: Vec<&TextCell> = cells
905            .iter()
906            .filter(|c| {
907                !c.text.trim().is_empty() && {
908                    let ca = area(c.l, c.t, c.r, c.b).max(1.0);
909                    inter(&r, c.l, c.t, c.r, c.b) / ca > 0.5
910                }
911            })
912            .collect();
913        // Group the contained cells into lines by vertical overlap (the same
914        // rule region_text orders by), tracking each line's union box.
915        let mut lines: Vec<(f32, f32, f32, f32)> = Vec::new(); // (t, b, l, r)
916        for c in &inside {
917            let (ct, cb) = (c.t.min(c.b), c.t.max(c.b));
918            match lines.iter_mut().find(|(lt, lb, _, _)| {
919                let ov = cb.min(*lb) - ct.max(*lt);
920                ov > 0.5 * (cb - ct).min(*lb - *lt).max(1.0)
921            }) {
922                Some((lt, lb, ll, lr)) => {
923                    *lt = lt.min(ct);
924                    *lb = lb.max(cb);
925                    *ll = ll.min(c.l);
926                    *lr = lr.max(c.r);
927                }
928                None => lines.push((ct, cb, c.l, c.r)),
929            }
930        }
931        if lines.len() < 3 {
932            out.push(r);
933            continue;
934        }
935        let panel_w = (r.r - r.l).max(1.0);
936        let coverage = inside.iter().map(|c| area(c.l, c.t, c.r, c.b)).sum::<f32>()
937            / area(r.l, r.t, r.r, r.b).max(1.0);
938        let mut widths: Vec<f32> = lines.iter().map(|(_, _, l, rr)| rr - l).collect();
939        widths.sort_by(f32::total_cmp);
940        // A figure's text is ragged: a title line, small axis/tick labels, and
941        // OCR boxes over the plot area come out at wildly different heights,
942        // whereas a real text panel is set in one face with constant leading.
943        // Require near-uniform line heights (median absolute deviation ≤ 35%
944        // of the median) so an uncaptioned chart keeps its crop even when its
945        // labels are dense enough to pass the coverage gate (#173) — garbled
946        // OCR of its bars is not content.
947        let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
948        heights.sort_by(f32::total_cmp);
949        let h_med = heights[heights.len() / 2].max(1.0);
950        let mut devs: Vec<f32> = heights.iter().map(|h| (h - h_med).abs()).collect();
951        devs.sort_by(f32::total_cmp);
952        let uniform = devs[devs.len() / 2] <= 0.35 * h_med;
953        let text_panel = coverage >= 0.2 && widths[widths.len() / 2] >= 0.45 * panel_w && uniform;
954        if !text_panel {
955            out.push(r);
956            continue;
957        }
958        lines.sort_by(|a, b| a.0.total_cmp(&b.0));
959        let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
960        heights.sort_by(f32::total_cmp);
961        let h = heights[heights.len() / 2].max(1.0);
962        let mut gaps: Vec<f32> = lines
963            .windows(2)
964            .map(|w| (w[1].0 - w[0].1).max(0.0))
965            .collect();
966        gaps.sort_by(f32::total_cmp);
967        let leading = if gaps.is_empty() {
968            0.0
969        } else {
970            gaps[gaps.len() / 2]
971        };
972        let brk = (1.8 * leading).max(0.75 * h);
973        let mut para: Option<(f32, f32, f32, f32)> = None; // (l, t, r, b) union
974        for (t, b, l, rr) in &lines {
975            match &mut para {
976                Some((pl, _, pr, pb)) if *t - *pb <= brk => {
977                    *pl = pl.min(*l);
978                    *pr = pr.max(*rr);
979                    *pb = pb.max(*b);
980                }
981                _ => {
982                    if let Some((pl, pt, pr, pb)) = para.take() {
983                        demoted_paras.push(Region {
984                            label: "text",
985                            score: r.score,
986                            l: pl,
987                            t: pt,
988                            r: pr,
989                            b: pb,
990                        });
991                    }
992                    para = Some((*l, *t, *rr, *b));
993                }
994            }
995        }
996        if let Some((pl, pt, pr, pb)) = para {
997            demoted_paras.push(Region {
998                label: "text",
999                score: r.score,
1000                l: pl,
1001                t: pt,
1002                r: pr,
1003                b: pb,
1004            });
1005        }
1006        demoted_boxes.push((r.l, r.t, r.r, r.b));
1007    }
1008    // The paragraphs are rebuilt from *all* of the panel's cells, so any
1009    // surviving text region inside a demoted panel (an orphan cluster or a
1010    // layout-detected fragment — pictures no longer swallow them, #165) would
1011    // say the same words twice. Consume those; wrappers and pictures stay.
1012    if !demoted_boxes.is_empty() {
1013        out.retain(|r| {
1014            r.label == "picture" || is_wrapper(r.label) || {
1015                let ra = area(r.l, r.t, r.r, r.b).max(1.0);
1016                !demoted_boxes
1017                    .iter()
1018                    .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.5)
1019            }
1020        });
1021    }
1022    // docling's "Remove regular clusters that are included in wrappers" (a
1023    // regular > 80 % inside a table is absorbed by it) already ran as
1024    // [`drop_contained_regulars`], but before this demotion created new
1025    // regulars. Apply it to them too: a panel that coincides with a table (a
1026    // dense data table detected as picture 0.80 and table 0.62 on one box;
1027    // `_handle_cross_type_overlaps` keeps both once the picture is ≥ 0.1 more
1028    // confident) rebuilds the table's words as a paragraph the grid already
1029    // renders. A panel inside another picture is left as it was.
1030    demoted_paras.retain(|p| {
1031        let pa = area(p.l, p.t, p.r, p.b).max(1.0);
1032        !out.iter()
1033            .any(|s| is_table_like(s.label) && inter(p, s.l, s.t, s.r, s.b) / pa > 0.8)
1034    });
1035    out.extend(demoted_paras);
1036    *regions = out;
1037}
1038
1039/// Drop a `picture` detection covering more than 90 % of the page — docling's
1040/// `LayoutPostprocessor._process_special_clusters` "Filter out full-page
1041/// pictures" (upstream since 2.15), applied to the thresholded detections
1042/// before overlap resolution. A box that big is the page itself, not a figure
1043/// on it: the layout model emits one for a whole-page diagram (a LaTeX figure
1044/// PDF cropped to its drawing), a plate, or a scan, and keeping it would swallow
1045/// every text cell on the page as picture children — the diagram's labels and
1046/// caption vanish behind a lone `<!-- image -->`, where docling reads them out as
1047/// text. `page_w`/`page_h` is the display-frame page box.
1048pub fn drop_full_page_pictures(regions: &mut Vec<Region>, page_w: f32, page_h: f32) {
1049    let page_area = (page_w * page_h).max(1.0);
1050    regions.retain(|r| r.label != "picture" || area(r.l, r.t, r.r, r.b) / page_area <= 0.90);
1051}
1052
1053/// Drop a `picture` detection that is a small, empty, low-confidence margin box on
1054/// a **text page** — a false positive the RT-DETR layout sometimes emits (e.g.
1055/// `right_to_left_02`'s phantom right-column picture, score 0.40); docling does not
1056/// emit it. The gate is deliberately narrow so a genuine figure is never dropped:
1057/// (1) only on pages with a digital text layer — image/scanned/figure pages have
1058/// no `cells` yet at this point (OCR runs later), so their pictures, which *are*
1059/// the content, are kept; (2) only a box covering < 25 % of the page (a margin
1060/// artifact, not a dominant figure); (3) only when it contains no text and scores
1061/// below 0.5 (real empty figures in the corpus all score ≥ 0.86).
1062pub fn drop_false_pictures(
1063    regions: &mut Vec<Region>,
1064    cells: &[TextCell],
1065    page_w: f32,
1066    page_h: f32,
1067) {
1068    if cells.iter().all(|c| c.text.trim().is_empty()) {
1069        return; // no digital text layer (image/scanned page) — keep all pictures
1070    }
1071    // A text-document page carries several text-bearing non-picture regions (so a
1072    // spurious margin picture is clearly extra). A slide / figure page has at most
1073    // one — there the picture is the content, so never drop it.
1074    let content_regions = regions
1075        .iter()
1076        .filter(|r| r.label != "picture" && !region_text(r, cells).trim().is_empty())
1077        .count();
1078    if content_regions < 2 {
1079        return;
1080    }
1081    let page_area = (page_w * page_h).max(1.0);
1082    regions.retain(|r| {
1083        if r.label != "picture" || r.score >= 0.5 {
1084            return true;
1085        }
1086        if area(r.l, r.t, r.r, r.b) / page_area >= 0.25 {
1087            return true; // a dominant figure, not a margin artifact
1088        }
1089        // Keep it if any text cell falls mostly inside (a real captioned/labelled
1090        // figure); drop only the genuinely empty low-confidence boxes.
1091        cells.iter().any(|c| {
1092            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1093            !c.text.trim().is_empty() && inter(r, c.l, c.t, c.r, c.b) / ca > 0.5
1094        })
1095    });
1096}
1097
1098/// A small digit-only region in the top/bottom margin: a page number. docling
1099/// emits `right_to_left_02`'s bottom `11` as the page's *first* text item (its
1100/// reading-order model floats the page number to the front), whereas our
1101/// position-based ordering would place a bottom region last.
1102fn is_page_number(region: &Region, cells: &[TextCell], page_h: f32) -> bool {
1103    let t = region_text(region, cells);
1104    let t = t.trim();
1105    !t.is_empty()
1106        && t.chars().all(|c| c.is_ascii_digit())
1107        && (region.b - region.t).abs() < 30.0
1108        && (region.t < page_h * 0.12 || region.b > page_h * 0.88)
1109}
1110
1111/// docling's `form` / `key_value_region` *containers* (2.123, docling#4064):
1112/// every region sitting > 0.8 inside one — text, list items, and since #4064
1113/// tables and pictures too — is that container's child. Children are
1114/// reading-ordered among themselves and emitted as one block where the
1115/// container falls in the page's top-level order (a `form_area` /
1116/// `key_value_area` group upstream), instead of interleaving with the text
1117/// around the form. A child inside several containers belongs to the smallest
1118/// (then most confident, then first); a container with children shrinks to
1119/// their union for the top-level ordering, like upstream's bbox adjustment.
1120///
1121/// The containers themselves are still not emitted (`is_skipped`), so the
1122/// Markdown is exactly upstream's — a group prints only its children.
1123///
1124/// `cids` are the items' positions in docling's assembly order
1125/// ([`cluster_cids`]) — the reading-order predictor's same-row rule (#424)
1126/// pairs consecutive ones, within the top level and within each container.
1127fn order_with_containers<T: Clone>(
1128    items: &mut Vec<T>,
1129    cids: &[usize],
1130    page_w: f32,
1131    page_h: f32,
1132    reg: impl Fn(&T) -> &Region,
1133) {
1134    let is_container = |r: &Region| matches!(r.label, "form" | "key_value_region");
1135    let containers: Vec<usize> = (0..items.len())
1136        .filter(|&i| is_container(reg(&items[i])))
1137        .collect();
1138    if containers.is_empty() {
1139        order_regions(items, cids, page_w, page_h, reg);
1140        return;
1141    }
1142    // Parent container per item (containers never nest in each other here —
1143    // upstream assigns regulars and tables/pictures only).
1144    let mut parent: Vec<Option<usize>> = vec![None; items.len()];
1145    for i in 0..items.len() {
1146        let r = reg(&items[i]);
1147        if is_container(r) {
1148            continue;
1149        }
1150        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
1151        let mut best: Option<(usize, f32, f32)> = None; // (idx, area, -score)
1152        for &c in &containers {
1153            let cr = reg(&items[c]);
1154            if inter(r, cr.l, cr.t, cr.r, cr.b) / ra > 0.8 {
1155                let key = (area(cr.l, cr.t, cr.r, cr.b), -cr.score);
1156                if best.is_none_or(|(_, a, s)| key.0 < a || (key.0 == a && key.1 < s)) {
1157                    best = Some((c, key.0, key.1));
1158                }
1159            }
1160        }
1161        parent[i] = best.map(|(c, _, _)| c);
1162    }
1163    // Top-level pass: non-children plus the containers, the latter shrunk to
1164    // their children's union.
1165    let mut top: Vec<(usize, Region)> = Vec::new();
1166    for i in 0..items.len() {
1167        if parent[i].is_some() {
1168            continue;
1169        }
1170        let mut r = reg(&items[i]).clone();
1171        if is_container(&r) {
1172            let kids: Vec<&Region> = (0..items.len())
1173                .filter(|&k| parent[k] == Some(i))
1174                .map(|k| reg(&items[k]))
1175                .collect();
1176            if !kids.is_empty() {
1177                r.l = kids.iter().map(|k| k.l).fold(f32::INFINITY, f32::min);
1178                r.t = kids.iter().map(|k| k.t).fold(f32::INFINITY, f32::min);
1179                r.r = kids.iter().map(|k| k.r).fold(f32::NEG_INFINITY, f32::max);
1180                r.b = kids.iter().map(|k| k.b).fold(f32::NEG_INFINITY, f32::max);
1181            }
1182        }
1183        top.push((i, r));
1184    }
1185    let top_cids: Vec<usize> = top.iter().map(|(i, _)| cids[*i]).collect();
1186    order_regions(&mut top, &top_cids, page_w, page_h, |it| &it.1);
1187    let mut out: Vec<T> = Vec::with_capacity(items.len());
1188    for (i, _) in top {
1189        if is_container(reg(&items[i])) {
1190            let kid_idx: Vec<usize> = (0..items.len()).filter(|&k| parent[k] == Some(i)).collect();
1191            let mut kids: Vec<T> = kid_idx.iter().map(|&k| items[k].clone()).collect();
1192            let kid_cids: Vec<usize> = kid_idx.iter().map(|&k| cids[k]).collect();
1193            order_regions(&mut kids, &kid_cids, page_w, page_h, &reg);
1194            out.push(items[i].clone());
1195            out.extend(kids);
1196        } else {
1197            out.push(items[i].clone());
1198        }
1199    }
1200    *items = out;
1201}
1202
1203/// Furniture / not-yet-emitted labels.
1204fn is_skipped(label: &str) -> bool {
1205    matches!(
1206        label,
1207        "page_header" | "page_footer" | "form" | "key_value_region"
1208    )
1209}
1210
1211/// Reading-order sort of a page's regions, via the ported rule-based
1212/// [`reading_order`](crate::reading_order) predictor (docling's
1213/// `ReadingOrderPredictor`): an up/down geometry graph with same-row links
1214/// between `cids`-consecutive elements (#424), horizontal dilation and a
1215/// depth-first traversal, with `page_header`/`page_footer` ordered as their own
1216/// groups (first/last) as docling does.
1217fn order_regions<T: Clone>(
1218    items: &mut Vec<T>,
1219    cids: &[usize],
1220    page_w: f32,
1221    page_h: f32,
1222    reg: impl Fn(&T) -> &Region,
1223) {
1224    let boxes: Vec<(f32, f32, f32, f32)> = items
1225        .iter()
1226        .map(|it| {
1227            let r = reg(it);
1228            (r.l, r.t, r.r, r.b)
1229        })
1230        .collect();
1231    let is_header: Vec<bool> = items
1232        .iter()
1233        .map(|it| reg(it).label == "page_header")
1234        .collect();
1235    let is_footer: Vec<bool> = items
1236        .iter()
1237        .map(|it| reg(it).label == "page_footer")
1238        .collect();
1239    let order =
1240        crate::reading_order::order_page(&boxes, cids, &is_header, &is_footer, page_w, page_h);
1241    *items = order.iter().map(|&i| items[i].clone()).collect();
1242}
1243
1244/// docling's assembly order of a page's clusters (`LayoutPostprocessor`'s
1245/// final `_sort_clusters(mode="id")`, #424): each region's rank when sorted by
1246/// its first source cell, then by top edge, then left edge; a region with no
1247/// cells sorts after every one that has some. docling numbers its page
1248/// elements (`cid`) in this order, and the reading-order predictor's same-row
1249/// rule pairs elements with consecutive numbers, so the ranks are what
1250/// [`order_with_containers`] hands the predictor.
1251///
1252/// A regular region's first cell is the smallest index among the cells it
1253/// claims. A table, picture or container has no cells of its own upstream
1254/// either — its cells are its *children's*: the regular clusters > 0.8 inside
1255/// it, and upstream every cell no regular cluster claimed is an orphan cluster
1256/// of its own, so a table's interior text (which no regular cluster claims)
1257/// reaches the table through those orphans. Here that is the cells > 0.8
1258/// inside the region plus the claimed cells of the regular regions > 0.8
1259/// inside it. Without the interior cells every table would sort last, and two
1260/// side-by-side tables would then be consecutive and row-linked — reading the
1261/// right table's caption ahead of the left column's headings (2206 page 8).
1262pub fn cluster_cids(regions: &[Region], cells: &[TextCell]) -> Vec<usize> {
1263    let owned = assign_cells(regions, cells);
1264    let first_cell: Vec<usize> = regions
1265        .iter()
1266        .enumerate()
1267        .map(|(i, r)| {
1268            if claims_cells(r) {
1269                return owned[i].iter().copied().min().unwrap_or(usize::MAX);
1270            }
1271            let interior = cells
1272                .iter()
1273                .enumerate()
1274                .filter(|(_, c)| {
1275                    !c.text.trim().is_empty()
1276                        && inter(r, c.l, c.t, c.r, c.b) / area(c.l, c.t, c.r, c.b).max(1.0) > 0.8
1277                })
1278                .map(|(ci, _)| ci)
1279                .min();
1280            let children = regions
1281                .iter()
1282                .enumerate()
1283                .filter(|(j, child)| {
1284                    *j != i && claims_cells(child) && {
1285                        let ca = area(child.l, child.t, child.r, child.b).max(1.0);
1286                        inter(r, child.l, child.t, child.r, child.b) / ca > 0.8
1287                    }
1288                })
1289                .filter_map(|(j, _)| owned[j].iter().copied().min())
1290                .min();
1291            interior
1292                .into_iter()
1293                .chain(children)
1294                .min()
1295                .unwrap_or(usize::MAX)
1296        })
1297        .collect();
1298    let mut by_source: Vec<usize> = (0..regions.len()).collect();
1299    // Stable, like Python's `sorted`: full ties keep the layout order.
1300    by_source.sort_by(|&a, &b| {
1301        first_cell[a]
1302            .cmp(&first_cell[b])
1303            .then(regions[a].t.total_cmp(&regions[b].t))
1304            .then(regions[a].l.total_cmp(&regions[b].l))
1305    });
1306    let mut cids = vec![0; regions.len()];
1307    for (rank, &i) in by_source.iter().enumerate() {
1308        cids[i] = rank;
1309    }
1310    cids
1311}
1312
1313/// Clean a region's assembled text: undo soft-hyphen line wraps, map curly
1314/// quotes and the ellipsis to ASCII (matching docling), and collapse runs of
1315/// whitespace. pdfium emits the line-wrap hyphen as U+0002 in this corpus
1316/// (U+00AD elsewhere), so `word\u{2} continuation` is one hyphenated word —
1317/// drop the hyphen + the joining space and merge (`com\u{2} pact` → `compact`,
1318/// `end-to\u{2} end` → `end-toend`), exactly as docling does.
1319///
1320/// Token spacing is otherwise left as the geometric join produced it. We do not
1321/// tighten punctuation spacing: docling preserves the PDF's own spaces (it keeps
1322/// `{ ahn }`, `Name 1 .`, `[ 9 ]`), and a geometric gap heuristic diverges from
1323/// it more than a plain single-space join does.
1324/// An ordered-list enumeration marker at the start of a list item: leading ASCII
1325/// digits followed by `.`, e.g. `1. Undo/Redo` → `(1, "Undo/Redo")`. Returns
1326/// `None` when the text doesn't start with `digits.`.
1327/// docling's `ListItemMarkerProcessor` bullet patterns
1328/// (`docling/models/postprocessing/list_marker_processor.py`), one glyph each.
1329const LIST_BULLET_MARKERS: &str = "\u{2022}\u{2023}\u{25E6}\u{2043}\u{204C}\u{204D}\u{2219}\u{25AA}\u{25AB}\u{25CF}\u{25CB}-*+•·‣⁃►▶▸➤➢✓✔✗✘";
1330
1331/// docling's numbered-marker patterns as byte-length scanners over the start
1332/// of the text, in its first-wins order (the compound ones first, as they are
1333/// the more specific). Each returns the marker's candidate lengths, longest
1334/// (greedy) first — the alternatives Python's regex would backtrack through
1335/// before the `\s(.+)` tail. Hand-rolled rather than `regex` because this file
1336/// is part of the `pdf-text` (wasm) build, where the `regex` crate is not.
1337/// `\d` is Unicode-aware in Python, hence `char::is_numeric`; the letters are
1338/// ASCII classes in both.
1339const LIST_NUMBERED_MARKERS: &[fn(&str) -> Vec<usize>] = &[
1340    // `\d+(?:\.\d+)+\.?` — 1.1  1.2.3  1.1.
1341    |s| {
1342        let mut i = digits(s, 0);
1343        if i == 0 {
1344            return Vec::new();
1345        }
1346        let mut groups = 0;
1347        while s[i..].starts_with('.') && digits(s, i + 1) > i + 1 {
1348            i = digits(s, i + 1);
1349            groups += 1;
1350        }
1351        if groups == 0 {
1352            return Vec::new();
1353        }
1354        if s[i..].starts_with('.') {
1355            vec![i + 1, i]
1356        } else {
1357            vec![i]
1358        }
1359    },
1360    // `\d+\.?[a-zA-Z]\.` — 9a. 3.a.
1361    |s| digits_dot_letter(s, 0, '.').into_iter().collect(),
1362    // `\d+\.?[a-zA-Z]\)` — 9a) 3.a)
1363    |s| digits_dot_letter(s, 0, ')').into_iter().collect(),
1364    // `\(\d+\.?[a-zA-Z]\)` — (9a) (3.a)
1365    |s| {
1366        if !s.starts_with('(') {
1367            return Vec::new();
1368        }
1369        digits_dot_letter(s, 1, ')').into_iter().collect()
1370    },
1371    // `\d+\.` — 1. 2. 3.
1372    |s| digits_then(s, 0, '.').into_iter().collect(),
1373    // `\d+\)` — 1) 2) 3)
1374    |s| digits_then(s, 0, ')').into_iter().collect(),
1375    // `\(\d+\)` — (1) (2) (3)
1376    |s| {
1377        if !s.starts_with('(') {
1378            return Vec::new();
1379        }
1380        digits_then(s, 1, ')').into_iter().collect()
1381    },
1382    // `\[\d+\]` — [1] [2] [3]
1383    |s| {
1384        if !s.starts_with('[') {
1385            return Vec::new();
1386        }
1387        digits_then(s, 1, ']').into_iter().collect()
1388    },
1389    // `[ivxlcdm]+\.` — i. ii. iii.
1390    |s| class_run_then(s, "ivxlcdm", '.').into_iter().collect(),
1391    // `[IVXLCDM]+\.` — I. II. III.
1392    |s| class_run_then(s, "IVXLCDM", '.').into_iter().collect(),
1393    // `[a-z]\.` / `[A-Z]\.` / `[a-z]\)` / `[A-Z]\)`
1394    |s| {
1395        letter_then(s, char::is_ascii_lowercase, '.')
1396            .into_iter()
1397            .collect()
1398    },
1399    |s| {
1400        letter_then(s, char::is_ascii_uppercase, '.')
1401            .into_iter()
1402            .collect()
1403    },
1404    |s| {
1405        letter_then(s, char::is_ascii_lowercase, ')')
1406            .into_iter()
1407            .collect()
1408    },
1409    |s| {
1410        letter_then(s, char::is_ascii_uppercase, ')')
1411            .into_iter()
1412            .collect()
1413    },
1414];
1415
1416/// Byte offset just past the run of `\d` characters starting at `from`
1417/// (`from` itself when there is none).
1418fn digits(s: &str, from: usize) -> usize {
1419    s[from..]
1420        .char_indices()
1421        .find(|(_, c)| !c.is_numeric())
1422        .map_or(s.len(), |(i, _)| from + i)
1423}
1424
1425/// `\d+<close>` from `from`: the length through `close`, if it matches.
1426fn digits_then(s: &str, from: usize, close: char) -> Option<usize> {
1427    let end = digits(s, from);
1428    (end > from && s[end..].starts_with(close)).then(|| end + close.len_utf8())
1429}
1430
1431/// `\d+\.?[a-zA-Z]<close>` from `from`.
1432fn digits_dot_letter(s: &str, from: usize, close: char) -> Option<usize> {
1433    let mut i = digits(s, from);
1434    if i == from {
1435        return None;
1436    }
1437    if s[i..].starts_with('.') {
1438        i += 1;
1439    }
1440    let letter = s[i..].chars().next().filter(char::is_ascii_alphabetic)?;
1441    i += letter.len_utf8();
1442    s[i..].starts_with(close).then(|| i + close.len_utf8())
1443}
1444
1445/// `[<class>]+<close>` at the start.
1446fn class_run_then(s: &str, class: &str, close: char) -> Option<usize> {
1447    let end = s
1448        .char_indices()
1449        .find(|(_, c)| !class.contains(*c))
1450        .map_or(s.len(), |(i, _)| i);
1451    (end > 0 && s[end..].starts_with(close)).then(|| end + close.len_utf8())
1452}
1453
1454/// `[<letter class>]<close>` at the start.
1455fn letter_then(s: &str, class: fn(&char) -> bool, close: char) -> Option<usize> {
1456    let letter = s.chars().next().filter(class)?;
1457    let i = letter.len_utf8();
1458    s[i..].starts_with(close).then(|| i + close.len_utf8())
1459}
1460
1461/// docling's `ListItemMarkerProcessor.process_list_item`: the item's original
1462/// text is matched against `^(marker)\s(.+)` (DOTALL) for the bullet patterns,
1463/// then the numbered ones in order; a hit splits it into
1464/// `(marker, text, enumerated)`. `\s` is one Unicode whitespace character and
1465/// `.+` everything after it, which must be non-empty.
1466fn split_list_marker(text: &str) -> Option<(&str, &str, bool)> {
1467    let tail_after = |len: usize| -> Option<&str> {
1468        let ws = text[len..].chars().next()?;
1469        if !ws.is_whitespace() {
1470            return None;
1471        }
1472        let rest = &text[len + ws.len_utf8()..];
1473        (!rest.is_empty()).then_some(rest)
1474    };
1475    let first = text.chars().next()?;
1476    if LIST_BULLET_MARKERS.contains(first) {
1477        if let Some(rest) = tail_after(first.len_utf8()) {
1478            return Some((&text[..first.len_utf8()], rest, false));
1479        }
1480    }
1481    for matcher in LIST_NUMBERED_MARKERS {
1482        for len in matcher(text) {
1483            if let Some(rest) = tail_after(len) {
1484                return Some((&text[..len], rest, true));
1485            }
1486        }
1487    }
1488    None
1489}
1490
1491/// A PDF `list_item` region as docling emits it: `ListItemMarkerProcessor`
1492/// splits the marker off (see [`split_list_marker`]), and docling-core's
1493/// Markdown list serializer (default `orig_list_item_marker_mode = AUTO`,
1494/// `ensure_valid_list_item_marker`) prints it as `N. text` for an `N.` marker
1495/// (`case_already_valid`: the marker verbatim); as `- text` for a bullet glyph
1496/// (no letter or digit in the marker: only the `-` the serializer adds); and
1497/// as `- a) text` / `- 1.2 text` / `- [3] text` for any other marker
1498/// (`case_auto`: the serializer's `-`, then the original marker) — spelled
1499/// here as a bullet item whose text carries the marker, the way the DOCX and
1500/// DOC backends already spell theirs. An item without a recognizable marker is
1501/// a plain bullet. The symbol-font bullets docling-parse filters out of its
1502/// cells (`•◦▪·*` glued to the text) are stripped before the match, as before.
1503fn list_item_node(text: &str, loc: [u16; 4], first_in_list: bool) -> Node {
1504    // docling's match runs on the text as docling-parse hands it over; the
1505    // glued symbol-font bullets it never sees are stripped only when the raw
1506    // text matches no marker (`•Text` → `Text`, but `• Text` → marker `•`).
1507    let stripped = text
1508        .trim_start_matches(['•', '◦', '▪', '·', '*'])
1509        .trim_start();
1510    let split = split_list_marker(text).or_else(|| split_list_marker(stripped));
1511    let bullet = |text: String, marker: &str| Node::ListItem {
1512        ordered: false,
1513        number: 0,
1514        first_in_list,
1515        text: md_escape(&text),
1516        level: 0,
1517        // docling keeps the marker as the DocLang list marker
1518        // (`<ldiv><marker>·</marker></ldiv>`); Markdown prints its own `-`.
1519        marker: Some(marker.to_string()),
1520        location: Some(loc),
1521        dclx: None,
1522        href: None,
1523        layer: None,
1524    };
1525    // docling-core's `case_already_valid` is a *full* `\d+\.` match: `3.a.`
1526    // and `1.2` are `case_auto` markers (`- 3.a. text`), not item numbers.
1527    let is_number_dot = |m: &str| {
1528        m.strip_suffix('.')
1529            .is_some_and(|d| !d.is_empty() && d.chars().all(char::is_numeric))
1530    };
1531    match split {
1532        Some((marker, body, true)) if is_number_dot(marker) => {
1533            let number = parse_ordered_marker(marker).map_or(0, |(n, _)| n);
1534            Node::ListItem {
1535                ordered: true,
1536                number,
1537                first_in_list,
1538                text: md_escape(body),
1539                level: 0,
1540                marker: Some(marker.to_string()),
1541                location: Some(loc),
1542                dclx: None,
1543                href: None,
1544                layer: None,
1545            }
1546        }
1547        // `case_auto`: a marker holding a letter or digit rides in the text.
1548        Some((marker, body, true)) => bullet(format!("{marker} {body}"), marker),
1549        Some((marker, body, false)) => bullet(body.to_string(), marker),
1550        None => bullet(stripped.to_string(), "·"),
1551    }
1552}
1553
1554fn parse_ordered_marker(s: &str) -> Option<(u64, String)> {
1555    let digits: String = s.chars().take_while(|c| c.is_ascii_digit()).collect();
1556    if digits.is_empty() {
1557        return None;
1558    }
1559    let rest = s[digits.len()..].strip_prefix('.')?;
1560    let number = digits.parse().ok()?;
1561    Some((number, rest.trim_start().to_string()))
1562}
1563
1564/// Escape markdown special characters the way docling-core's markdown serializer
1565/// does (`markdown.py` post_process): `_` → `\_`, then HTML-escape `&`, `<`, `>`
1566/// (quote=False, so quotes are left). Applied to prose (headings, list items,
1567/// paragraphs); code blocks, the formula placeholder, and table cells are left raw.
1568fn md_escape(text: &str) -> String {
1569    text.replace('_', "\\_")
1570        .replace('&', "&amp;")
1571        .replace('<', "&lt;")
1572        .replace('>', "&gt;")
1573}
1574
1575fn clean_text(text: &str) -> String {
1576    // Typographic-quote normalization follows docling-parse's sanitizer table
1577    // (`pdf_sanitators/constants.h`): every curly quote — single *and double* —
1578    // becomes the ASCII apostrophe `'`, and `‚` a comma. A `"` in docling's
1579    // output only ever comes from a literal `quotedbl` glyph, never from `“ ”`
1580    // (2206's `'text in the wild"` pairs a curly open with a literal-quote
1581    // close). This replaces an earlier Hangul-only special case that patched
1582    // one symptom of mapping `“ ”` to `"`.
1583    let replaced = text
1584        .replace("\u{2} ", "")
1585        .replace("\u{ad} ", "")
1586        .replace(['\u{2}', '\u{ad}'], "") // any stray wrap hyphens not at a join
1587        .replace(
1588            [
1589                '\u{2018}', '\u{2019}', '\u{201b}', '\u{201c}', '\u{201d}', '\u{201e}', '\u{201f}',
1590            ],
1591            "'",
1592        ) // ‘ ’ ‛ “ ” „ ‟ → '
1593        .replace('\u{201a}', ",") // ‚ → ,
1594        .replace(
1595            [
1596                '\u{2010}', '\u{2011}', '\u{2012}', '\u{2013}', '\u{2014}', '\u{2015}', '\u{2212}',
1597            ],
1598            "-",
1599        ) // hyphen/dash family → -
1600        .replace('\u{2044}', "/") // ⁄ fraction slash → /
1601        .replace('\u{2022}', "\u{b7}") // • → · (docling never emits •; inline CCS-concept separators)
1602        .replace('\u{2026}', "..."); // … → ...
1603                                     // The docling-parse sanitizer already placed the correct spacing (e.g.
1604                                     // justified double spaces); preserve internal runs of spaces, only
1605                                     // normalizing line breaks/tabs and trimming the ends.
1606    let out = replaced.replace(['\n', '\r', '\t'], " ").trim().to_string();
1607    fix_arabic_lam_alef(&out)
1608}
1609
1610/// pdfium decomposes the Arabic lam-alef ligature (لا / لإ / لأ / لآ) into its
1611/// glyph constituents in *visual* order — `alef-variant, lam` — but docling keeps
1612/// logical order, `lam, alef-variant`. Swap a mid-word `alef-variant + lam` back
1613/// to `lam + alef-variant`. "Mid-word" (the previous char is an Arabic letter)
1614/// distinguishes the ligature from the definite article `ال` (word-initial
1615/// `alef + lam`), which must stay. No-op for non-Arabic text.
1616fn fix_arabic_lam_alef(s: &str) -> String {
1617    let is_arabic_letter = |c: char| ('\u{0620}'..='\u{064A}').contains(&c);
1618    let chars: Vec<char> = s.chars().collect();
1619    if !chars.iter().any(|&c| is_arabic_letter(c)) {
1620        return s.to_string(); // no-op for non-Arabic text
1621    }
1622    // Pass 1: swap mid-word `alef-variant + lam` → `lam + alef-variant`. Only the
1623    // hamza/madda alef variants (إ أ آ) are safe: the definite article is always
1624    // plain `ا + ل`, so plain `alef + lam` is ambiguous (a legitimate `فعالة` vs a
1625    // reversed `لا` ligature look identical) — leaving plain alef alone avoids
1626    // corrupting legitimate words.
1627    let mut a: Vec<char> = Vec::with_capacity(chars.len());
1628    let mut i = 0;
1629    while i < chars.len() {
1630        let c = chars[i];
1631        if matches!(c, '\u{0622}' | '\u{0623}' | '\u{0625}')
1632            && chars.get(i + 1) == Some(&'\u{0644}')
1633            && i > 0
1634            && is_arabic_letter(chars[i - 1])
1635            // A preceding lam means this alef-variant is *already* the logical
1636            // `lam + alef` ligature; the following lam is the next syllable's
1637            // letter, not a reversed ligature — swapping it corrupts `لآل` → `للآ`
1638            // (e.g. التعلم الآلي → الآلي, not اللآي).
1639            && chars[i - 1] != '\u{0644}'
1640        {
1641            a.push('\u{0644}');
1642            a.push(c);
1643            i += 2;
1644            continue;
1645        }
1646        a.push(c);
1647        i += 1;
1648    }
1649    // Pass 2: insert a space at Arabic↔Latin boundaries (bidi script switch) that
1650    // pdfium runs together — docling separates the embedded Latin run (`وPython`
1651    // → `و Python`).
1652    let mut out: Vec<char> = Vec::with_capacity(a.len());
1653    for (j, &c) in a.iter().enumerate() {
1654        if j > 0 {
1655            let p = a[j - 1];
1656            if (is_arabic_letter(p) && c.is_ascii_alphabetic())
1657                || (p.is_ascii_alphabetic() && is_arabic_letter(c))
1658            {
1659                out.push(' ');
1660            }
1661        }
1662        out.push(c);
1663    }
1664    out.into_iter().collect()
1665}
1666
1667/// docling's `PageAssembleModel._match_hyperlink`: the URI whose link
1668/// annotations cover at least half of the region's box, or `None`. Coverage is
1669/// intersection-over-region-area, **accumulated per URI** — a URL that wraps
1670/// across lines carries several annotation rects that sum toward the same
1671/// target. Ties resolve to the first-seen URI (Python's `max` over dict
1672/// insertion order); the winner still needs `>= 0.5`
1673/// (`_HYPERLINK_COVERAGE_THRESHOLD`).
1674pub(crate) fn region_hyperlink(
1675    region: &Region,
1676    links: &[crate::pdfium_backend::LinkAnnot],
1677) -> Option<String> {
1678    if links.is_empty() {
1679        return None;
1680    }
1681    let area = (region.r - region.l).max(0.0) * (region.b - region.t).max(0.0);
1682    if area <= 0.0 {
1683        return None;
1684    }
1685    let mut coverage: Vec<(&str, f32)> = Vec::new();
1686    for link in links {
1687        let ix = (region.r.min(link.r) - region.l.max(link.l)).max(0.0);
1688        let iy = (region.b.min(link.b) - region.t.max(link.t)).max(0.0);
1689        let c = ix * iy / area;
1690        match coverage.iter_mut().find(|(uri, _)| *uri == link.uri) {
1691            Some((_, acc)) => *acc += c,
1692            None => coverage.push((&link.uri, c)),
1693        }
1694    }
1695    let mut best: Option<(&str, f32)> = None;
1696    for (uri, c) in coverage {
1697        // Strictly greater keeps the first-seen URI on ties, like Python's max.
1698        if best.is_none_or(|(_, bc)| c > bc) {
1699            best = Some((uri, c));
1700        }
1701    }
1702    let (uri, c) = best?;
1703    (c >= 0.5).then(|| normalize_uri(uri))
1704}
1705
1706/// The pydantic-`AnyUrl` normalization docling's hyperlink value passes
1707/// through on its way to the serializer: a URL with an authority but no path
1708/// gains a trailing `/` (`https://arxiv.org` → `https://arxiv.org/`). Other
1709/// AnyUrl canonicalizations (scheme/host lowercasing, percent-encoding) don't
1710/// occur in PDF link annotations in practice, so they are not reproduced.
1711fn normalize_uri(uri: &str) -> String {
1712    if let Some((_, rest)) = uri.split_once("://") {
1713        if !rest.is_empty() && !rest.contains(['/', '?', '#']) {
1714            return format!("{uri}/");
1715        }
1716    }
1717    uri.to_string()
1718}
1719
1720/// Resolve each page hyperlink to the visible text it covers, as `(anchor, uri)`
1721/// in reading order. The anchor is the cells whose centre falls in the link rect,
1722/// joined left-to-right and cleaned the same way prose is (so it matches the
1723/// serialized text), deduped against the immediately-preceding link so pdfium's
1724/// occasional duplicate annotation doesn't double-list. Empty anchors are dropped.
1725pub(crate) fn resolve_link_anchors(page: &PdfPage) -> Vec<(String, String)> {
1726    let mut out: Vec<(String, String)> = Vec::new();
1727    // Use per-word cells, not the line-merged `cells`: a link rect covers a few
1728    // words on a line, and a whole merged line cell would over-capture (its centre
1729    // lands in one link's rect, grabbing the entire line as that link's anchor).
1730    let words = if page.word_cells.is_empty() {
1731        &page.cells
1732    } else {
1733        &page.word_cells
1734    };
1735    for link in &page.links {
1736        // A cell participates when its centre row is inside the rect and it
1737        // overlaps the rect horizontally. A cell can be *wider* than the rect:
1738        // PDFs often draw a whole header line as one text run ("LinkedIn |
1739        // GitHub | Credly"), which docling-parse's word grouping keeps as one
1740        // cell even though each label carries its own link annotation —
1741        // centre-in-rect alone would hand the entire line to every link.
1742        // [`cell_text_in_rect`] clips such a cell to the tokens under the rect.
1743        let mut inside: Vec<(&TextCell, String)> = words
1744            .iter()
1745            .filter(|c| {
1746                let cy = (c.t + c.b) / 2.0;
1747                cy >= link.t && cy <= link.b && c.r.min(link.r) > c.l.max(link.l)
1748            })
1749            .filter_map(|c| {
1750                let text = cell_text_in_rect(c, link.l, link.r);
1751                (!text.is_empty()).then_some((c, text))
1752            })
1753            .collect();
1754        // Reading order: top band then left-to-right (link anchors are LTR).
1755        let band = inside
1756            .iter()
1757            .map(|(c, _)| (c.b - c.t).abs())
1758            .fold(0.0f32, f32::max)
1759            .max(1.0);
1760        inside.sort_by_key(|(c, _)| ((c.t / band).round() as i64, (c.l * 10.0) as i64));
1761        let anchor = clean_text(
1762            &inside
1763                .iter()
1764                .map(|(_, t)| t.trim())
1765                .filter(|t| !t.is_empty())
1766                .collect::<Vec<_>>()
1767                .join(" "),
1768        );
1769        if anchor.is_empty() {
1770            continue;
1771        }
1772        if out
1773            .last()
1774            .is_some_and(|(a, u)| a == &anchor && u == &link.uri)
1775        {
1776            continue;
1777        }
1778        out.push((anchor, link.uri.clone()));
1779    }
1780    out
1781}
1782
1783/// The part of a cell's text that lies under a link rect's x-range. A cell
1784/// fully inside the rect (by centre) returns its whole text. A wider cell is
1785/// split into whitespace tokens whose x-spans are estimated proportionally to
1786/// their character positions (kerning makes this approximate, so selection
1787/// snaps to whole tokens, never characters); tokens whose estimated centre
1788/// falls inside the rect are kept. Returns "" when nothing falls inside.
1789fn cell_text_in_rect(c: &TextCell, l: f32, r: f32) -> String {
1790    let cx = (c.l + c.r) / 2.0;
1791    if cx >= l && cx <= r && c.l >= l - (c.r - c.l) * 0.25 && c.r <= r + (c.r - c.l) * 0.25 {
1792        return c.text.trim().to_string();
1793    }
1794    let chars: Vec<char> = c.text.chars().collect();
1795    let n = chars.len();
1796    if n == 0 || c.r <= c.l {
1797        return String::new();
1798    }
1799    let per = (c.r - c.l) / n as f32;
1800    let mut out: Vec<String> = Vec::new();
1801    let mut token = String::new();
1802    let mut start = 0usize;
1803    // A trailing sentinel space flushes the last token.
1804    for (i, &ch) in chars.iter().enumerate().chain(std::iter::once((n, &' '))) {
1805        if ch.is_whitespace() {
1806            if !token.is_empty() {
1807                let mid = c.l + (start as f32 + (i - start) as f32 / 2.0) * per;
1808                if mid >= l && mid <= r {
1809                    out.push(std::mem::take(&mut token));
1810                } else {
1811                    token.clear();
1812                }
1813            }
1814        } else {
1815            if token.is_empty() {
1816                start = i;
1817            }
1818            token.push(ch);
1819        }
1820    }
1821    out.join(" ")
1822}
1823
1824/// Cells assigned to a region (best container), in reading order, joined.
1825fn region_text(region: &Region, cells: &[TextCell]) -> String {
1826    let inside: Vec<&TextCell> = cells
1827        .iter()
1828        .filter(|c| {
1829            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1830            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1831        })
1832        .collect();
1833    cells_text(inside)
1834}
1835
1836/// docling's exclusive cell assignment (`_assign_cells_to_clusters`): every
1837/// non-empty cell goes to the single best-overlapping *regular* region at
1838/// intersection-over-self > 0.2, and each region serializes exactly its
1839/// assigned cells. A cell under two overlapping boxes is emitted once (by the
1840/// better-covering one), and a cell only partially under its region — e.g.
1841/// normal_4pages' big section numeral, ~30 % inside the heading box — still
1842/// joins it (`## 들어가며 1`) instead of leaking as an orphan. Pictures and
1843/// wrappers never claim (docling walks regular clusters only); ties go to the
1844/// first region, like docling's strict `>` best-overlap scan.
1845pub fn region_texts_exclusive(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
1846    let owned = assign_cells(regions, cells);
1847    // Non-claimers (tables/wrappers/pictures) keep the inclusive > 0.5 text:
1848    // docling fills a special cluster's cells from its contained children, and
1849    // downstream table assembly gates on that text being non-empty.
1850    regions
1851        .iter()
1852        .zip(owned)
1853        .map(|(r, cs)| {
1854            if claims_cells(r) {
1855                cells_text(cs.iter().map(|&i| &cells[i]).collect())
1856            } else {
1857                region_text(r, cells)
1858            }
1859        })
1860        .collect()
1861}
1862
1863/// A *regular* region in docling's sense — one that claims cells. Pictures and
1864/// the wrappers (`table`, `document_index`, `form`, `key_value_region`) fill
1865/// their cells from contained children instead.
1866fn claims_cells(r: &Region) -> bool {
1867    r.label != "picture" && !is_wrapper(r.label)
1868}
1869
1870/// docling's `_assign_cells_to_clusters`: each non-empty cell's index goes to
1871/// the single best-overlapping regular region at intersection-over-self > 0.2
1872/// (ties to the first region, like docling's strict `>` scan). One entry per
1873/// region, in region order.
1874fn assign_cells(regions: &[Region], cells: &[TextCell]) -> Vec<Vec<usize>> {
1875    let mut owned: Vec<Vec<usize>> = vec![Vec::new(); regions.len()];
1876    for (ci, c) in cells.iter().enumerate() {
1877        if c.text.trim().is_empty() {
1878            continue;
1879        }
1880        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1881        let mut best: Option<(usize, f32)> = None;
1882        for (i, r) in regions.iter().enumerate() {
1883            if !claims_cells(r) {
1884                continue;
1885            }
1886            let ov = inter(r, c.l, c.t, c.r, c.b) / ca;
1887            if ov > 0.2 && best.is_none_or(|(_, b)| ov > b) {
1888                best = Some((i, ov));
1889            }
1890        }
1891        if let Some((i, _)) = best {
1892            owned[i].push(ci);
1893        }
1894    }
1895    owned
1896}
1897
1898/// docling's regular-cluster refinement after cell assignment
1899/// (`LayoutPostprocessor._process_regular_clusters`, #419), run once the page's
1900/// cells are final and before reading order:
1901///
1902/// 1. every regular region's box becomes the union of the cells it claimed
1903///    (`_adjust_cluster_bboxes` — a regular cluster's bbox *is* its cells'
1904///    bbox; a table's is the union with the model box, and pictures keep
1905///    theirs, so neither is touched here);
1906/// 2. a regular region that claimed no cell is dropped (`keep_empty_clusters`
1907///    is off; a `formula` is kept, as upstream keeps it);
1908/// 3. an orphan text region (`score == 0.0`, from [`add_orphan_regions`]) that
1909///    now sits > 0.8 inside another regular region's fitted box is folded into
1910///    it (`_remove_overlapping_clusters` at containment 0.8, the larger box
1911///    winning the group) — up to three rounds, like upstream's loop.
1912///
1913/// Why it matters: the layout model's box can end partway through a line. That
1914/// line fails the 0.2 claim and becomes an orphan — recoverable — but the
1915/// *model* box still overlaps the orphan's line by a few points, so the
1916/// reading-order graph, which links only strictly-above pairs, gets no edge
1917/// between them and may emit the next paragraph first, stranding the line
1918/// after the paragraph it belongs in (1540 of 6050 text blocks on the #419
1919/// book began mid-sentence). Fitted to its cells, the box ends on a line
1920/// boundary and the orphan slots in between; an orphan the fitted box
1921/// swallows joins the paragraph outright. Cell assignment is untouched: a
1922/// region's fitted box contains every cell it claimed, so
1923/// [`region_texts_exclusive`] hands it the same cells afterwards.
1924///
1925/// A page with no cells yet (a scan before OCR) is left alone: dropping every
1926/// text region for want of cells would be wrong, and the OCR paths call this
1927/// again once the cells exist.
1928pub fn fit_regions_to_cells(regions: &mut Vec<Region>, cells: &[TextCell]) {
1929    if !cells.iter().any(|c| !c.text.trim().is_empty()) {
1930        return;
1931    }
1932    for _ in 0..3 {
1933        let owned = assign_cells(regions, cells);
1934        let mut fitted: Vec<Region> = Vec::with_capacity(regions.len());
1935        for (r, own) in regions.iter().zip(&owned) {
1936            if !claims_cells(r) {
1937                fitted.push(r.clone());
1938                continue;
1939            }
1940            if own.is_empty() {
1941                if r.label == "formula" {
1942                    fitted.push(r.clone());
1943                }
1944                continue;
1945            }
1946            let mut f = r.clone();
1947            f.l = own
1948                .iter()
1949                .map(|&i| cells[i].l)
1950                .fold(f32::INFINITY, f32::min);
1951            f.t = own
1952                .iter()
1953                .map(|&i| cells[i].t)
1954                .fold(f32::INFINITY, f32::min);
1955            f.r = own
1956                .iter()
1957                .map(|&i| cells[i].r)
1958                .fold(f32::NEG_INFINITY, f32::max);
1959            f.b = own
1960                .iter()
1961                .map(|&i| cells[i].b)
1962                .fold(f32::NEG_INFINITY, f32::max);
1963            fitted.push(f);
1964        }
1965        let mut changed = fitted.len() != regions.len();
1966        // Fold orphans into the regular region whose fitted box holds them.
1967        let mut drop = vec![false; fitted.len()];
1968        for i in 0..fitted.len() {
1969            let o = &fitted[i];
1970            if !(o.score == 0.0 && o.label == "text") {
1971                continue;
1972            }
1973            let oa = area(o.l, o.t, o.r, o.b).max(1.0);
1974            let mut best: Option<(usize, f32)> = None;
1975            for (j, r) in fitted.iter().enumerate() {
1976                if j == i || drop[j] || r.score == 0.0 || !claims_cells(r) {
1977                    continue;
1978                }
1979                let ov = inter(r, o.l, o.t, o.r, o.b) / oa;
1980                if ov > 0.8 && best.is_none_or(|(_, b)| ov > b) {
1981                    best = Some((j, ov));
1982                }
1983            }
1984            if let Some((j, _)) = best {
1985                let (l, t, r, b) = (o.l, o.t, o.r, o.b);
1986                let host = &mut fitted[j];
1987                host.l = host.l.min(l);
1988                host.t = host.t.min(t);
1989                host.r = host.r.max(r);
1990                host.b = host.b.max(b);
1991                drop[i] = true;
1992                changed = true;
1993            }
1994        }
1995        let mut drop = drop.into_iter();
1996        fitted.retain(|_| !drop.next().expect("aligned"));
1997        *regions = fitted;
1998        if !changed {
1999            break;
2000        }
2001    }
2002}
2003
2004/// Join a prefiltered cell list into the region's text (docling's
2005/// `sanitize_text` over the sanitizer's cell order).
2006fn cells_text(inside: Vec<&TextCell>) -> String {
2007    // docling orders a cluster's cells by their docling-parse cell index
2008    // alone (`LayoutPostprocessor._sort_cells`: `sorted(cells, key=c.index)`)
2009    // — the sanitizer's output order, which our `cells` slice already is.
2010    // No geometric re-sort: normal_4pages' big section numerals paint
2011    // *after* their heading text, and docling's `## 들어가며 1` (numeral
2012    // last) only falls out of pure index order — a band sort dragged the
2013    // numeral to the front. The overlap-grouped line restore this replaced
2014    // measured strictly worse on the corpus (it fixed nothing the index
2015    // order broke, and broke the numerals).
2016    let joined = {
2017        // docling's `PageAssembleModel.sanitize_text`, ported verbatim over the
2018        // parse-index-ordered lines: append a separating space to a line —
2019        // unless it ends with `-`. A dash-ending line whose last word and the
2020        // next line's first word are both alphanumeric is a wrapped word: the
2021        // dash is dropped and the lines fuse (`platforms-` + `reflects` →
2022        // `platformsreflects`, `pp. 545-` + `561` → `545561`). Any other
2023        // dash-ending line — e.g. the *bare* `-` cell a superscript ORCID or an
2024        // inline `–` bullet splits off (its word list is empty, so the fuse
2025        // test fails) — keeps its dash and still takes no trailing space:
2026        // `[0000` `-` `0002` joins as docling's `[0000 -0002`, and the OTSL
2027        // list's `-` + `"C" cell -` + `a new table cell` collapses to
2028        // `-"C" cell a new table cell`. Our cells still carry the raw dash
2029        // family (docling-parse normalizes to `-` before this; clean_text does
2030        // it after), so the endswith test matches them all.
2031        let texts: Vec<&str> = inside
2032            .iter()
2033            .map(|c| c.text.trim())
2034            // Skip whitespace-only cells (a justified line's trailing space
2035            // glyph): an empty line would double the separator.
2036            .filter(|t| !t.is_empty())
2037            .collect();
2038        let last_word_alnum = |s: &str| {
2039            s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
2040                .rfind(|w| !w.is_empty())
2041                .is_some_and(|w| w.chars().all(char::is_alphanumeric))
2042        };
2043        let first_word_alnum = |s: &str| {
2044            s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
2045                .find(|w| !w.is_empty())
2046                .is_some_and(|w| w.chars().all(char::is_alphanumeric))
2047        };
2048        let mut out = String::new();
2049        for (i, t) in texts.iter().enumerate() {
2050            if i > 0 {
2051                let prev = texts[i - 1];
2052                let dashish = matches!(
2053                    prev.chars().last(),
2054                    Some(
2055                        '-' | '\u{2010}'
2056                            | '\u{2011}'
2057                            | '\u{2012}'
2058                            | '\u{2013}'
2059                            | '\u{2014}'
2060                            | '\u{2015}'
2061                            | '\u{2212}'
2062                    )
2063                );
2064                // docling#4052 (2.122): a dash only splits a word when it is
2065                // *attached* to one — the character before it is alphanumeric.
2066                // A dash that follows whitespace (a separator dash, a bullet
2067                // marker, a wrapped `-prefixed` token, the bare `-` cell an
2068                // ORCID splits off) is a literal character: it is kept and the
2069                // lines join with the ordinary space.
2070                let attached = prev.chars().rev().nth(1).is_some_and(char::is_alphanumeric);
2071                if dashish && attached {
2072                    if last_word_alnum(prev) && first_word_alnum(t) {
2073                        out.pop(); // wrapped word: fuse without the dash
2074                    }
2075                    // an attached dash never takes a separating space
2076                } else {
2077                    out.push(' ');
2078                }
2079            }
2080            out.push_str(t);
2081        }
2082        out
2083    };
2084    clean_text(&joined)
2085}
2086
2087/// Tighten the spaces pdfium leaves around tight punctuation in a code line
2088/// (`console .log` → `console.log`, `add (3 , 5)` → `add(3, 5)`), matching
2089/// docling-parse's source spacing.
2090fn tighten_code_punct(s: &str) -> String {
2091    s.replace(" .", ".")
2092        .replace(" ,", ",")
2093        .replace(" ;", ";")
2094        .replace(" )", ")")
2095        .replace(" (", "(")
2096}
2097
2098/// Assemble a **code** region's text with its line structure preserved.
2099///
2100/// Unlike [`region_text`] — which joins every cell with a single space, the right
2101/// thing for prose reflow — a code block's line breaks and indentation are
2102/// significant. The `code_cells` are already one physical source line each
2103/// (grouped space-glyph-only, so monospace runs keep their spacing), so this:
2104///
2105/// 1. groups the cells into vertical line bands and orders them top→bottom,
2106///    left→right;
2107/// 2. joins the lines with `\n` (rather than spaces), keeping the carriage
2108///    returns; and
2109/// 3. reconstructs each line's leading indentation from its left offset, in units
2110///    of the block's estimated monospace character width, so nesting survives.
2111///
2112/// Typography is normalized per line via [`clean_text`] (smart quotes, dashes,
2113/// ellipsis), which never merges lines. Returns an empty string if the region has
2114/// no code cells (the caller falls back to the prose text).
2115fn code_region_text(region: &Region, cells: &[TextCell]) -> String {
2116    let mut inside: Vec<&TextCell> = cells
2117        .iter()
2118        .filter(|c| {
2119            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2120            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
2121        })
2122        .filter(|c| !c.text.trim().is_empty())
2123        .collect();
2124    if inside.is_empty() {
2125        return String::new();
2126    }
2127
2128    // Quantize the top edge into ~line bands (like `region_text`), then order the
2129    // cells by band (top→bottom) and, within a band, by left edge.
2130    let band = inside
2131        .iter()
2132        .map(|c| (c.b - c.t).abs())
2133        .fold(0.0f32, f32::max)
2134        .max(1.0);
2135    let line_of = |c: &TextCell| (c.t / band).round() as i64;
2136    inside.sort_by_key(|c| (line_of(c), (c.l * 10.0) as i64));
2137
2138    // Estimate one monospace character's width (total ink width / total glyphs) to
2139    // convert a line's left offset into a count of leading spaces. Measured over
2140    // all lines so a single short line can't skew it.
2141    let (mut total_w, mut total_chars) = (0.0f32, 0usize);
2142    for c in &inside {
2143        let n = c.text.trim().chars().count();
2144        if n > 0 {
2145            total_w += (c.r - c.l).max(0.0);
2146            total_chars += n;
2147        }
2148    }
2149    let char_w = if total_chars > 0 {
2150        (total_w / total_chars as f32).max(1.0)
2151    } else {
2152        1.0
2153    };
2154    // The block's own left margin is the zero-indent baseline.
2155    let base_l = inside.iter().map(|c| c.l).fold(f32::INFINITY, f32::min);
2156
2157    let mut lines: Vec<String> = Vec::new();
2158    let mut cur: Option<i64> = None;
2159    for c in &inside {
2160        // Tighten pdfium's spaced punctuation per line (on the trimmed content, so
2161        // the reconstructed leading indentation is never nibbled).
2162        let text = tighten_code_punct(&clean_text(c.text.trim()));
2163        if Some(line_of(c)) == cur {
2164            // A second cell sharing this band (rare — e.g. split columns): keep it
2165            // on the same source line, separated by a space.
2166            if let Some(last) = lines.last_mut() {
2167                last.push(' ');
2168                last.push_str(&text);
2169            }
2170            continue;
2171        }
2172        let indent = ((c.l - base_l) / char_w).round().max(0.0) as usize;
2173        lines.push(format!("{}{}", " ".repeat(indent), text));
2174        cur = Some(line_of(c));
2175    }
2176    lines.join("\n")
2177}
2178
2179/// Reconstruct a table's grid geometrically from the text cells inside its
2180/// region: cluster cells into rows (by vertical centre) and columns (by clustered
2181/// left edges), then place each cell. A model-free stand-in for TableFormer that
2182/// recovers grid-aligned tables from the precise PDF text layer (it does not
2183/// resolve row/column spans).
2184pub fn reconstruct_table(region: &Region, cells: &[TextCell]) -> Vec<Vec<String>> {
2185    let mut inside: Vec<&TextCell> = cells
2186        .iter()
2187        .filter(|c| {
2188            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2189            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
2190        })
2191        .collect();
2192    if inside.is_empty() {
2193        return Vec::new();
2194    }
2195    inside.sort_by(|a, b| a.t.total_cmp(&b.t));
2196
2197    // Rows: consecutive cells whose vertical centre is within ~0.7 line height.
2198    let mut rows: Vec<(f32, Vec<&TextCell>)> = Vec::new();
2199    for c in &inside {
2200        let cyc = (c.t + c.b) / 2.0;
2201        let lh = (c.b - c.t).abs().max(1.0);
2202        if let Some((ryc, row)) = rows.last_mut() {
2203            if (cyc - *ryc).abs() < lh * 0.7 {
2204                row.push(c);
2205                continue;
2206            }
2207        }
2208        rows.push((cyc, vec![c]));
2209    }
2210
2211    // Columns: cluster left edges (merge those within a tolerance).
2212    let tol = {
2213        let mut hs: Vec<f32> = inside.iter().map(|c| (c.b - c.t).abs()).collect();
2214        hs.sort_by(f32::total_cmp);
2215        hs[hs.len() / 2].max(4.0) * 1.5
2216    };
2217    let mut lefts: Vec<f32> = inside.iter().map(|c| c.l).collect();
2218    lefts.sort_by(f32::total_cmp);
2219    let mut col_starts: Vec<f32> = Vec::new();
2220    for l in lefts {
2221        if col_starts.last().is_none_or(|&last| l - last > tol) {
2222            col_starts.push(l);
2223        }
2224    }
2225    let ncols = col_starts.len().max(1);
2226    let col_of = |l: f32| -> usize {
2227        col_starts
2228            .iter()
2229            .rposition(|&s| l + tol * 0.5 >= s)
2230            .unwrap_or(0)
2231            .min(ncols - 1)
2232    };
2233
2234    let mut grid = Vec::with_capacity(rows.len());
2235    for (_, mut row) in rows {
2236        row.sort_by(|a, b| a.l.total_cmp(&b.l));
2237        let mut cols = vec![String::new(); ncols];
2238        for c in row {
2239            let ci = col_of(c.l);
2240            // Strip the wrap-hyphen control char so it never lands in a cell.
2241            let t = c.text.trim().replace(['\u{2}', '\u{ad}'], "");
2242            if cols[ci].is_empty() {
2243                cols[ci] = t;
2244            } else {
2245                cols[ci].push(' ');
2246                cols[ci].push_str(&t);
2247            }
2248        }
2249        grid.push(cols);
2250    }
2251    grid
2252}
2253
2254/// Does the geometric reconstruction of a table look trustworthy enough to use
2255/// as-is, instead of paying for TableFormer?
2256///
2257/// [`reconstruct_table`] derives columns by clustering cell **left edges**. On a
2258/// clean grid that is exact, but when a column's entries are not left-aligned
2259/// (or the OCR boxes wobble) the clustering splits one real column into several,
2260/// and the result is a wide, mostly-empty grid — the "spurious empty columns"
2261/// failure TableFormer exists to fix.
2262///
2263/// Two symptoms separate the two cases, and both are properties of the grid
2264/// alone (no model needed):
2265/// * **density** — a real table is mostly full; a split-up one is mostly holes;
2266/// * **thin columns** — a column carrying at most one entry across several rows
2267///   is almost always a split artefact rather than a real column.
2268///
2269/// Deliberately conservative: it answers `true` only for grids that are plainly
2270/// well-formed, so the expensive path stays the default whenever there is doubt.
2271/// A caller that skips TableFormer on `true` trades no quality for the time.
2272pub fn geometric_table_is_reliable(rows: &[Vec<String>]) -> bool {
2273    let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
2274    // Fewer than two columns is not a grid this heuristic can vouch for: it is
2275    // exactly the shape a collapsed table takes, and TableFormer may recover
2276    // real structure from it.
2277    if rows.len() < 2 || ncols < 2 {
2278        return false;
2279    }
2280    let filled = |c: &String| !c.trim().is_empty();
2281    let total = rows.len() * ncols;
2282    let full = rows.iter().flatten().filter(|c| filled(c)).count();
2283    if (full as f32) < MIN_TABLE_FILL * total as f32 {
2284        return false;
2285    }
2286    // A column used by at most one row, when there are rows enough to tell.
2287    if rows.len() >= 3 {
2288        for ci in 0..ncols {
2289            let used = rows
2290                .iter()
2291                .filter(|r| r.get(ci).is_some_and(filled))
2292                .count();
2293            if used <= 1 {
2294                return false;
2295            }
2296        }
2297    }
2298    true
2299}
2300
2301/// Share of a geometric grid's cells that must carry text for it to be trusted
2302/// without TableFormer. Chosen well above the density a left-edge split
2303/// produces (those land nearer a third) and below what a genuine table with a
2304/// few blank cells reaches.
2305const MIN_TABLE_FILL: f32 = 0.6;
2306
2307/// The union bbox of the text cells assigned to a region (same >50%-overlap
2308/// rule as [`region_text`]), or `None` when no cell lands in it. docling's
2309/// LayoutPostprocessor shrinks a regular cluster's bbox to its cells, and the
2310/// enrichment crops are taken from that cell-tight box — cropping the raw
2311/// detector box instead hands the VLM surrounding chrome (e.g. the `Listing N:`
2312/// caption under a code block) that changes its output.
2313pub fn region_cell_bbox(region: &Region, cells: &[TextCell]) -> Option<[f32; 4]> {
2314    let mut bbox: Option<[f32; 4]> = None;
2315    for c in cells {
2316        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2317        if inter(region, c.l, c.t, c.r, c.b) / ca <= 0.5 {
2318            continue;
2319        }
2320        bbox = Some(match bbox {
2321            None => [c.l, c.t, c.r, c.b],
2322            Some([l, t, r, b]) => [l.min(c.l), t.min(c.t), r.max(c.r), b.max(c.b)],
2323        });
2324    }
2325    bbox
2326}
2327
2328/// One region's enrichment-model result, produced by the pipeline's opt-in
2329/// passes (issue #76) and applied during assembly.
2330#[derive(Debug, Clone)]
2331pub enum Enrichment {
2332    /// DocumentPictureClassifier predictions, descending confidence.
2333    PictureClasses(Vec<PictureClass>),
2334    /// CodeFormulaV2 output for a `code` region: the rewritten source text and
2335    /// the `<_language_>` prefix (when the model emitted one).
2336    Code {
2337        language: Option<String>,
2338        text: String,
2339    },
2340    /// CodeFormulaV2 output for a `formula` region: the decoded LaTeX.
2341    Formula { latex: String },
2342}
2343
2344/// Crop a region (page points, already expanded by the caller if needed) from
2345/// the rendered page image and resize it to `target_scale` pixels per point —
2346/// the enrichment-model equivalent of docling's
2347/// `page.get_image(scale=…, cropbox=…)`, sourced from the existing
2348/// [`crate::pdfium_backend::RENDER_SCALE`] render instead of a fresh pdfium
2349/// pass (the page bitmap is already the exact docling render at scale 2).
2350#[cfg(feature = "ml")]
2351pub fn crop_region_scaled(page: &PdfPage, bbox: [f32; 4], target_scale: f32) -> Option<RgbImage> {
2352    let s = page.scale;
2353    let [l, t, r, b] = bbox;
2354    let (iw, ih) = (page.image.width(), page.image.height());
2355    let x = (l * s).max(0.0) as u32;
2356    let y = (t * s).max(0.0) as u32;
2357    if x >= iw || y >= ih {
2358        return None;
2359    }
2360    let w = (((r - l.max(0.0)) * s) as u32).min(iw - x);
2361    let h = (((b - t.max(0.0)) * s) as u32).min(ih - y);
2362    if w == 0 || h == 0 {
2363        return None;
2364    }
2365    let crop = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
2366    // docling renders the crop at `target_scale` directly; from the scale-2
2367    // page render that is a resize to the same pixel geometry
2368    // (`round(width_points * scale)`, PIL's BICUBIC ≙ CatmullRom).
2369    let tw = ((w as f32 / s) * target_scale).round().max(1.0) as u32;
2370    let th = ((h as f32 / s) * target_scale).round().max(1.0) as u32;
2371    if (tw, th) == (w, h) {
2372        return Some(crop);
2373    }
2374    Some(image::imageops::resize(
2375        &crop,
2376        tw,
2377        th,
2378        image::imageops::FilterType::CatmullRom,
2379    ))
2380}
2381
2382/// Crop a layout region from the rendered page image and encode it as PNG (the
2383/// figure bytes docling stores on a `PictureItem`). Region coordinates are page
2384/// points; the image is rendered at `page.scale`.
2385#[cfg(feature = "ocr-prep")]
2386fn crop_region(page: &PdfPage, region: &Region) -> Option<PictureImage> {
2387    let s = page.scale;
2388    let (iw, ih) = (page.image.width(), page.image.height());
2389    let x = (region.l * s).max(0.0) as u32;
2390    let y = (region.t * s).max(0.0) as u32;
2391    if x >= iw || y >= ih {
2392        return None;
2393    }
2394    let w = (((region.r - region.l) * s) as u32).min(iw - x);
2395    let h = (((region.b - region.t) * s) as u32).min(ih - y);
2396    if w == 0 || h == 0 {
2397        return None;
2398    }
2399    let sub = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
2400    let mut buf = std::io::Cursor::new(Vec::new());
2401    sub.write_to(&mut buf, image::ImageFormat::Png).ok()?;
2402    Some(PictureImage {
2403        mimetype: "image/png".into(),
2404        width: w,
2405        height: h,
2406        data: buf.into_inner(),
2407    })
2408}
2409
2410/// For each `picture` region, find the `caption` region closest below it (and
2411/// horizontally overlapping); docling pairs them and emits the caption first.
2412/// Each caption is claimed by at most one picture.
2413fn pair_captions(regions: &[Region]) -> Vec<Option<usize>> {
2414    let mut pairs = vec![None; regions.len()];
2415    let mut taken = vec![false; regions.len()];
2416    for (pi, p) in regions.iter().enumerate() {
2417        if p.label != "picture" {
2418            continue;
2419        }
2420        let mut best: Option<(usize, f32)> = None;
2421        for (ci, c) in regions.iter().enumerate() {
2422            if c.label != "caption" || taken[ci] {
2423                continue;
2424            }
2425            let line_h = (c.b - c.t).abs().max(1.0);
2426            let gap = c.t - p.b; // caption sits below the picture
2427            let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
2428            if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
2429                let dist = gap.abs();
2430                if best.is_none_or(|(_, bd)| dist < bd) {
2431                    best = Some((ci, dist));
2432                }
2433            }
2434        }
2435        if let Some((ci, _)) = best {
2436            pairs[pi] = Some(ci);
2437            taken[ci] = true;
2438        }
2439    }
2440    pairs
2441}
2442
2443/// Pair each `code` region with the `caption` region just **above** it (a
2444/// `Listing N:` label). docling renders the code block first, then its caption,
2445/// so the caption is consumed from its own (earlier) reading-order slot and
2446/// re-emitted after the code.
2447fn pair_code_captions(regions: &[Region]) -> Vec<Option<usize>> {
2448    let mut pairs = vec![None; regions.len()];
2449    let mut taken = vec![false; regions.len()];
2450    for (pi, p) in regions.iter().enumerate() {
2451        if p.label != "code" {
2452            continue;
2453        }
2454        let mut best: Option<(usize, f32)> = None;
2455        for (ci, c) in regions.iter().enumerate() {
2456            if c.label != "caption" || taken[ci] {
2457                continue;
2458            }
2459            let line_h = (c.b - c.t).abs().max(1.0);
2460            let gap = p.t - c.b; // caption sits above the code
2461            let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
2462            if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
2463                let dist = gap.abs();
2464                if best.is_none_or(|(_, bd)| dist < bd) {
2465                    best = Some((ci, dist));
2466                }
2467            }
2468        }
2469        if let Some((ci, _)) = best {
2470            pairs[pi] = Some(ci);
2471            taken[ci] = true;
2472        }
2473    }
2474    pairs
2475}
2476
2477/// Pair each `table`/`document_index` region with its `caption` (#265) the way
2478/// docling's `ReadingOrderPredictor._find_to_captions` does: by **reading-order
2479/// adjacency**, not geometry. A caption claims the media element
2480/// (table/picture/code) immediately next to it in the ordered region sequence,
2481/// and only when exactly one side holds one — a caption sandwiched between two
2482/// media elements stays unattached, and a text paragraph between caption and
2483/// table breaks the bond. This is what lets a flush-left "Table 3: …" label
2484/// bind a centered grid it doesn't horizontally overlap, while a caption in
2485/// the neighbouring column of a two-column page — geometrically close — never
2486/// pairs across the gutter. Runs after the picture and code pairings (the
2487/// picture/code arms of the same upstream matcher), so a caption they claimed
2488/// stays claimed. docling attaches these as `TableItem.captions` refs; the
2489/// paired caption is consumed from its own reading-order slot and rides on the
2490/// table node instead.
2491fn pair_table_captions(regions: &[Region], taken: &mut [bool]) -> Vec<Option<usize>> {
2492    let is_media = |label: &str| is_table_like(label) || matches!(label, "picture" | "code");
2493    let mut pairs: Vec<Option<usize>> = vec![None; regions.len()];
2494    for ci in 0..regions.len() {
2495        if regions[ci].label != "caption" || taken[ci] {
2496            continue;
2497        }
2498        // Furniture (headers/footers, form chrome) is not part of docling's
2499        // body-element sequence, so it neither bonds nor blocks.
2500        let prev = regions[..ci].iter().rposition(|r| !is_skipped(r.label));
2501        let next = regions[ci + 1..]
2502            .iter()
2503            .position(|r| !is_skipped(r.label))
2504            .map(|off| ci + 1 + off);
2505        let prev_media = prev.is_some_and(|j| is_media(regions[j].label));
2506        let next_media = next.is_some_and(|j| is_media(regions[j].label));
2507        let target = match (prev_media, next_media) {
2508            (true, false) => prev,
2509            (false, true) => next,
2510            // Ambiguous (media on both sides) or no media at all: leave the
2511            // caption in its own reading-order slot, as docling does.
2512            _ => None,
2513        };
2514        if let Some(ti) = target {
2515            // A first claim wins (a table with captions above *and* below
2516            // keeps the earlier one — docling's nearest-first tiebreak).
2517            if is_table_like(regions[ti].label) && pairs[ti].is_none() {
2518                pairs[ti] = Some(ci);
2519                taken[ci] = true;
2520            }
2521        }
2522    }
2523    pairs
2524}
2525
2526/// Assemble one page from its (already overlap-resolved) layout regions and
2527/// text cells.
2528/// Normalize a layout region (page points, top-left origin) to DocLang's 0–511
2529/// location grid: `clamp(round(512 · coord / page_dim), 0, 511)`, per axis,
2530/// order `[x0, y0, x1, y1]`. Mirrors docling_core's
2531/// `_doclang_utils._create_location_tokens_for_bbox` (resolution 512) so the
2532/// emitted `<location>` tokens line up with the Python groundtruth. Our heron
2533/// cluster boxes match docling's to within ~1 grid unit; the residual (mainly
2534/// the aspect-ratio-stretch vs letterbox preprocessing difference) is absorbed
2535/// by the conformance harness's geometry tolerance.
2536fn norm_loc(region: &Region, page_w: f32, page_h: f32) -> [u16; 4] {
2537    let q = |v: f32, dim: f32| -> u16 {
2538        if dim <= 0.0 {
2539            return 0;
2540        }
2541        let g = (512.0 * (v as f64) / (dim as f64)).round() as i64;
2542        g.clamp(0, 511) as u16
2543    };
2544    [
2545        q(region.l, page_w),
2546        q(region.t, page_h),
2547        q(region.r, page_w),
2548        q(region.b, page_h),
2549    ]
2550}
2551
2552/// Wrap a node in its layout provenance so the DocLang serializer emits the four
2553/// `<location>` tokens as the element's head (Markdown/JSON render `inner`
2554/// unchanged).
2555fn located(loc: [u16; 4], inner: Node) -> Node {
2556    Node::Located {
2557        location: loc,
2558        inner: Box::new(inner),
2559    }
2560}
2561
2562/// Stamp the real 1-based page number onto a page's leading marker (see
2563/// [`assemble_page`], which emits it with `page_no: 0` because only the
2564/// document-level collector knows the true index — `--pages` windows shift it).
2565pub fn stamp_page_no(nodes: &mut [Node], page_no: usize) {
2566    if let Some(Node::PageInfo { page_no: p, .. }) = nodes.first_mut() {
2567        *p = page_no;
2568    }
2569}
2570
2571/// A dense table grid plus its first-class cells (#240): `rows` is the text
2572/// grid every serializer renders (spans replicate their anchor's text);
2573/// `cells` are the docling-parity per-cell records (text, page-point bbox,
2574/// span rectangle, OTSL header roles). Produced by the TableFormer paths
2575/// (`tf_core`); lives in this always-compiled module so the pure-text (wasm
2576/// `pdf-text`) build sees the type.
2577#[derive(Clone, Debug)]
2578pub struct TableGrid {
2579    pub rows: Vec<Vec<String>>,
2580    pub cells: Vec<docling_core::TableCell>,
2581}
2582
2583/// docling's `_RICH_CELL_PICTURE_COVERAGE_THRESHOLD`.
2584const RICH_CELL_PICTURE_COVERAGE: f32 = 0.8;
2585
2586/// docling `ReadingOrderModel._match_table_pictures` (#3906, 2.118.1): every
2587/// picture ≥ 80 % inside a TableFormer-structured table on the page is matched
2588/// to the cell covering it, and returned per table as `cell index → pictures`.
2589/// A picture that pairs with a caption stays a standalone figure (upstream
2590/// would nest it and lose the caption; keeping the caption is the better
2591/// failure). Tables without first-class cells (geometric fallback) have no cell
2592/// boxes to match against and nest nothing.
2593fn match_table_pictures(
2594    regions: &[Region],
2595    table_rows: &[Option<TableGrid>],
2596    caption_for: &[Option<usize>],
2597) -> std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> {
2598    let mut out: std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> =
2599        std::collections::HashMap::new();
2600    for (p, pic) in regions.iter().enumerate() {
2601        if pic.label != "picture" || caption_for.get(p).is_some_and(Option::is_some) {
2602            continue;
2603        }
2604        let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2605        let mut best: Option<(f32, usize, usize)> = None; // (coverage, table, cell)
2606        for (t, tbl) in regions.iter().enumerate() {
2607            if !is_table_like(tbl.label) {
2608                continue;
2609            }
2610            let Some(grid) = table_rows.get(t).and_then(Option::as_ref) else {
2611                continue;
2612            };
2613            if inter(pic, tbl.l, tbl.t, tbl.r, tbl.b) / pa < RICH_CELL_PICTURE_COVERAGE {
2614                continue;
2615            }
2616            if let Some((cov, cell)) = match_picture_to_cell(pic, &grid.cells) {
2617                if best.is_none_or(|(b, _, _)| cov > b) {
2618                    best = Some((cov, t, cell));
2619                }
2620            }
2621        }
2622        if let Some((_, t, cell)) = best {
2623            let entry = out.entry(t).or_default();
2624            match entry.iter_mut().find(|(c, _)| *c == cell) {
2625                Some((_, pics)) => pics.push(p),
2626                None => entry.push((cell, vec![p])),
2627            }
2628        }
2629    }
2630    out
2631}
2632
2633/// docling `_match_picture_to_table_cell`: among the cells covering ≥ 80 % of
2634/// the picture, prefer the one at the picture's inferred grid position (the
2635/// row / column whose median cell center is nearest the picture's center —
2636/// cell boxes can overlap across logical rows and columns), else the best
2637/// coverage. Returns `(coverage, cell index)`.
2638fn match_picture_to_cell(pic: &Region, cells: &[docling_core::TableCell]) -> Option<(f32, usize)> {
2639    let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2640    let cover = |b: &[f32; 4]| inter(pic, b[0], b[1], b[2], b[3]) / pa;
2641    let eligible: Vec<(f32, usize)> = cells
2642        .iter()
2643        .enumerate()
2644        .filter_map(|(i, c)| {
2645            let b = c.bbox.as_ref()?;
2646            let cov = cover(b);
2647            (cov >= RICH_CELL_PICTURE_COVERAGE).then_some((cov, i))
2648        })
2649        .collect();
2650    if eligible.is_empty() {
2651        return None;
2652    }
2653    let mut row_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2654    let mut col_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2655    for c in cells {
2656        let Some(b) = c.bbox.as_ref() else { continue };
2657        for r in c.start_row..c.start_row + c.row_span {
2658            row_centers.entry(r).or_default().push((b[1] + b[3]) / 2.0);
2659        }
2660        for k in c.start_col..c.start_col + c.col_span {
2661            col_centers.entry(k).or_default().push((b[0] + b[2]) / 2.0);
2662        }
2663    }
2664    let median = |v: &mut Vec<f32>| -> f32 {
2665        v.sort_by(f32::total_cmp);
2666        let n = v.len();
2667        if n % 2 == 1 {
2668            v[n / 2]
2669        } else {
2670            (v[n / 2 - 1] + v[n / 2]) / 2.0
2671        }
2672    };
2673    let (px, py) = ((pic.l + pic.r) / 2.0, (pic.t + pic.b) / 2.0);
2674    let nearest = |centers: &mut std::collections::BTreeMap<usize, Vec<f32>>, target: f32| {
2675        centers
2676            .iter_mut()
2677            .map(|(&i, v)| (i, (median(v) - target).abs()))
2678            .min_by(|a, b| a.1.total_cmp(&b.1))
2679            .map(|(i, _)| i)
2680    };
2681    let row = nearest(&mut row_centers, py);
2682    let col = nearest(&mut col_centers, px);
2683    let logical: Vec<(f32, usize)> = eligible
2684        .iter()
2685        .copied()
2686        .filter(|&(_, i)| {
2687            let c = &cells[i];
2688            row.is_some_and(|r| c.start_row <= r && r < c.start_row + c.row_span)
2689                && col.is_some_and(|k| c.start_col <= k && k < c.start_col + c.col_span)
2690        })
2691        .collect();
2692    let pool = if logical.is_empty() {
2693        &eligible
2694    } else {
2695        &logical
2696    };
2697    // Python's `max` over `(coverage, cell_index, cell)` tuples: highest
2698    // coverage, ties to the higher index.
2699    pool.iter()
2700        .copied()
2701        .max_by(|a, b| a.0.total_cmp(&b.0).then(a.1.cmp(&b.1)))
2702}
2703
2704/// The DocLang structure overlay derived from first-class cells: span
2705/// continuations (`lcel`/`ucel`/`xcel`) and per-cell header roles, so the
2706/// PDF path's DCLX carries real spans instead of a flat grid.
2707fn structure_from_cells(
2708    cells: &[docling_core::TableCell],
2709    nrows: usize,
2710    ncols: usize,
2711) -> docling_core::TableStructure {
2712    let grid = || vec![vec![false; ncols]; nrows];
2713    let mut col_cont = grid();
2714    let mut row_cont = grid();
2715    let mut row_header = grid();
2716    let mut col_header = grid();
2717    for c in cells {
2718        for r in c.start_row..(c.start_row + c.row_span).min(nrows) {
2719            for k in c.start_col..(c.start_col + c.col_span).min(ncols) {
2720                col_cont[r][k] = k > c.start_col;
2721                row_cont[r][k] = r > c.start_row;
2722                row_header[r][k] = c.row_header;
2723                col_header[r][k] = c.column_header;
2724            }
2725        }
2726    }
2727    docling_core::TableStructure {
2728        header_row: Vec::new(),
2729        col_continuation: col_cont,
2730        row_continuation: row_cont,
2731        row_header,
2732        col_header,
2733    }
2734}
2735
2736pub fn assemble_page(
2737    page: &PdfPage,
2738    regions: Vec<Region>,
2739    table_rows: &[Option<TableGrid>],
2740    enrichments: &[Option<Enrichment>],
2741) -> (Vec<Node>, Vec<(String, String)>) {
2742    let mut nodes: Vec<Node> = Vec::new();
2743    // Every page opens with an invisible page marker carrying its size in
2744    // points — what the JSON export needs to build docling's `pages` map and
2745    // denormalize the 0–511 `<location>` grid into point bboxes (#171). The
2746    // page *number* is stamped by the document-level collector (which knows
2747    // the real 1-based index, `--pages` windows included); every serializer
2748    // except JSON skips the marker, so Markdown/DocLang stay byte-identical.
2749    nodes.push(Node::PageInfo {
2750        page_no: 0,
2751        width: page.width,
2752        height: page.height,
2753    });
2754    // Recover this page's hyperlinks (anchor-precise pairs for strict
2755    // Markdown; whole-item docling-parity links are baked below and their
2756    // pairs dropped from this list so strict output doesn't double-wrap).
2757    let mut links = resolve_link_anchors(page);
2758    // Pair each region with its precomputed TableFormer grid and enrichment
2759    // (indexed by original order) and order by reading order together, so they
2760    // stay aligned.
2761    // A picture's children (docling's `_set_cluster_children`: the regulars
2762    // > 80 % inside it) are not page elements — they leave the reading order
2763    // here and ride with their picture, to be written under it in the JSON.
2764    let parents = picture_parents(&regions);
2765    let mut kids: Vec<Vec<Region>> = vec![Vec::new(); regions.len()];
2766    let mut top: Vec<(usize, Region)> = Vec::with_capacity(regions.len());
2767    for (i, (r, parent)) in regions.into_iter().zip(parents).enumerate() {
2768        match parent {
2769            Some(p) => kids[p].push(r),
2770            None => top.push((i, r)),
2771        }
2772    }
2773    // docling's assembly order of the regions — what its reading-order
2774    // predictor knows as `cid` (#424) — before they are shuffled.
2775    let top_regions: Vec<Region> = top.iter().map(|(_, r)| r.clone()).collect();
2776    let cids = cluster_cids(&top_regions, &page.cells);
2777    type RegionItem = (Region, Option<TableGrid>, Option<Enrichment>, Vec<Region>);
2778    let mut items: Vec<RegionItem> = top
2779        .into_iter()
2780        .map(|(i, r)| {
2781            (
2782                r,
2783                table_rows.get(i).cloned().flatten(),
2784                enrichments.get(i).cloned().flatten(),
2785                std::mem::take(&mut kids[i]),
2786            )
2787        })
2788        .collect();
2789    order_with_containers(&mut items, &cids, page.width, page.height, |it| &it.0);
2790    // Float a margin page number to the front of reading order (docling parity:
2791    // right_to_left_02's bottom `11` is its first item). Stable, so everything
2792    // else keeps its order; no-op on pages without such a region.
2793    let page_h = page.height;
2794    items.sort_by_key(|(r, _, _, _)| !is_page_number(r, &page.cells, page_h));
2795    let table_rows: Vec<Option<TableGrid>> = items.iter().map(|(_, t, _, _)| t.clone()).collect();
2796    let enrichments: Vec<Option<Enrichment>> = items.iter().map(|(_, _, e, _)| e.clone()).collect();
2797    let mut picture_children: Vec<Vec<Region>> = items
2798        .iter_mut()
2799        .map(|it| std::mem::take(&mut it.3))
2800        .collect();
2801    let regions: Vec<Region> = items.into_iter().map(|(r, _, _, _)| r).collect();
2802    // Children in docling's `_sort_clusters(mode="id")` order: first source
2803    // cell, then top, then left.
2804    for kids in picture_children.iter_mut().filter(|k| k.len() > 1) {
2805        let rank = cluster_cids(kids, &page.cells);
2806        let mut ranked: Vec<(usize, Region)> = rank.into_iter().zip(kids.drain(..)).collect();
2807        ranked.sort_by_key(|(k, _)| *k);
2808        kids.extend(ranked.into_iter().map(|(_, r)| r));
2809    }
2810    // docling emits a figure's caption *before* the image marker. Pair each
2811    // picture with the caption region nearest below it and consume that caption,
2812    // so it isn't also emitted in its own (lower) reading-order position.
2813    let caption_for = pair_captions(&regions);
2814    let code_caption_for = pair_code_captions(&regions);
2815    let mut consumed = vec![false; regions.len()];
2816    for ci in caption_for.iter().flatten() {
2817        consumed[*ci] = true;
2818    }
2819    for ci in code_caption_for.iter().flatten() {
2820        consumed[*ci] = true;
2821    }
2822    // Table captions (#265) claim from what the picture/code pairings left.
2823    let mut caption_taken = consumed.clone();
2824    let table_caption_for = pair_table_captions(&regions, &mut caption_taken);
2825    for ci in table_caption_for.iter().flatten() {
2826        consumed[*ci] = true;
2827    }
2828    // Pictures inside a table become rich-cell content (docling#3906, 2.118.1):
2829    // the picture is nested in the cell it covers and not emitted standalone.
2830    let rich_cell_pictures = match_table_pictures(&regions, &table_rows, &caption_for);
2831    for (_, pics) in rich_cell_pictures.values().flatten() {
2832        for &p in pics {
2833            consumed[p] = true;
2834        }
2835    }
2836    // A code block's language label (`XML`, `C#`, …) is chrome, not content — the
2837    // detector emits it as its own region above the code; consume it.
2838    for (i, is_label) in code_language_labels(&regions, &page.cells)
2839        .into_iter()
2840        .enumerate()
2841    {
2842        if is_label {
2843            consumed[i] = true;
2844        }
2845    }
2846
2847    // docling `ReadingOrderPredictor.predict_merges`: join a text fragment with a
2848    // following text fragment strictly to its right (an author column that wraps
2849    // into the next, a paragraph continuing in the next column) into one block —
2850    // the intra-page half of docling's reading-order merges (cross-page/vertical
2851    // continuations stay with [`merge_continuations`]). Already-consumed regions
2852    // (paired captions, code labels) are excluded.
2853    // Exclusive docling cell assignment: computed once for the ordered region
2854    // list and reused for every serialization below, so a cell can never render
2855    // in two regions. The picture children take part (docling assigns cells to
2856    // every regular cluster before it nests any); their texts are split off.
2857    let with_children: Vec<Region> = regions
2858        .iter()
2859        .chain(picture_children.iter().flatten())
2860        .cloned()
2861        .collect();
2862    let mut region_texts: Vec<String> = region_texts_exclusive(&with_children, &page.cells);
2863    let mut kid_texts = region_texts.split_off(regions.len()).into_iter();
2864    let child_texts: Vec<Vec<String>> = picture_children
2865        .iter()
2866        .map(|k| kid_texts.by_ref().take(k.len()).collect())
2867        .collect();
2868    let is_text: Vec<bool> = regions
2869        .iter()
2870        .enumerate()
2871        .map(|(i, r)| r.label == "text" && !consumed[i])
2872        .collect();
2873    let is_skip: Vec<bool> = regions
2874        .iter()
2875        .enumerate()
2876        .map(|(i, r)| {
2877            consumed[i]
2878                || matches!(
2879                    r.label,
2880                    "page_header" | "page_footer" | "table" | "picture" | "caption" | "footnote"
2881                )
2882        })
2883        .collect();
2884    let boxes: Vec<(f32, f32, f32, f32)> = regions.iter().map(|r| (r.l, r.t, r.r, r.b)).collect();
2885    if docling_core::env::flag("DOCLING_RS_DEBUG_MERGES") {
2886        for (i, r) in regions.iter().enumerate() {
2887            eprintln!(
2888                "MRG {i:2} {} text={} skip={} [{:.0},{:.0},{:.0},{:.0}] {:?}",
2889                r.label,
2890                is_text[i],
2891                is_skip[i],
2892                r.l,
2893                r.t,
2894                r.r,
2895                r.b,
2896                region_texts[i].chars().take(40).collect::<String>()
2897            );
2898        }
2899    }
2900    let mut merge_suffix: Vec<String> = vec![String::new(); regions.len()];
2901    for (head, children) in
2902        crate::reading_order::predict_merges(&boxes, &region_texts, &is_text, &is_skip)
2903            .into_iter()
2904            .enumerate()
2905    {
2906        for c in children {
2907            let t = region_texts[c].trim();
2908            if !t.is_empty() {
2909                merge_suffix[head].push(' ');
2910                merge_suffix[head].push_str(t);
2911            }
2912            consumed[c] = true;
2913        }
2914    }
2915
2916    for (i, region) in regions.iter().enumerate() {
2917        if consumed[i] {
2918            continue;
2919        }
2920        // Page headers/footers: docling emits them as furniture blocks
2921        // (`<page_header>`/`<page_footer>` with a layer + location + text) at
2922        // their reading-order position, not as body — emit them, don't skip.
2923        if matches!(region.label, "page_header" | "page_footer") {
2924            let text = region_texts[i].clone();
2925            if !text.is_empty() {
2926                nodes.push(Node::PageFurniture {
2927                    footer: region.label == "page_footer",
2928                    location: norm_loc(region, page.width, page_h),
2929                    text: md_escape(&text),
2930                });
2931            }
2932            continue;
2933        }
2934        if is_skipped(region.label) {
2935            continue;
2936        }
2937        // Layout provenance for this region, normalized to docling's 0–511 grid.
2938        let loc = norm_loc(region, page.width, page_h);
2939        if region.label == "picture" {
2940            // The figure pixels are cropped from the page render for image export.
2941            // Captions are prose: markdown-escaped like a paragraph (the JSON
2942            // export unescapes back to the raw text, matching docling).
2943            let caption = caption_for[i]
2944                .map(|ci| md_escape(&region_texts[ci]))
2945                .filter(|t| !t.is_empty());
2946            let classification = match &enrichments[i] {
2947                Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
2948                _ => None,
2949            };
2950            // Without the page render (text-layer-only build) a picture keeps
2951            // its caption/classification but carries no cropped pixels.
2952            #[cfg(feature = "ocr-prep")]
2953            let image = crate::timing::timed("crop_region", || crop_region(page, region));
2954            #[cfg(not(feature = "ocr-prep"))]
2955            let image: Option<PictureImage> = None;
2956            nodes.push(located(
2957                loc,
2958                Node::Picture {
2959                    caption,
2960                    caption_href: None,
2961                    image,
2962                    classification,
2963                    // docling's layout pipeline parents a figure's caption to
2964                    // the picture itself (#390) — the one backend that does.
2965                    caption_parent: CaptionParent::Item,
2966                },
2967            ));
2968            let children: Vec<Node> = picture_children[i]
2969                .iter()
2970                .zip(&child_texts[i])
2971                .filter_map(|(r, text)| {
2972                    picture_child_node(r, text, norm_loc(r, page.width, page_h))
2973                })
2974                .collect();
2975            if !children.is_empty() {
2976                nodes.push(Node::PictureChildren(children));
2977            }
2978            continue;
2979        }
2980        let mut text = region_texts[i].clone();
2981        text.push_str(&merge_suffix[i]);
2982        if text.is_empty() {
2983            continue;
2984        }
2985        match region.label {
2986            // docling assembles checkboxes as TEXT_ELEM items (the region's
2987            // cells are the option label, e.g. right_to_left_03's بلی/خير)
2988            // and its Markdown serializer renders them as task-list lines
2989            // (`- [x] …`) — mirrored by [`Node::CheckboxItem`].
2990            "checkbox_selected" | "checkbox_unselected" => nodes.push(Node::CheckboxItem {
2991                checked: region.label == "checkbox_selected",
2992                text: md_escape(&text),
2993            }),
2994            // docling renders both the document title and section headers as
2995            // `##` (it never emits a top-level `#` for PDFs), so match that.
2996            "title" | "section_header" => nodes.push(located(
2997                loc,
2998                Node::Heading {
2999                    level: 2,
3000                    text: md_escape(&text),
3001                },
3002            )),
3003            // docling's `ListItemMarkerProcessor.process_list_item` runs on
3004            // every PDF list item: a leading bullet glyph or enumeration marker
3005            // followed by whitespace is split off into the item's `marker`, and
3006            // docling-core's Markdown then prints `- text` for a bullet, `N. text`
3007            // for an `N.` marker and `- a) text` for any other marker holding a
3008            // letter or digit (see [`list_item_node`]). The symbol-font bullets
3009            // docling-parse filters out of its cells are stripped first.
3010            "list_item" => nodes.push(list_item_node(&text, loc, false)),
3011            // TableFormer structure (cells + spans, text matched from word cells)
3012            // when available; otherwise geometric grid reconstruction; finally a
3013            // single cell.
3014            "table" | "document_index" => {
3015                // TableFormer grids carry first-class cells (#240: text +
3016                // page-point bbox + span rectangle + OTSL header roles) into
3017                // the public model, and the DocLang structure overlay derives
3018                // from them so DCLX emits real span/header tokens. The
3019                // geometric fallback has no per-cell records.
3020                let (mut rows, cells, structure) = match table_rows[i].clone() {
3021                    Some(grid) => {
3022                        let nrows = grid.rows.len();
3023                        let ncols = grid.rows.first().map_or(0, Vec::len);
3024                        let structure = structure_from_cells(&grid.cells, nrows, ncols);
3025                        (grid.rows, Some(grid.cells), Some(structure))
3026                    }
3027                    None => {
3028                        let rows = reconstruct_table(region, &page.cells);
3029                        let rows = if rows.iter().any(|r| r.len() > 1) {
3030                            rows
3031                        } else {
3032                            vec![vec![text.clone()]]
3033                        };
3034                        (rows, None, None)
3035                    }
3036                };
3037                // The paired caption (#265) rides on the table — docling's
3038                // TableItem.captions ref; Markdown prints it above the grid,
3039                // the JSON export emits the $ref, DocLang the <caption>.
3040                let caption = table_caption_for[i]
3041                    .map(|ci| md_escape(&region_texts[ci]))
3042                    .filter(|t| !t.is_empty());
3043                // Rich cells (docling#3906): the covering cell's blocks are its
3044                // text followed by the nested picture(s). docling's Markdown
3045                // renders a `RichTableCell` through the serializer — the
3046                // group's children joined by blank lines, newlines flattened
3047                // to spaces — so the flat `rows` text becomes
3048                // `text  <!-- image -->`; the first-class `cells` (the JSON
3049                // `table_cells` / `grid`) keep the plain text, as upstream.
3050                let mut cell_blocks: Option<Vec<Vec<Vec<Node>>>> = None;
3051                if let (Some(by_cell), Some(fc)) = (rich_cell_pictures.get(&i), cells.as_ref()) {
3052                    let nrows = rows.len();
3053                    let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
3054                    let mut blocks = vec![vec![Vec::<Node>::new(); ncols]; nrows];
3055                    for (cell_idx, pics) in by_cell {
3056                        let cell = &fc[*cell_idx];
3057                        let (r, c) = (cell.start_row, cell.start_col);
3058                        if r >= nrows || c >= ncols {
3059                            continue;
3060                        }
3061                        let mut parts: Vec<String> = Vec::new();
3062                        let mut cell_nodes: Vec<Node> = Vec::new();
3063                        if !cell.text.trim().is_empty() {
3064                            parts.push(cell.text.clone());
3065                            cell_nodes.push(Node::Paragraph {
3066                                text: cell.text.clone(),
3067                            });
3068                        }
3069                        for &p in pics {
3070                            parts.push("<!-- image -->".to_string());
3071                            let classification = match &enrichments[p] {
3072                                Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
3073                                _ => None,
3074                            };
3075                            #[cfg(feature = "ocr-prep")]
3076                            let image = crop_region(page, &regions[p]);
3077                            #[cfg(not(feature = "ocr-prep"))]
3078                            let image: Option<PictureImage> = None;
3079                            cell_nodes.push(located(
3080                                norm_loc(&regions[p], page.width, page_h),
3081                                Node::Picture {
3082                                    caption: None,
3083                                    caption_href: None,
3084                                    image,
3085                                    classification,
3086                                    caption_parent: Default::default(),
3087                                },
3088                            ));
3089                        }
3090                        let rendered = parts.join("  ");
3091                        for row in rows.iter_mut().skip(r).take(cell.row_span) {
3092                            for slot in row.iter_mut().skip(c).take(cell.col_span) {
3093                                *slot = rendered.clone();
3094                            }
3095                        }
3096                        blocks[r][c] = cell_nodes;
3097                    }
3098                    cell_blocks = Some(blocks);
3099                }
3100                nodes.push(located(
3101                    loc,
3102                    Node::Table(Table {
3103                        rows,
3104                        location: None,
3105                        structure,
3106                        cell_blocks,
3107                        cells,
3108                        caption,
3109                        // As for pictures: the caption is the table's child.
3110                        caption_parent: CaptionParent::Item,
3111                    }),
3112                ));
3113            }
3114            // With formula enrichment the CodeFormula model decodes the region
3115            // to LaTeX; otherwise docling emits a placeholder comment rather
3116            // than the (garbled) raw glyph text.
3117            "formula" => match &enrichments[i] {
3118                Some(Enrichment::Formula { latex }) => nodes.push(Node::Formula {
3119                    latex: latex.clone(),
3120                    orig: text.clone(),
3121                    location: Some(loc),
3122                }),
3123                _ => nodes.push(Node::Paragraph {
3124                    text: "<!-- formula-not-decoded -->".into(),
3125                }),
3126            },
3127            // Code blocks: use the space-glyph-only grouping (monospace keeps its
3128            // source spacing) and emit a fenced block, preserving the line breaks
3129            // and indentation of the source (unlike prose, which reflows). pdfium
3130            // still inserts spaces around tight punctuation (`console .log`,
3131            // `add (3 , 5)`); tighten them to match docling-parse's source spacing.
3132            "code" => {
3133                // `code_region_text` preserves line breaks/indentation and tightens
3134                // each line itself; the fallback prose `text` is tightened here.
3135                let code = code_region_text(region, &page.code_cells);
3136                let code = if code.is_empty() {
3137                    tighten_code_punct(&text)
3138                } else {
3139                    code
3140                };
3141                // With code enrichment the CodeFormula model rewrites the block
3142                // (and names its language); `orig` keeps the raw extraction in
3143                // docling's shape — its parser has no line-preserving code
3144                // path, so its `orig` is the same code with the lines joined
3145                // by single spaces (indentation collapsed).
3146                // docling's parser has no line-preserving code path — its code
3147                // items carry the lines joined by single spaces. That flat
3148                // form is what every byte-conformance surface serializes
3149                // (legacy Markdown, JSON, DocLang); the line-preserving
3150                // extraction rides in `pretty` for strict Markdown only.
3151                let flat = code
3152                    .lines()
3153                    .map(str::trim)
3154                    .filter(|l| !l.is_empty())
3155                    .collect::<Vec<_>>()
3156                    .join(" ");
3157                let node = match &enrichments[i] {
3158                    Some(Enrichment::Code {
3159                        language,
3160                        text: enriched,
3161                    }) => Node::Code {
3162                        language: language.clone(),
3163                        text: enriched.clone(),
3164                        orig: Some(flat),
3165                        pretty: None,
3166                    },
3167                    _ => Node::Code {
3168                        language: None,
3169                        text: flat,
3170                        orig: None,
3171                        pretty: Some(code),
3172                    },
3173                };
3174                nodes.push(located(loc, node));
3175                // docling emits the `Listing N:` caption after the code block.
3176                if let Some(ci) = code_caption_for[i] {
3177                    let cap = md_escape(&region_texts[ci]);
3178                    if !cap.is_empty() {
3179                        nodes.push(Node::Paragraph { text: cap });
3180                    }
3181                }
3182            }
3183            // text, caption, footnote → paragraph
3184            _ => {
3185                // docling parity (`PageAssembleModel._match_hyperlink`): when
3186                // link annotations cover ≥ half of the region's box, the
3187                // hyperlink attaches to the item and the legacy Markdown
3188                // serializer wraps its full text — 2206.01062's footnote URLs
3189                // render as `[1 https://…](https://…)`. Sparse in-paragraph
3190                // citation links stay below the 0.5 coverage threshold and
3191                // remain plain text, exactly like docling.
3192                //
3193                // Scope: **footnote regions only.** Upstream's page_assemble
3194                // matches every TEXT_ELEM label, but published docling
3195                // observably carries the hyperlink into the document only for
3196                // footnote items — in both committed groundtruth generations
3197                // (docling-JSON and Markdown, independent runs) the fully
3198                // covered plain-text DOI line of 2206.01062 page 1 has
3199                // `hyperlink: None` while the equally covered footnotes carry
3200                // theirs. The corpus is the conformance reference, so match
3201                // the observed behavior; widen the label set if a future
3202                // groundtruth refresh starts linking plain text too.
3203                let escaped = md_escape(&text);
3204                let hyperlink = (region.label == "footnote")
3205                    .then(|| region_hyperlink(region, &page.links))
3206                    .flatten();
3207                let text = match hyperlink {
3208                    Some(uri) => {
3209                        // The strict-mode anchor pairs this item covers are
3210                        // superseded by the baked whole-item link.
3211                        links.retain(|(anchor, href)| {
3212                            !(href == &uri && region_texts[i].contains(anchor.as_str()))
3213                        });
3214                        format!("[{escaped}]({uri})")
3215                    }
3216                    None => escaped,
3217                };
3218                nodes.push(located(loc, Node::Paragraph { text }))
3219            }
3220        }
3221    }
3222    // A `/Rotate`-normalized scanned page (see `pdfium_backend`) was assembled
3223    // in upright space; rotate the finished geometry back so locations and the
3224    // page size are display-space, like docling and every viewer report them.
3225    if page.rotation != 0 {
3226        rotate_nodes_to_display(&mut nodes, page.rotation);
3227    }
3228    (nodes, links)
3229}
3230
3231/// One child of a picture as docling's `ReadingOrderModel._add_child_elements`
3232/// writes it under the `PictureItem`: a heading for a `section_header` /
3233/// `title` (upstream remaps title to section header), a list item for a
3234/// `list_item`, a furniture-layer text for a page header/footer, a caption,
3235/// otherwise a text item. `None` for a child that claimed no text.
3236fn picture_child_node(region: &Region, text: &str, loc: [u16; 4]) -> Option<Node> {
3237    if text.is_empty() {
3238        return None;
3239    }
3240    Some(match region.label {
3241        "title" | "section_header" => located(
3242            loc,
3243            Node::Heading {
3244                level: 2,
3245                text: md_escape(text),
3246            },
3247        ),
3248        // docling-core's `add_list_item` under a non-list parent opens a
3249        // list group per item, so every child item starts its own list;
3250        // `_add_child_elements` runs the marker processor on it too.
3251        "list_item" => list_item_node(text, loc, true),
3252        "page_header" | "page_footer" => Node::PageFurniture {
3253            footer: region.label == "page_footer",
3254            location: loc,
3255            text: md_escape(text),
3256        },
3257        "caption" => located(
3258            loc,
3259            Node::Caption {
3260                text: md_escape(text),
3261                href: None,
3262            },
3263        ),
3264        _ => located(
3265            loc,
3266            Node::Paragraph {
3267                text: md_escape(text),
3268            },
3269        ),
3270    })
3271}
3272
3273/// Rotate one 0–511 location bbox 90° clockwise on the grid (top-left origin):
3274/// `(x, y) → (511 - y, x)`.
3275fn rot_loc_cw(l: [u16; 4]) -> [u16; 4] {
3276    [511 - l[3], l[0], 511 - l[1], l[2]]
3277}
3278
3279/// Map upright-space geometry back to display space for a page whose `/Rotate`
3280/// was normalized away before inference: every `<location>` rotates `rot`°
3281/// clockwise on the 0–511 grid (the grid is per-axis normalized, so no page
3282/// dims are needed), and the `PageInfo` size returns to the display box. Node
3283/// text and order are untouched — reading order was decided upright, which is
3284/// the whole point.
3285fn rotate_nodes_to_display(nodes: &mut [Node], rot: u16) {
3286    let quarter_turns = (rot / 90) as usize;
3287    let rot_loc = |l: &mut [u16; 4]| {
3288        for _ in 0..quarter_turns {
3289            *l = rot_loc_cw(*l);
3290        }
3291    };
3292    fn walk(node: &mut Node, rot_loc: &impl Fn(&mut [u16; 4]), swap_dims: bool) {
3293        match node {
3294            Node::PageInfo { width, height, .. } => {
3295                if swap_dims {
3296                    std::mem::swap(width, height);
3297                }
3298            }
3299            Node::Located { location, inner } => {
3300                rot_loc(location);
3301                walk(inner, rot_loc, swap_dims);
3302            }
3303            Node::Furniture { inner, .. } => walk(inner, rot_loc, swap_dims),
3304            Node::Group { children, .. } | Node::PictureChildren(children) => {
3305                for c in children {
3306                    walk(c, rot_loc, swap_dims);
3307                }
3308            }
3309            Node::ListItem { location, .. }
3310            | Node::Formula { location, .. }
3311            | Node::Chart { location, .. } => {
3312                if let Some(l) = location {
3313                    rot_loc(l);
3314                }
3315            }
3316            Node::PageFurniture { location, .. } => rot_loc(location),
3317            Node::Table(t) => {
3318                if let Some(l) = &mut t.location {
3319                    rot_loc(l);
3320                }
3321            }
3322            _ => {}
3323        }
3324    }
3325    let swap_dims = quarter_turns % 2 == 1;
3326    for node in nodes {
3327        walk(node, &rot_loc, swap_dims);
3328    }
3329}
3330
3331/// Merge paragraph fragments split across a column or page break. docling joins a
3332/// paragraph whose previous fragment ends mid-sentence (a letter, not sentence
3333/// punctuation) with a lowercase continuation: `…definition of` + `lists in…` →
3334/// `…definition of lists in…`. The fragments are consecutive paragraphs, or
3335/// separated only by figure(s) the text wraps around: a column whose body flows
3336/// past a figure resumes below it (`…The wing type that is` ⟶[figure]⟶ `the most
3337/// common…`), and docling emits the whole paragraph before the figure. A heading,
3338/// table, or list between them ends the paragraph (no merge).
3339/// A paragraph that is really a figure/table caption (`Fig. 1. …`, `Table 2 …`).
3340/// Used to skip an unpaired caption when stitching a paragraph that wraps around
3341/// a figure.
3342fn looks_like_caption(text: &str) -> bool {
3343    let head: String = text.trim_start().chars().take(14).collect();
3344    (head.starts_with("Fig") || head.starts_with("Table"))
3345        && head.contains(|c: char| c.is_ascii_digit())
3346}
3347
3348/// A paragraph fragment is "open" — i.e. it might continue into the next
3349/// paragraph — when it ends mid-word (a letter) or with a wrap hyphen/dash.
3350/// docling joins `vocab-` + `ulary` → `vocab- ulary`.
3351fn paragraph_is_open(text: &str) -> bool {
3352    // docling's merge head test (`.+([a-z,\-\u00AD])\s*`): at least two chars,
3353    // ending in an ASCII lowercase letter, a comma, a hyphen, or a soft
3354    // hyphen. The comma matters: 2206's "…In phase four," resumes across the
3355    // page break. Uppercase/non-Latin endings do not merge, exactly as
3356    // upstream (the dash family is already `-` here — clean_text normalized).
3357    let t = text.trim_end();
3358    t.chars().count() >= 2
3359        && t.chars()
3360            .next_back()
3361            .is_some_and(|c| matches!(c, 'a'..='z' | ',' | '-' | '\u{ad}'))
3362}
3363
3364/// The paragraph text inside a node, looking through a [`Node::Located`]
3365/// provenance wrapper (PDF body paragraphs are wrapped since they carry a
3366/// `<location>`). Returns `None` for non-paragraph nodes.
3367fn as_paragraph(n: &Node) -> Option<&str> {
3368    match n {
3369        Node::Paragraph { text } => Some(text),
3370        Node::Located { inner, .. } => match inner.as_ref() {
3371            Node::Paragraph { text } => Some(text),
3372            _ => None,
3373        },
3374        _ => None,
3375    }
3376}
3377
3378/// Whether a node is a picture, looking through a [`Node::Located`] wrapper.
3379fn is_picture_node(n: &Node) -> bool {
3380    match n {
3381        Node::Picture { .. } => true,
3382        Node::Located { inner, .. } => matches!(inner.as_ref(), Node::Picture { .. }),
3383        _ => false,
3384    }
3385}
3386
3387/// A node a forward paragraph merge looks straight past: a figure or *table*
3388/// the text wraps around, or a page header/footer that falls between the two
3389/// fragments of a paragraph continuing across a page break (docling's merge
3390/// skip-labels: page_header, page_footer, table, picture, caption, footnote —
3391/// 2206's "…In phase four," resumes after a full caption+table+figure block).
3392fn is_merge_trailer(n: &Node) -> bool {
3393    is_picture_node(n)
3394        || matches!(
3395            n,
3396            Node::PageFurniture { .. }
3397                | Node::PageInfo { .. }
3398                | Node::Table(_)
3399                | Node::PictureChildren(_)
3400        )
3401        || matches!(n, Node::Located { inner, .. } if matches!(inner.as_ref(), Node::Table(_)))
3402        || as_paragraph(n).is_some_and(looks_like_caption)
3403}
3404
3405/// Rebuild node `i` as a paragraph with `text`, preserving its `<location>`
3406/// wrapper (and thus provenance) if it had one.
3407fn reparagraph(node: &Node, text: String) -> Node {
3408    match node {
3409        Node::Located { location, .. } => located(*location, Node::Paragraph { text }),
3410        _ => Node::Paragraph { text },
3411    }
3412}
3413
3414pub(crate) fn merge_continuations(nodes: &mut Vec<Node>) {
3415    let mut i = 0;
3416    while i + 1 < nodes.len() {
3417        let Some(a) = as_paragraph(&nodes[i]) else {
3418            i += 1;
3419            continue;
3420        };
3421        // A figure/table caption is a self-contained unit; body text resuming
3422        // after a figure is the continuation case, not the caption itself. Never
3423        // stitch *from* a caption — otherwise a caption that ends in a lone glyph
3424        // (`Fig. 5. … PubTabNet. μ`) would swallow a following stray figure label
3425        // (a standalone `μ`) into `… μ μ`.
3426        if looks_like_caption(a) {
3427            i += 1;
3428            continue;
3429        }
3430        if !paragraph_is_open(a) {
3431            i += 1;
3432            continue;
3433        }
3434        // The continuation is the next paragraph, looking past any figures the
3435        // text wraps around — and a figure/table caption that was emitted as its
3436        // own paragraph (an above-the-figure caption that didn't pair), since the
3437        // body text resumes after the whole figure+caption block.
3438        let mut j = i + 1;
3439        while nodes.get(j).is_some_and(is_merge_trailer) {
3440            j += 1;
3441        }
3442        // docling's continuation regex allows either case, but its merge runs
3443        // over the pre-assembly element stream; at node level an uppercase
3444        // start is overwhelmingly a new sentence/heading fragment (allowing it
3445        // swallowed 2305's formula blocks and redp's chapter openers), so the
3446        // continuation stays lowercase-start here.
3447        let cont = nodes.get(j).and_then(as_paragraph).is_some_and(|b| {
3448            b.trim_start()
3449                .chars()
3450                .next()
3451                .is_some_and(char::is_lowercase)
3452        });
3453        if cont {
3454            let a = as_paragraph(&nodes[i]).unwrap().trim_end().to_string();
3455            let b = as_paragraph(&nodes[j]).unwrap().trim_start().to_string();
3456            // A soft hyphen -- or a hard hyphen followed by a lowercase
3457            // continuation (guaranteed lowercase by the `cont` gate above) --
3458            // is a word split across the break: strip it and join without a
3459            // space, docling#3888 ("vocab-" + "ulary" -> "vocabulary");
3460            // docling's older serializer kept the artifact ("vocab- ulary").
3461            // Everything else joins with the space, as before.
3462            let merged = match a.strip_suffix('\u{ad}').or_else(|| a.strip_suffix('-')) {
3463                Some(stem) => format!("{stem}{b}"),
3464                None => format!("{a} {b}"),
3465            };
3466            // Keep node i's provenance wrapper; docling's merged paragraph keeps
3467            // the first fragment's geometry as its primary location.
3468            nodes[i] = reparagraph(&nodes[i], merged);
3469            nodes.remove(j);
3470            // Re-check i: the merged paragraph may continue further.
3471        } else {
3472            i += 1;
3473        }
3474    }
3475}
3476
3477/// How many leading nodes of `nodes` are safe to flush now — i.e. cannot be
3478/// rewritten by a future [`merge_continuations`] once more pages are appended.
3479///
3480/// A forward merge can only start from an "open" paragraph (ends mid-word) and
3481/// only reaches across trailing pictures and figure/table captions. So we scan
3482/// from the end past those skippable trailers: if the first non-skippable node is
3483/// an open paragraph, it (and the trailers after it) must be held; anything else —
3484/// a closed paragraph, a heading, a table, a list — blocks any forward merge, so
3485/// the whole buffer is safe to flush.
3486fn hold_start(nodes: &[Node]) -> usize {
3487    for k in (0..nodes.len()).rev() {
3488        // Skippable trailers (figures, page furniture, captions): a forward merge
3489        // looks straight past them.
3490        if is_merge_trailer(&nodes[k]) {
3491            continue;
3492        }
3493        match as_paragraph(&nodes[k]) {
3494            // An open body paragraph might still pull a continuation off the next
3495            // page — hold from here to the end.
3496            Some(text) if paragraph_is_open(text) => return k,
3497            // A closed paragraph, heading, table, list, etc. ends the paragraph:
3498            // nothing after it can merge backwards across it. Flush everything.
3499            _ => return nodes.len(),
3500        }
3501    }
3502    // Only skippable trailers (or empty) and no open paragraph to anchor a merge.
3503    nodes.len()
3504}
3505
3506/// Streaming counterpart of [`merge_continuations`]: feed per-page node batches in
3507/// document order and get back the prefix that is final (its cross-page merges are
3508/// resolved and no future page can change it), holding back only the small tail
3509/// that might still merge into the next page. Concatenating every flushed batch
3510/// (then [`finish`](Self::finish)) yields exactly the same nodes as running
3511/// [`merge_continuations`] once over the whole document.
3512pub(crate) struct StreamAssembler {
3513    pending: Vec<Node>,
3514}
3515
3516impl StreamAssembler {
3517    pub(crate) fn new() -> Self {
3518        Self {
3519            pending: Vec::new(),
3520        }
3521    }
3522
3523    /// Append one page's nodes, resolve merges within the buffer, and return the
3524    /// now-final prefix to emit (possibly empty).
3525    pub(crate) fn push(&mut self, mut nodes: Vec<Node>) -> Vec<Node> {
3526        self.pending.append(&mut nodes);
3527        merge_continuations(&mut self.pending);
3528        let cut = hold_start(&self.pending);
3529        let tail = self.pending.split_off(cut);
3530        std::mem::replace(&mut self.pending, tail)
3531    }
3532
3533    /// Flush whatever is left after the last page (the held tail is final once no
3534    /// more pages can follow).
3535    pub(crate) fn finish(self) -> Vec<Node> {
3536        self.pending
3537    }
3538}
3539
3540#[cfg(test)]
3541mod tests {
3542    use super::{cells_text, clean_text, merge_overlapping_regulars, reclaim_heading_body_footers};
3543
3544    /// docling drops a picture covering > 90 % of the page (its labels then
3545    /// read out as text); a dominant-but-not-full figure and any other label
3546    /// stay whatever their size.
3547    #[test]
3548    fn full_page_pictures_are_dropped_like_docling() {
3549        use super::drop_full_page_pictures;
3550        use crate::layout::Region;
3551        let region = |label: &'static str, l, t, r, b| Region {
3552            label,
3553            score: 0.99,
3554            l,
3555            t,
3556            r,
3557            b,
3558        };
3559        let mut regions = vec![
3560            region("picture", 0.0, 0.5, 478.9, 241.8),
3561            region("picture", 10.0, 10.0, 400.0, 200.0),
3562            region("table", 0.0, 0.0, 480.0, 243.0),
3563            region("text", 5.0, 5.0, 100.0, 20.0),
3564        ];
3565        drop_full_page_pictures(&mut regions, 480.75, 243.75);
3566        let labels: Vec<_> = regions.iter().map(|r| (r.label, r.l)).collect();
3567        assert_eq!(
3568            labels,
3569            vec![("picture", 10.0), ("table", 0.0), ("text", 5.0)]
3570        );
3571    }
3572    use super::{code_region_text, merge_continuations, resolve_link_anchors, StreamAssembler};
3573    use crate::layout::Region;
3574    use crate::pdfium_backend::{LinkAnnot, PdfPage, TextCell};
3575    use docling_core::Node;
3576
3577    /// The int8-layout guard's coverage metric: cells under detections count,
3578    /// cells outside don't, whitespace cells are ignored, and a cell-less page
3579    /// reads as fully covered (nothing to rescue).
3580    #[test]
3581    fn layout_cell_coverage_counts_claimed_text_cells() {
3582        let cell = |text: &str, l: f32, t: f32| TextCell {
3583            text: text.into(),
3584            l,
3585            t,
3586            r: l + 40.0,
3587            b: t + 10.0,
3588        };
3589        let region = Region {
3590            label: "text",
3591            score: 0.9,
3592            l: 0.0,
3593            t: 0.0,
3594            r: 100.0,
3595            b: 50.0,
3596        };
3597        let cells = vec![
3598            cell("inside", 10.0, 10.0),
3599            cell("also inside", 10.0, 30.0),
3600            cell("outside", 10.0, 200.0),
3601            cell("   ", 10.0, 210.0), // whitespace: not counted at all
3602        ];
3603        let cov = super::layout_cell_coverage(std::slice::from_ref(&region), &cells);
3604        assert!((cov - 2.0 / 3.0).abs() < 1e-6, "got {cov}");
3605        assert_eq!(super::layout_cell_coverage(&[], &[]), 1.0);
3606        assert_eq!(super::layout_cell_coverage(&[], &cells), 0.0);
3607    }
3608
3609    /// #165: a picture no longer claims cells at 0.2 intersection-over-self.
3610    /// A line straddling the figure border (≤80 % contained) becomes an orphan
3611    /// region and is emitted as page text — before the fix its cells were
3612    /// silently erased. A line fully inside the picture is the picture's child
3613    /// (docling's `_set_cluster_children`): it survives the containment drop,
3614    /// leaves the page's reading order, and is written only under the picture
3615    /// in the JSON — never in the Markdown, like docling's picture serializer.
3616    #[test]
3617    fn border_straddlers_are_page_text_picture_interior_is_a_picture_child() {
3618        let pic = Region {
3619            label: "picture",
3620            score: 0.9,
3621            l: 0.0,
3622            t: 0.0,
3623            r: 100.0,
3624            b: 100.0,
3625        };
3626        // ~35 % of this cell overlaps the picture (l=90..120 of 0..100): above
3627        // the old 0.2 claim (was swallowed), below full containment (survives).
3628        let straddler = TextCell {
3629            text: "axis label".into(),
3630            l: 90.0,
3631            t: 40.0,
3632            r: 120.0,
3633            b: 48.0,
3634        };
3635        let interior = TextCell {
3636            text: "in-figure callout".into(),
3637            l: 10.0,
3638            t: 10.0,
3639            r: 60.0,
3640            b: 18.0,
3641        };
3642        let cells = vec![straddler, interior];
3643        let mut regions = vec![pic];
3644        super::add_orphan_regions(&mut regions, &cells);
3645        super::drop_contained_regulars(&mut regions);
3646        assert_eq!(
3647            regions.iter().filter(|r| r.label == "text").count(),
3648            2,
3649            "both unclaimed lines become orphans, and a picture swallows neither"
3650        );
3651        let parents = super::picture_parents(&regions);
3652        let parent_of = |l: f32| {
3653            regions
3654                .iter()
3655                .zip(&parents)
3656                .find(|(r, _)| r.label == "text" && r.l == l)
3657                .and_then(|(_, p)| *p)
3658        };
3659        assert_eq!(
3660            parent_of(10.0),
3661            Some(0),
3662            "the callout is the picture's child"
3663        );
3664        assert_eq!(parent_of(90.0), None, "the straddler is a page element");
3665
3666        let page = PdfPage::from_cells(200.0, 200.0, 2.0, cells);
3667        let n = regions.len();
3668        let (nodes, _) = super::assemble_page(&page, regions, &vec![None; n], &vec![None; n]);
3669        let children: Vec<&Node> = nodes
3670            .iter()
3671            .filter_map(|n| match n {
3672                Node::PictureChildren(c) => Some(c),
3673                _ => None,
3674            })
3675            .flatten()
3676            .collect();
3677        assert!(
3678            matches!(children.as_slice(), [Node::Located { inner, .. }]
3679                if matches!(inner.as_ref(), Node::Paragraph { text } if text == "in-figure callout")),
3680            "{children:?}"
3681        );
3682        let mut doc = docling_core::DoclingDocument::new("t");
3683        doc.nodes = nodes;
3684        let md = doc.export_to_markdown();
3685        assert!(md.contains("axis label"), "{md}");
3686        assert!(!md.contains("in-figure callout"), "{md}");
3687        let json = doc.export_to_json_value();
3688        let pic = &json["pictures"][0];
3689        let child = pic["children"][0]["$ref"].as_str().expect("a child ref");
3690        let idx: usize = child.rsplit('/').next().unwrap().parse().unwrap();
3691        assert_eq!(json["texts"][idx]["text"], "in-figure callout");
3692        assert_eq!(json["texts"][idx]["parent"]["$ref"], "#/pictures/0");
3693        assert_eq!(json["texts"][idx]["content_layer"], "body");
3694    }
3695
3696    /// docling#3906's concern, pinned on our side: a picture detected fully
3697    /// inside a table region must survive the containment drop (upstream now
3698    /// attaches it to the table's cell; we keep it as a body sibling — either
3699    /// way it must not vanish). The text region inside the same table is the
3700    /// control: regulars are the ones the drop swallows.
3701    #[test]
3702    fn picture_inside_a_table_region_survives_the_containment_drop() {
3703        let mut regions = vec![
3704            region("table", 0.9, 0.0, 0.0, 200.0, 200.0),
3705            region("picture", 0.9, 20.0, 20.0, 120.0, 120.0),
3706            region("text", 0.9, 20.0, 140.0, 180.0, 180.0),
3707        ];
3708        super::drop_contained_regulars(&mut regions);
3709        let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
3710        assert_eq!(
3711            labels,
3712            ["table", "picture"],
3713            "the in-table picture stays; the in-table regular is the special's child"
3714        );
3715    }
3716
3717    /// Table–caption pairing (#265) is reading-order adjacency, docling's
3718    /// `_find_to_captions`: a caption binds the table directly next to it in
3719    /// the region sequence — above-caption and below-caption both work, and
3720    /// geometry is irrelevant (a same-page caption in the other column of a
3721    /// two-column layout is *not* adjacent, however close its box is). A
3722    /// caption with media on both sides, or separated from the table by a
3723    /// text paragraph, stays unattached.
3724    #[test]
3725    fn table_captions_pair_by_reading_order_adjacency() {
3726        // caption → table (above-caption), then table → caption (below-caption),
3727        // then a caption fenced off by a paragraph, then one between two tables.
3728        let regions = vec![
3729            region("text", 0.9, 0.0, 0.0, 100.0, 10.0), // 0 body text
3730            region("caption", 0.9, 0.0, 12.0, 60.0, 20.0), // 1 above-caption
3731            region("table", 0.9, 20.0, 22.0, 90.0, 60.0), // 2 ← pairs with 1
3732            region("table", 0.9, 0.0, 70.0, 100.0, 110.0), // 3 ← pairs with 4
3733            region("caption", 0.9, 0.0, 112.0, 60.0, 120.0), // 4 below-caption
3734            region("text", 0.9, 0.0, 130.0, 100.0, 140.0), // 5 body text
3735            region("caption", 0.9, 0.0, 142.0, 60.0, 150.0), // 6 fenced by 5/7
3736            region("text", 0.9, 0.0, 152.0, 100.0, 162.0), // 7 body text
3737            region("table", 0.9, 0.0, 170.0, 100.0, 200.0), // 8 unpaired
3738            region("caption", 0.9, 0.0, 202.0, 60.0, 210.0), // 9 ambiguous
3739            region("table", 0.9, 0.0, 212.0, 100.0, 240.0), // 10 unpaired
3740        ];
3741        let mut taken = vec![false; regions.len()];
3742        let pairs = super::pair_table_captions(&regions, &mut taken);
3743        assert_eq!(pairs[2], Some(1), "caption directly above its table pairs");
3744        assert_eq!(pairs[3], Some(4), "caption directly below its table pairs");
3745        assert_eq!(
3746            pairs[8], None,
3747            "a text paragraph between caption and table breaks the bond"
3748        );
3749        assert_eq!(
3750            pairs[10], None,
3751            "a caption between two tables is ambiguous and stays loose"
3752        );
3753        assert!(taken[1] && taken[4] && !taken[6] && !taken[9]);
3754    }
3755
3756    /// A colored terms-and-conditions panel detected as `picture` demotes into
3757    /// per-paragraph `text` regions (the blank line between C.7 and C.8 splits
3758    /// them); a chart whose only text is a few narrow axis labels keeps its
3759    /// crop untouched.
3760    #[test]
3761    fn text_panels_demote_to_paragraphs_but_charts_keep_their_crop() {
3762        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3763            text: text.to_string(),
3764            l,
3765            t,
3766            r,
3767            b,
3768        };
3769        let panel = Region {
3770            label: "picture",
3771            score: 0.9,
3772            l: 0.0,
3773            t: 0.0,
3774            r: 100.0,
3775            b: 100.0,
3776        };
3777        // Three tight lines, a blank-line gap, two more: two paragraphs.
3778        let cells = vec![
3779            cell(
3780                "C.7. Wenn Sie diesen Vertrag widerrufen,",
3781                5.0,
3782                10.0,
3783                95.0,
3784                18.0,
3785            ),
3786            cell(
3787                "haben wir Ihnen alle Zahlungen, die wir",
3788                5.0,
3789                20.0,
3790                95.0,
3791                28.0,
3792            ),
3793            cell(
3794                "von Ihnen erhalten haben, zurückzuzahlen.",
3795                5.0,
3796                30.0,
3797                90.0,
3798                38.0,
3799            ),
3800            cell(
3801                "C.8. Wir können die Rückzahlung verweigern,",
3802                5.0,
3803                52.0,
3804                95.0,
3805                60.0,
3806            ),
3807            cell(
3808                "bis wir die Waren wieder zurückerhalten haben.",
3809                5.0,
3810                62.0,
3811                92.0,
3812                70.0,
3813            ),
3814        ];
3815        let mut regions = vec![panel.clone()];
3816        super::recover_text_panels(&mut regions, &cells);
3817        assert_eq!(
3818            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3819            ["text", "text"],
3820            "dense panel must demote into one text region per paragraph"
3821        );
3822        assert!(regions[0].b < regions[1].t, "paragraphs split at the gap");
3823        // Sparse narrow labels (a chart): picture survives.
3824        let labels = vec![
3825            cell("0", 5.0, 90.0, 8.0, 95.0),
3826            cell("50", 5.0, 50.0, 10.0, 55.0),
3827            cell("100", 5.0, 10.0, 12.0, 15.0),
3828            cell("t, s", 45.0, 96.0, 55.0, 100.0),
3829        ];
3830        let mut regions = vec![panel];
3831        super::recover_text_panels(&mut regions, &labels);
3832        assert_eq!(
3833            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3834            ["picture"]
3835        );
3836    }
3837
3838    /// An uncaptioned chart on a scanned page whose title, axis labels, and
3839    /// OCR boxes over the plot area are dense and wide enough to pass the
3840    /// coverage/width gates still keeps its crop: its line heights are ragged
3841    /// (title face vs tick labels vs bar-area OCR), failing the uniform-leading
3842    /// gate — a real text panel is set with constant leading (#173).
3843    #[test]
3844    fn dense_titled_chart_keeps_its_crop() {
3845        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3846            text: text.to_string(),
3847            l,
3848            t,
3849            r,
3850            b,
3851        };
3852        let chart = Region {
3853            label: "picture",
3854            score: 0.9,
3855            l: 0.0,
3856            t: 0.0,
3857            r: 100.0,
3858            b: 100.0,
3859        };
3860        // Five wide lines at wildly different heights: a 12-pt title, 20-pt OCR
3861        // boxes over the bars, 4–5-pt tick/axis labels. Coverage and median
3862        // width both clear the panel thresholds.
3863        let cells = vec![
3864            cell("Underground Water Storage", 10.0, 5.0, 90.0, 17.0),
3865            cell("aquifer recharge zone", 15.0, 30.0, 75.0, 50.0),
3866            cell("confined | unconfined | perched", 12.0, 55.0, 80.0, 59.0),
3867            cell("saturated thickness", 8.0, 70.0, 60.0, 90.0),
3868            cell("distance from well, km", 20.0, 92.0, 85.0, 97.0),
3869        ];
3870        let mut regions = vec![chart];
3871        super::recover_text_panels(&mut regions, &cells);
3872        assert_eq!(
3873            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3874            ["picture"],
3875            "ragged line heights mark a figure, not a text panel"
3876        );
3877    }
3878
3879    /// docling serializes a cluster's cells in docling-parse index order
3880    /// (`_sort_cells`) and joins them with `PageAssembleModel.sanitize_text`:
3881    /// a space after every line except one ending in `-`, which either fuses a
3882    /// wrapped word (alnum on both sides — dash dropped) or glues verbatim (a
3883    /// bare `-` cell: `[0000` `-` `0002` → `[0000 -0002`, the 2305 ORCID line;
3884    /// `-` + `"C" cell -` + `a new table cell` → `-"C" cell a new table cell`,
3885    /// its OTSL list). Verified against the corpus: pure index order beats any
3886    /// geometric re-sort (normal_4pages' heading numerals paint after their
3887    /// text and belong last: `## 들어가며 1`).
3888    #[test]
3889    fn cells_join_in_index_order_with_sanitize_text_rules() {
3890        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3891            text: text.to_string(),
3892            l,
3893            t,
3894            r,
3895            b,
3896        };
3897        let region = Region {
3898            label: "text",
3899            score: 1.0,
3900            l: 0.0,
3901            t: 95.0,
3902            r: 200.0,
3903            b: 130.0,
3904        };
3905        // ORCID superscript: a bare dash cell is a *detached* dash — kept, and
3906        // since docling#4052 (2.122) it joins with the ordinary space on both
3907        // sides (`[0000 -0002 -6960]` before that fix).
3908        let orcid = vec![
3909            cell("[0000", 10.0, 100.0, 30.0, 110.0),
3910            cell("−", 30.0, 100.0, 34.0, 110.0),
3911            cell("0002", 34.0, 100.0, 50.0, 110.0),
3912            cell("−", 50.0, 100.0, 54.0, 110.0),
3913            cell("6960]", 54.0, 100.0, 70.0, 110.0),
3914        ];
3915        assert_eq!(super::region_text(&region, &orcid), "[0000 - 0002 - 6960]");
3916        // Wrapped word: dash dropped, lines fused (both boundary words alnum).
3917        let wrapped = vec![
3918            cell("platforms-", 10.0, 100.0, 60.0, 110.0),
3919            cell("reflects the design", 10.0, 112.0, 90.0, 122.0),
3920        ];
3921        assert_eq!(
3922            super::region_text(&region, &wrapped),
3923            "platformsreflects the design"
3924        );
3925        // Dash-ending lines that are *detached* dashes (a bare bullet cell, a
3926        // `cell -` separator): the dash stays and the lines join with a space
3927        // — docling#4052; before it they glued (`-"C" cell a new table cell`,
3928        // 2305's OTSL list bullets).
3929        let otsl = vec![
3930            cell("–", 10.0, 100.0, 14.0, 110.0),
3931            cell("\"C\" cell -", 16.0, 100.0, 60.0, 110.0),
3932            cell("a new table cell", 10.0, 112.0, 80.0, 122.0),
3933        ];
3934        assert_eq!(
3935            super::region_text(&region, &otsl),
3936            "- \"C\" cell - a new table cell"
3937        );
3938        // Index order is authoritative — no geometric re-sort.
3939        let numeral = vec![
3940            cell("들어가며", 30.0, 100.0, 80.0, 110.0),
3941            cell("1", 10.0, 98.0, 25.0, 112.0), // big numeral painted last
3942        ];
3943        assert_eq!(super::region_text(&region, &numeral), "들어가며 1");
3944    }
3945
3946    /// The geometric-reliability gate, on the two shapes it has to tell apart.
3947    #[test]
3948    fn geometric_reliability_rejects_split_column_grids() {
3949        let g = |rows: &[&[&str]]| -> Vec<Vec<String>> {
3950            rows.iter()
3951                .map(|r| r.iter().map(|c| c.to_string()).collect())
3952                .collect()
3953        };
3954        // A genuine grid: dense, every column carrying entries. Nothing for
3955        // TableFormer to improve, so geometry is used as-is.
3956        assert!(super::geometric_table_is_reliable(&g(&[
3957            &["Datum", "Leistung", "Anzahl", "Kosten"],
3958            &["04.07", "Internet", "1", "40.30"],
3959            &["04.07", "Telefon", "2", "8.06"],
3960        ])));
3961        // The left-edge split artefact (the shape a scanned invoice produced):
3962        // one real label column plus values scattered across three sparse ones.
3963        assert!(!super::geometric_table_is_reliable(&g(&[
3964            &["www.magenta.at/faq", "", "", ""],
3965            &["Serviceteam", "", "", ""],
3966            &["Telefon", "0676/2000", "", ""],
3967            &["Kundennummer", "", "", "1.21699482"],
3968            &["Rechnungsnummer", "", "922769430725", ""],
3969            &["Rechnungsdatum", "", "", "04.07.2025"],
3970        ])));
3971        // A column only one row ever uses is a split artefact even when the
3972        // grid is otherwise dense.
3973        assert!(!super::geometric_table_is_reliable(&g(&[
3974            &["a", "b", ""],
3975            &["c", "d", ""],
3976            &["e", "f", "g"],
3977        ])));
3978        // Degenerate shapes are never vouched for — TableFormer may recover
3979        // structure a collapsed reconstruction lost.
3980        assert!(!super::geometric_table_is_reliable(&g(&[&[
3981            "only one column"
3982        ]])));
3983        assert!(!super::geometric_table_is_reliable(&[]));
3984    }
3985
3986    /// A `picture` region is cropped out of the rendered page, whatever built
3987    /// that page. The browser pipeline (#157) has no pdfium but does hand over
3988    /// the rasterized bitmap through `from_cells_with_image`, so it must get
3989    /// the same figure bytes the native path does — that is what makes
3990    /// `images = "embedded"` inline real pixels instead of a placeholder.
3991    #[cfg(feature = "ocr-prep")]
3992    #[test]
3993    fn picture_regions_are_cropped_from_a_host_supplied_page_image() {
3994        let mut img = image::RgbImage::new(200, 200);
3995        // Paint the figure area so the crop is distinguishable from the page.
3996        for y in 100..160 {
3997            for x in 20..120 {
3998                img.put_pixel(x, y, image::Rgb([255, 0, 0]));
3999            }
4000        }
4001        // scale 2.0: the region is in page points, the bitmap in pixels.
4002        let page = PdfPage::from_cells_with_image(100.0, 100.0, 2.0, Vec::new(), img);
4003        let region = Region {
4004            label: "picture",
4005            score: 0.9,
4006            l: 10.0,
4007            t: 50.0,
4008            r: 60.0,
4009            b: 80.0,
4010        };
4011        let (nodes, _) = super::assemble_page(&page, vec![region], &[None], &[None]);
4012        // Layout-derived nodes carry provenance, so the picture arrives wrapped.
4013        let image = nodes
4014            .iter()
4015            .find_map(|n| match n {
4016                Node::Located { inner, .. } => match &**inner {
4017                    Node::Picture { image, .. } => image.as_ref(),
4018                    _ => None,
4019                },
4020                Node::Picture { image, .. } => image.as_ref(),
4021                _ => None,
4022            })
4023            .expect("a picture node with cropped pixels");
4024        assert_eq!(image.mimetype, "image/png");
4025        assert_eq!((image.width, image.height), (100, 60), "region × scale");
4026        assert!(!image.data.is_empty(), "PNG bytes were encoded");
4027    }
4028
4029    #[test]
4030    fn link_anchors_split_a_shared_word_cell_between_adjacent_links() {
4031        // A common header layout: one text run holds several pipe-separated
4032        // labels, each carrying its own link annotation. Every link must get
4033        // its own label as the anchor (and the "|" separators must belong to
4034        // none), not the whole run.
4035        let annot = |l: f32, r: f32, uri: &str| LinkAnnot {
4036            l,
4037            t: 100.0,
4038            r,
4039            b: 114.0,
4040            uri: uri.into(),
4041        };
4042        let page = PdfPage {
4043            width: 600.0,
4044            height: 800.0,
4045            scale: 2.0,
4046            cells: Vec::new(),
4047            code_cells: Vec::new(),
4048            // "LinkedIn | GitHub | Credly" = 26 chars over x 100..360.
4049            word_cells: vec![cell(
4050                "LinkedIn | GitHub | Credly",
4051                100.0,
4052                100.0,
4053                360.0,
4054                114.0,
4055            )],
4056            image: image::RgbImage::new(1, 1),
4057            image_layout: None,
4058            links: vec![
4059                annot(100.0, 180.0, "https://l"),
4060                annot(200.0, 260.0, "https://g"),
4061                annot(290.0, 360.0, "https://c"),
4062            ],
4063            rotation: 0,
4064        };
4065        assert_eq!(
4066            resolve_link_anchors(&page),
4067            vec![
4068                ("LinkedIn".to_string(), "https://l".to_string()),
4069                ("GitHub".to_string(), "https://g".to_string()),
4070                ("Credly".to_string(), "https://c".to_string()),
4071            ]
4072        );
4073    }
4074
4075    /// A one-line code cell at `[l, r] × [t, b]` (top-left coords).
4076    fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
4077        TextCell {
4078            text: text.into(),
4079            l,
4080            t,
4081            r,
4082            b,
4083        }
4084    }
4085
4086    /// OCR-path grouping (docling's `_remove_overlapping_clusters("regular")`):
4087    /// the low-score paragraph box RT-DETR draws over its own high-score line
4088    /// boxes collapses to one region — the group's union, with the survivor's
4089    /// label and score — so region-scoped OCR reads each line once. Regions
4090    /// that merely sit near each other, and specials, are untouched.
4091    #[test]
4092    fn merge_overlapping_regulars_collapses_a_block_over_its_lines() {
4093        let mut regions = vec![
4094            region("text", 0.84, 60.0, 186.0, 270.0, 198.0),
4095            region("text", 0.80, 60.0, 160.0, 294.0, 172.0),
4096            region("text", 0.79, 59.0, 107.0, 272.0, 119.0),
4097            // The paragraph box, lower score, containing all three lines.
4098            region("text", 0.52, 59.0, 107.0, 295.0, 200.0),
4099            // Elsewhere on the page: stays as is.
4100            region("section_header", 0.77, 60.0, 71.0, 253.0, 86.0),
4101            // A picture the block overlaps is not a regular — never grouped.
4102            region("picture", 0.9, 50.0, 100.0, 300.0, 210.0),
4103        ];
4104        merge_overlapping_regulars(&mut regions);
4105        assert_eq!(regions.len(), 3, "{regions:?}");
4106        let block = regions
4107            .iter()
4108            .find(|r| r.label == "text")
4109            .expect("one text");
4110        // docling keeps the largest passing candidate unless a rival is both
4111        // comparable in size and > 0.05 more confident; the 16× larger block
4112        // passes, and a smaller line never replaces a larger current best.
4113        // Either way the survivor spans the whole group.
4114        assert_eq!(
4115            (block.l, block.t, block.r, block.b),
4116            (59.0, 107.0, 295.0, 200.0)
4117        );
4118        assert!(regions.iter().any(|r| r.label == "section_header"));
4119        assert!(regions.iter().any(|r| r.label == "picture"));
4120    }
4121
4122    /// The pairwise rules, each in the arrangement where it decides the
4123    /// outcome: docling seeds the survivor with the group's first passing
4124    /// cluster and a later one replaces it only when larger *and* within
4125    /// 0.05 confidence, so a rule that merely lets a cluster pass matters
4126    /// exactly when that cluster comes first — a same-sized list item ahead
4127    /// of a far more confident text box, a code box ahead of the text it
4128    /// contains. Without the rule either would be rejected outright (similar
4129    /// size, rival > 0.05 more confident) and the text box would win.
4130    #[test]
4131    fn merge_overlapping_regulars_follows_the_preference_rules() {
4132        let mut regions = vec![
4133            region("list_item", 0.6, 0.0, 0.0, 102.0, 20.0),
4134            region("text", 0.9, 0.0, 0.0, 100.0, 20.0),
4135        ];
4136        merge_overlapping_regulars(&mut regions);
4137        assert_eq!(regions.len(), 1);
4138        assert_eq!(regions[0].label, "list_item");
4139
4140        let mut regions = vec![
4141            region("code", 0.6, 0.0, 0.0, 100.0, 100.0),
4142            region("text", 0.9, 2.0, 2.0, 98.0, 98.0),
4143        ];
4144        merge_overlapping_regulars(&mut regions);
4145        assert_eq!(regions.len(), 1);
4146        assert_eq!(regions[0].label, "code");
4147
4148        // No rule applies: a near-identical rival that is > 0.05 more
4149        // confident rejects the candidate whatever the order.
4150        for order in [[0.9, 0.6], [0.6, 0.9]] {
4151            let mut regions = vec![
4152                region("text", order[0], 0.0, 0.0, 100.0, 20.0),
4153                region("text", order[1], 0.0, 0.0, 105.0, 21.0),
4154            ];
4155            merge_overlapping_regulars(&mut regions);
4156            assert_eq!(regions.len(), 1);
4157            assert_eq!(regions[0].score, 0.9, "the confident twin wins");
4158            assert_eq!(
4159                (regions[0].r, regions[0].b),
4160                (105.0, 21.0),
4161                "on the union box"
4162            );
4163        }
4164
4165        // Side by side (no containment, IoU 0): nothing to merge.
4166        let mut regions = vec![
4167            region("text", 0.9, 0.0, 0.0, 100.0, 20.0),
4168            region("text", 0.9, 0.0, 22.0, 100.0, 42.0),
4169        ];
4170        merge_overlapping_regulars(&mut regions);
4171        assert_eq!(regions.len(), 2);
4172    }
4173
4174    #[test]
4175    fn footer_under_a_body_less_heading_is_its_text() {
4176        // The reporting CV's last page: `## Languages` at t=773.7..783.4 and
4177        // its one-line body at 798.5..808.7, labelled page_footer (A4, 595 pt).
4178        let mut regions = vec![
4179            region("list_item", 0.9, 66.0, 754.0, 353.0, 765.0),
4180            region("section_header", 0.94, 43.0, 773.7, 89.5, 783.4),
4181            region("page_footer", 0.88, 43.0, 798.5, 527.8, 808.7),
4182        ];
4183        reclaim_heading_body_footers(&mut regions, 595.28);
4184        assert_eq!(regions[2].label, "text");
4185        assert_eq!(regions[1].label, "section_header");
4186
4187        // A heading with its own paragraph and a running footer below: kept.
4188        let mut regions = vec![
4189            region("section_header", 0.9, 43.0, 700.0, 120.0, 710.0),
4190            region("text", 0.9, 43.0, 714.0, 520.0, 780.0),
4191            region("page_footer", 0.9, 43.0, 798.0, 520.0, 808.0),
4192        ];
4193        reclaim_heading_body_footers(&mut regions, 595.28);
4194        assert_eq!(regions[2].label, "page_footer");
4195
4196        // A page number under a trailing heading is too narrow to be a body.
4197        let mut regions = vec![
4198            region("section_header", 0.9, 43.0, 773.0, 120.0, 783.0),
4199            region("page_footer", 0.9, 280.0, 798.0, 300.0, 808.0),
4200        ];
4201        reclaim_heading_body_footers(&mut regions, 595.28);
4202        assert_eq!(regions[1].label, "page_footer");
4203
4204        // Too far below the heading (a real footer after a heading that ends
4205        // the page): kept.
4206        let mut regions = vec![
4207            region("section_header", 0.9, 43.0, 700.0, 120.0, 710.0),
4208            region("page_footer", 0.9, 43.0, 798.0, 520.0, 808.0),
4209        ];
4210        reclaim_heading_body_footers(&mut regions, 595.28);
4211        assert_eq!(regions[1].label, "page_footer");
4212    }
4213
4214    fn region(label: &'static str, score: f32, l: f32, t: f32, r: f32, b: f32) -> Region {
4215        Region {
4216            label,
4217            score,
4218            l,
4219            t,
4220            r,
4221            b,
4222        }
4223    }
4224
4225    #[test]
4226    fn resolve_collapses_nested_code_keeping_the_larger_box() {
4227        // A tight high-score `code` box and a taller lower-score near-duplicate that
4228        // contains it must collapse to one — the *larger* box, so every cell stays
4229        // covered and nothing leaks out as orphan text.
4230        let tight = region("code", 0.95, 78.0, 292.0, 300.0, 330.0);
4231        let wide = region("code", 0.66, 63.0, 260.0, 320.0, 346.0);
4232        let kept = super::resolve(vec![tight, wide]);
4233        assert_eq!(kept.len(), 1, "nested code boxes must collapse to one");
4234        assert!(
4235            kept[0].l == 63.0 && kept[0].b == 346.0,
4236            "the larger containing box is kept"
4237        );
4238    }
4239
4240    #[test]
4241    fn resolve_keeps_distinct_and_differently_typed_regions() {
4242        // A text box fully inside a lower-score *table* must NOT be collapsed (the
4243        // code dedup is code-only), and two separate code blocks stay separate.
4244        let text = region("text", 0.95, 90.0, 210.0, 200.0, 230.0);
4245        let table = region("table", 0.60, 80.0, 200.0, 400.0, 500.0);
4246        assert_eq!(super::resolve(vec![text, table]).len(), 2);
4247
4248        let code_a = region("code", 0.9, 78.0, 100.0, 300.0, 140.0);
4249        let code_b = region("code", 0.9, 78.0, 300.0, 300.0, 360.0); // far below, no overlap
4250        assert_eq!(super::resolve(vec![code_a, code_b]).len(), 2);
4251    }
4252
4253    /// A two-column glossary page came out as three column
4254    /// tables *and* one low-score whole-page table over them. docling's wrapper
4255    /// `_remove_overlapping_clusters` keeps one table per overlapping group
4256    /// (here the whole-page one: > 2× every rival's area and ≤ 0.2 less
4257    /// confident than the running best); `greedy` alone kept all four and
4258    /// emitted every cell twice.
4259    #[test]
4260    fn resolve_keeps_one_table_per_nested_group() {
4261        let kept = super::resolve(vec![
4262            region("table", 0.71, 26.0, 203.0, 183.0, 558.0),
4263            region("table", 0.67, 26.0, 55.0, 183.0, 196.0),
4264            region("table", 0.66, 196.0, 56.0, 354.0, 561.0),
4265            region("table", 0.53, 25.0, 53.0, 354.0, 561.0),
4266        ]);
4267        assert_eq!(kept.len(), 1, "one survivor per overlapping group");
4268        assert_eq!((kept[0].l, kept[0].b), (25.0, 561.0));
4269        // Side-by-side tables that don't overlap stay separate.
4270        let kept = super::resolve(vec![
4271            region("table", 0.9, 26.0, 55.0, 183.0, 558.0),
4272            region("table", 0.9, 196.0, 56.0, 354.0, 561.0),
4273        ]);
4274        assert_eq!(kept.len(), 2);
4275    }
4276
4277    /// A dense data table detected as both picture (0.80) and table (0.62) on one box.
4278    /// The picture is ≥ 0.1 more confident, so `_handle_cross_type_overlaps`
4279    /// keeps both, and the dense table text passed the text-panel gates: the
4280    /// demoted paragraph repeated every cell the table grid renders. A
4281    /// paragraph > 80 % inside a surviving table is the table's child and is
4282    /// not emitted; a panel with no table under it still demotes.
4283    #[test]
4284    fn text_panel_over_a_table_does_not_repeat_its_cells() {
4285        let lines = |t0: f32| -> Vec<TextCell> {
4286            (0..4)
4287                .map(|i| {
4288                    let t = t0 + 10.0 * i as f32;
4289                    cell("1 2 3 4 5 6 7 8 9 10 11 12", 5.0, t, 95.0, t + 8.0)
4290                })
4291                .collect()
4292        };
4293        let mut cells = lines(0.0);
4294        cells.extend(lines(200.0));
4295        let mut regions = vec![
4296            region("picture", 0.80, 0.0, 0.0, 100.0, 45.0),
4297            region("table", 0.62, 0.0, 0.0, 100.0, 45.0),
4298            region("picture", 0.80, 0.0, 200.0, 100.0, 245.0),
4299        ];
4300        super::recover_text_panels(&mut regions, &cells);
4301        assert_eq!(
4302            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
4303            ["table", "text"]
4304        );
4305        assert_eq!(regions[1].t, 200.0, "the table-free panel still demotes");
4306    }
4307
4308    /// docling's `ListItemMarkerProcessor` + docling-core's Markdown list rules
4309    /// (see `list_item_node`): a bullet glyph or `-` is split off and the
4310    /// serializer's own `-` printed (2305's OTSL list reads `- "C" cell`, not
4311    /// `- - "C" cell`); `N.` is an ordered item; any other marker with a letter
4312    /// or digit stays in the text behind the bullet; no marker → plain bullet.
4313    #[test]
4314    fn list_item_markers_split_like_docling() {
4315        let text_of = |n: &Node| match n {
4316            Node::ListItem {
4317                ordered,
4318                number,
4319                text,
4320                marker,
4321                ..
4322            } => (*ordered, *number, text.clone(), marker.clone()),
4323            other => panic!("{other:?}"),
4324        };
4325        let loc = [0, 0, 100, 10];
4326        assert_eq!(
4327            text_of(&super::list_item_node(
4328                "- \"C\" cell - a new table cell",
4329                loc,
4330                false
4331            )),
4332            (
4333                false,
4334                0,
4335                "\"C\" cell - a new table cell".into(),
4336                Some("-".into())
4337            )
4338        );
4339        assert_eq!(
4340            text_of(&super::list_item_node("• Bullet text", loc, false)),
4341            (false, 0, "Bullet text".into(), Some("•".into()))
4342        );
4343        assert_eq!(
4344            text_of(&super::list_item_node("3. Third step", loc, false)),
4345            (true, 3, "Third step".into(), Some("3.".into()))
4346        );
4347        assert_eq!(
4348            text_of(&super::list_item_node("a) Option", loc, false)),
4349            (false, 0, "a) Option".into(), Some("a)".into()))
4350        );
4351        assert_eq!(
4352            text_of(&super::list_item_node("1.2 Nested outline", loc, false)),
4353            (false, 0, "1.2 Nested outline".into(), Some("1.2".into()))
4354        );
4355        // 2203's `3.a. If all IOU scores…`: a compound marker is not an item
4356        // number — docling prints `- 3.a. If all…`.
4357        assert_eq!(
4358            text_of(&super::list_item_node("3.a. If all IOU scores", loc, false)),
4359            (
4360                false,
4361                0,
4362                "3.a. If all IOU scores".into(),
4363                Some("3.a.".into())
4364            )
4365        );
4366        // A glued symbol-font bullet is stripped, a spaced one is the marker.
4367        assert_eq!(
4368            text_of(&super::list_item_node("•Glued", loc, false)),
4369            (false, 0, "Glued".into(), Some("·".into()))
4370        );
4371        // No whitespace after the glyph → not a marker (docling's `\s` is required).
4372        assert_eq!(
4373            text_of(&super::list_item_node("-5 degrees", loc, false)),
4374            (false, 0, "-5 degrees".into(), Some("·".into()))
4375        );
4376        // The remaining numbered shapes, first-wins like docling's list.
4377        for (input, marker, body) in [
4378            ("1.2.3. Deep", "1.2.3.", "Deep"),
4379            ("9a) Nine-a", "9a)", "Nine-a"),
4380            ("(3.a) Paren", "(3.a)", "Paren"),
4381            ("12) Twelve", "12)", "Twelve"),
4382            ("(4) Four", "(4)", "Four"),
4383            ("[7] Seven", "[7]", "Seven"),
4384            ("iv. Roman", "iv.", "Roman"),
4385            ("IX. Roman", "IX.", "Roman"),
4386            ("b. Letter", "b.", "Letter"),
4387            ("B) Letter", "B)", "Letter"),
4388        ] {
4389            assert_eq!(
4390                super::split_list_marker(input),
4391                Some((marker, body, true)),
4392                "{input}"
4393            );
4394        }
4395        // A `1.2.` whose optional dot would eat the separator backtracks like
4396        // Python's regex; a marker with nothing after the whitespace is none.
4397        assert_eq!(
4398            super::split_list_marker("1.2.\tx"),
4399            Some(("1.2.", "x", true))
4400        );
4401        assert_eq!(super::split_list_marker("1. "), None);
4402        assert_eq!(super::split_list_marker("• "), None);
4403        assert_eq!(
4404            text_of(&super::list_item_node("Plain item", loc, false)),
4405            (false, 0, "Plain item".into(), Some("·".into()))
4406        );
4407    }
4408
4409    #[test]
4410    fn code_language_label_above_code_is_detected() {
4411        // A bare "XML" token directly above a code box is a language label; a real
4412        // heading above the same code is not; a language word with no code below is
4413        // left alone.
4414        let label = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
4415        let code = region("code", 0.7, 77.0, 552.0, 290.0, 640.0);
4416        let heading = region("section_header", 0.9, 76.0, 500.0, 260.0, 512.0);
4417        let cells = vec![
4418            cell("XML", 78.0, 541.0, 94.0, 548.0),       // inside `label`
4419            cell("Overview", 78.0, 501.0, 250.0, 511.0), // inside `heading`
4420        ];
4421        let drop = super::code_language_labels(&[label, code, heading], &cells);
4422        assert_eq!(drop, vec![true, false, false], "only the label is consumed");
4423
4424        // Same label with no code region present → not consumed.
4425        let label2 = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
4426        let only = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
4427        assert_eq!(super::code_language_labels(&[label2], &only), vec![false]);
4428
4429        // A label swallowed into the top of a wider code box (negative gap) is still
4430        // recognized.
4431        let inside_lbl = region("text", 0.9, 76.0, 540.0, 96.0, 549.0);
4432        let wide_code = region("code", 0.7, 63.0, 531.0, 320.0, 654.0);
4433        let cells2 = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
4434        assert_eq!(
4435            super::code_language_labels(&[inside_lbl, wide_code], &cells2),
4436            vec![true, false]
4437        );
4438
4439        assert!(super::is_code_language("XML") && super::is_code_language("c#"));
4440        assert!(!super::is_code_language("Configure") && !super::is_code_language("XML schema"));
4441    }
4442
4443    #[test]
4444    fn code_region_text_keeps_lines_and_indentation() {
4445        // Three source lines; each glyph is 6 units wide (width / chars = 6), so the
4446        // `int X;` line indented to x=22 is (22-10)/6 = 2 spaces in.
4447        let region = Region {
4448            label: "code",
4449            score: 1.0,
4450            l: 0.0,
4451            t: -5.0,
4452            r: 100.0,
4453            b: 40.0,
4454        };
4455        let cells = vec![
4456            cell("struct P {", 10.0, 0.0, 70.0, 10.0),
4457            cell("int X;", 22.0, 12.0, 58.0, 22.0),
4458            cell("}", 10.0, 24.0, 16.0, 34.0),
4459        ];
4460        assert_eq!(code_region_text(&region, &cells), "struct P {\n  int X;\n}");
4461    }
4462
4463    #[test]
4464    fn code_region_text_tightens_punctuation_without_eating_indentation() {
4465        // A fluent `.Foo()` line at x=22 (2 chars in). Per-line tightening must not
4466        // consume the leading indent space by matching " ." across it.
4467        let region = Region {
4468            label: "code",
4469            score: 1.0,
4470            l: 0.0,
4471            t: -5.0,
4472            r: 100.0,
4473            b: 40.0,
4474        };
4475        let cells = vec![
4476            cell("builder", 10.0, 0.0, 52.0, 10.0),
4477            // pdfium spaced the call: ".Foo (x)" tightens to ".Foo(x)", still 2-indented.
4478            cell(".Foo (x)", 22.0, 12.0, 70.0, 22.0),
4479        ];
4480        assert_eq!(code_region_text(&region, &cells), "builder\n  .Foo(x)");
4481    }
4482
4483    #[test]
4484    fn code_region_text_orders_out_of_order_cells_and_ignores_blank_lines() {
4485        let region = Region {
4486            label: "code",
4487            score: 1.0,
4488            l: 0.0,
4489            t: -5.0,
4490            r: 100.0,
4491            b: 60.0,
4492        };
4493        // Fed bottom-up and with a whitespace-only cell; output is top-down, no blank.
4494        let cells = vec![
4495            cell("b();", 10.0, 24.0, 34.0, 34.0),
4496            cell("   ", 10.0, 12.0, 20.0, 22.0),
4497            cell("a();", 10.0, 0.0, 34.0, 10.0),
4498        ];
4499        assert_eq!(code_region_text(&region, &cells), "a();\nb();");
4500        // No code cells → empty, so the caller falls back to the prose text.
4501        assert_eq!(code_region_text(&region, &[]), "");
4502    }
4503
4504    fn para(text: &str) -> Node {
4505        Node::Paragraph { text: text.into() }
4506    }
4507
4508    /// Run a node sequence through [`StreamAssembler`] with the given page splits
4509    /// and assert the flushed result equals one-shot [`merge_continuations`].
4510    fn assert_stream_eq(nodes: &[Node], splits: &[usize]) {
4511        let mut want = nodes.to_vec();
4512        merge_continuations(&mut want);
4513
4514        let mut asm = StreamAssembler::new();
4515        let mut got = Vec::new();
4516        let mut start = 0;
4517        for &end in splits {
4518            got.extend(asm.push(nodes[start..end].to_vec()));
4519            start = end;
4520        }
4521        got.extend(asm.push(nodes[start..].to_vec()));
4522        got.extend(asm.finish());
4523        assert_eq!(got, want, "stream assembly diverged (splits={splits:?})");
4524    }
4525
4526    #[test]
4527    fn stream_assembler_matches_merge_continuations() {
4528        // Open fragment + lowercase continuation split across a page boundary.
4529        let cross = [para("the definition of"), para("lists in scope")];
4530        assert_stream_eq(&cross, &[1]);
4531        assert_stream_eq(&cross, &[]);
4532
4533        // Continuation that wraps around a figure (+ its caption) on the boundary.
4534        let wrap = [
4535            para("the wing type that is"),
4536            Node::Picture {
4537                caption: None,
4538                caption_href: None,
4539                image: None,
4540                classification: None,
4541                caption_parent: Default::default(),
4542            },
4543            para("Fig. 1. a diagram"),
4544            para("the most common kind"),
4545        ];
4546        for splits in [&[][..], &[1][..], &[2][..], &[3][..], &[1, 3][..]] {
4547            assert_stream_eq(&wrap, splits);
4548        }
4549
4550        // A heading between fragments blocks the merge (must still flush correctly).
4551        let blocked = [
4552            para("ends mid word and"),
4553            Node::Heading {
4554                level: 2,
4555                text: "New Section".into(),
4556            },
4557            para("more body here"),
4558        ];
4559        for splits in [&[][..], &[1][..], &[2][..]] {
4560            assert_stream_eq(&blocked, splits);
4561        }
4562
4563        // A chain across three pages: each page is one open lowercase fragment.
4564        let chain = [
4565            para("alpha beta"),
4566            para("gamma delta"),
4567            para("epsilon zeta"),
4568        ];
4569        assert_stream_eq(&chain, &[1, 2]);
4570    }
4571
4572    #[test]
4573    fn clean_text_dehyphenates_and_normalizes_typography() {
4574        // U+0002 line-wrap hyphen + the join space → merged word (like docling).
4575        assert_eq!(clean_text("com\u{2} pact"), "compact");
4576        assert_eq!(clean_text("end-to\u{2} end deep"), "end-toend deep");
4577        // A stray wrap hyphen (no following join) is dropped.
4578        assert_eq!(clean_text("word\u{2}"), "word");
4579        // Typographic punctuation → ASCII: every curly quote becomes `'`
4580        // (docling-parse's sanitizer table), a literal `"` stays.
4581        assert_eq!(
4582            clean_text("Graph\u{2019}s \u{201c}x\u{201d} \"y\""),
4583            "Graph's 'x' \"y\""
4584        );
4585        assert_eq!(clean_text("a\u{2026}"), "a...");
4586        // The docling-parse sanitizer's internal spacing is preserved as
4587        // placed; line breaks/tabs normalize to a space, ends trim.
4588        assert_eq!(clean_text("a   b\nc"), "a   b c");
4589    }
4590
4591    /// docling#4064: a form's children are emitted together where the form
4592    /// sits in the top-level order, not interleaved with surrounding text.
4593    #[test]
4594    fn form_children_stay_together_in_reading_order() {
4595        let reg = |label: &'static str, l: f32, t: f32, r: f32, b: f32| Region {
4596            label,
4597            score: 0.9,
4598            l,
4599            t,
4600            r,
4601            b,
4602        };
4603        // Page: intro text, then a form spanning the left column with two
4604        // fields and a table inside, while a right-column paragraph sits
4605        // level with the form's first field (it would otherwise be read
4606        // between the form's children).
4607        let mut items = vec![
4608            reg("text", 50.0, 50.0, 550.0, 70.0),    // 0 intro
4609            reg("form", 50.0, 100.0, 300.0, 400.0),  // 1 container
4610            reg("text", 60.0, 110.0, 290.0, 130.0),  // 2 field A (child)
4611            reg("text", 320.0, 110.0, 550.0, 130.0), // 3 right column paragraph
4612            reg("table", 60.0, 150.0, 290.0, 300.0), // 4 table (child)
4613            reg("text", 60.0, 320.0, 290.0, 340.0),  // 5 field B (child)
4614            reg("text", 50.0, 450.0, 550.0, 470.0),  // 6 outro
4615        ];
4616        let cids = super::cluster_cids(&items, &[]);
4617        super::order_with_containers(&mut items, &cids, 600.0, 800.0, |r| r);
4618        let order: Vec<(&str, f32)> = items.iter().map(|r| (r.label, r.t)).collect();
4619        // The form block (container, then its children top-down) is one unit.
4620        let form_pos = order.iter().position(|(l, _)| *l == "form").unwrap();
4621        assert_eq!(
4622            &order[form_pos..form_pos + 4],
4623            &[
4624                ("form", 100.0),
4625                ("text", 110.0),
4626                ("table", 150.0),
4627                ("text", 320.0)
4628            ]
4629        );
4630        assert_eq!(order[0], ("text", 50.0));
4631        assert_eq!(order[order.len() - 1], ("text", 450.0));
4632        // Without a container the plain order interleaves by geometry.
4633        let mut flat: Vec<Region> = items
4634            .iter()
4635            .filter(|r| r.label != "form")
4636            .cloned()
4637            .collect();
4638        let cids = super::cluster_cids(&flat, &[]);
4639        super::order_regions(&mut flat, &cids, 600.0, 800.0, |r| r);
4640        assert_ne!(
4641            flat.iter().map(|r| r.t).collect::<Vec<_>>(),
4642            order
4643                .iter()
4644                .filter(|(l, _)| *l != "form")
4645                .map(|(_, t)| *t)
4646                .collect::<Vec<_>>()
4647        );
4648    }
4649
4650    /// docling#3906: a picture inside a table lands in the covering cell,
4651    /// chosen by the picture's inferred grid position when cell boxes overlap.
4652    #[test]
4653    fn picture_matches_the_cell_at_its_grid_position() {
4654        let cell = |r: usize, c: usize, bbox: [f32; 4]| docling_core::TableCell {
4655            text: format!("r{r}c{c}"),
4656            bbox: Some(bbox),
4657            start_row: r,
4658            start_col: c,
4659            row_span: 1,
4660            col_span: 1,
4661            column_header: false,
4662            row_header: false,
4663            row_section: false,
4664        };
4665        // 2×2 grid; the (1,0) cell box is generous and also covers the picture.
4666        let cells = vec![
4667            cell(0, 0, [0.0, 0.0, 100.0, 50.0]),
4668            cell(0, 1, [100.0, 0.0, 200.0, 50.0]),
4669            cell(1, 0, [0.0, 50.0, 100.0, 100.0]),
4670            cell(1, 1, [100.0, 50.0, 200.0, 100.0]),
4671        ];
4672        let pic = Region {
4673            label: "picture",
4674            score: 0.9,
4675            l: 110.0,
4676            t: 60.0,
4677            r: 190.0,
4678            b: 95.0,
4679        };
4680        assert_eq!(super::match_picture_to_cell(&pic, &cells), Some((1.0, 3)));
4681        // A picture only half inside any cell is not nested.
4682        let straddling = Region {
4683            label: "picture",
4684            score: 0.9,
4685            l: 60.0,
4686            t: 60.0,
4687            r: 160.0,
4688            b: 95.0,
4689        };
4690        assert_eq!(super::match_picture_to_cell(&straddling, &cells), None);
4691    }
4692
4693    /// docling#4052 (2.122): a line-final dash fuses the wrapped word only
4694    /// when attached to it; a detached dash is a literal and the lines join
4695    /// with a space.
4696    #[test]
4697    fn line_final_hyphen_fuses_only_when_attached_to_a_word() {
4698        let line = |text: &str, t: f32| TextCell {
4699            text: text.to_string(),
4700            l: 0.0,
4701            t,
4702            r: 100.0,
4703            b: t + 10.0,
4704        };
4705        // `algo-` / `rithms`: attached hyphen, alnum on both sides → fused.
4706        assert_eq!(
4707            cells_text(vec![&line("algo-", 0.0), &line("rithms", 12.0)]),
4708            "algorithms"
4709        );
4710        // `pp. 545-` / `561`: attached, digits count as alnum → `545561` (upstream).
4711        assert_eq!(
4712            cells_text(vec![&line("pp. 545-", 0.0), &line("561", 12.0)]),
4713            "pp. 545561"
4714        );
4715        // A dash after whitespace — a separator or a lone `-` cell — is kept and
4716        // the lines take the ordinary joining space.
4717        assert_eq!(
4718            cells_text(vec![&line("range -", 0.0), &line("wide", 12.0)]),
4719            "range - wide"
4720        );
4721        assert_eq!(
4722            cells_text(vec![&line("-", 0.0), &line("item", 12.0)]),
4723            "- item"
4724        );
4725        // Attached but the next line opens with no word (`x-` / `...`): dash
4726        // kept and, as before, no separating space.
4727        assert_eq!(
4728            cells_text(vec![&line("x-", 0.0), &line("...", 12.0)]),
4729            "x-..."
4730        );
4731    }
4732
4733    #[test]
4734    fn lam_alef_only_swaps_a_genuinely_reversed_ligature() {
4735        // A mid-word `alef-variant + lam` is pdfium's reversed lam-alef ligature and
4736        // is swapped back to logical `lam + alef-variant` (`ب أ ل` → `ب ل أ`).
4737        assert_eq!(
4738            clean_text("\u{0628}\u{0623}\u{0644}"),
4739            "\u{0628}\u{0644}\u{0623}"
4740        );
4741        // But when the alef-variant is *already* preceded by a lam it is the logical
4742        // ligature `لآ`; the following lam is the next syllable's letter and must not
4743        // move. `التعلم الآلي` must stay `الآلي`, not become `اللآي`.
4744        assert_eq!(
4745            clean_text("\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"),
4746            "\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"
4747        );
4748    }
4749
4750    /// The #419 page, in points: three layout boxes over one paragraph, two of
4751    /// them ending partway through a line. The sliced lines miss the 0.2 claim
4752    /// and become orphans; the third model box starts above the second orphan,
4753    /// so unfitted the reading order emits that box first and strands the line.
4754    fn sliced_paragraph() -> (Vec<Region>, Vec<TextCell>) {
4755        let line = |text: &str, t: f32, r: f32| cell(text, 60.0, t, r, t + 11.0);
4756        let cells = vec![
4757            line("The mission of this series is to improve", 135.0, 458.0),
4758            line("The books in this series are technical,", 147.0, 458.0),
4759            line("substantial. The authors are", 159.0, 458.0),
4760            line("highly experienced craftsmen and", 171.5, 458.0), // sliced: 1.5/11 under box A
4761            line("actually works in practice, as opposed", 185.0, 458.0),
4762            line("about what the author has done, not", 197.0, 458.0),
4763            line("about programming, there will be lots", 210.5, 458.0), // sliced: 1.5/11 under box B
4764            line("will be lots of case studies from real", 223.0, 206.0), // C's line
4765        ];
4766        let regions = vec![
4767            region("text", 0.9, 60.0, 132.0, 458.0, 173.0), // A: three lines + a sliver of the 4th
4768            region("text", 0.9, 60.0, 184.0, 458.0, 212.0), // B: two lines + a sliver of the 7th
4769            region("text", 0.9, 60.0, 216.0, 206.0, 227.0), // C: last line, box opening 5.5pt too early
4770        ];
4771        (regions, cells)
4772    }
4773
4774    fn ordered_texts(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
4775        let mut items: Vec<Region> = regions.to_vec();
4776        let cids = super::cluster_cids(&items, cells);
4777        super::order_regions(&mut items, &cids, 500.0, 700.0, |r| r);
4778        super::region_texts_exclusive(&items, cells)
4779            .into_iter()
4780            .map(|t| t.chars().take(9).collect())
4781            .collect()
4782    }
4783
4784    /// #419: fitted to its cells, a model box that cut a line in half no longer
4785    /// overlaps the orphan that line became, so the orphan orders where it
4786    /// reads; unfitted, the same page strands the line after the paragraph.
4787    #[test]
4788    fn fitting_boxes_to_cells_puts_a_sliced_line_back_in_order() {
4789        let (mut regions, cells) = sliced_paragraph();
4790        super::add_orphan_regions(&mut regions, &cells);
4791        assert_eq!(regions.len(), 5, "two orphan lines");
4792        // The defect, for the record: C (top 216) is not strictly below the
4793        // orphan at 210.5–221.5, so the graph orders C first.
4794        assert_eq!(
4795            ordered_texts(&regions, &cells).last().map(String::as_str),
4796            Some("about pro")
4797        );
4798
4799        super::fit_regions_to_cells(&mut regions, &cells);
4800        assert_eq!(regions.len(), 5);
4801        // A ends on its last claimed line, C starts on its only one.
4802        assert_eq!((regions[0].t, regions[0].b), (135.0, 170.0));
4803        assert_eq!((regions[2].t, regions[2].b), (223.0, 234.0));
4804        assert_eq!(
4805            ordered_texts(&regions, &cells),
4806            [
4807                "The missi",
4808                "highly ex",
4809                "actually ",
4810                "about pro",
4811                "will be l"
4812            ]
4813        );
4814    }
4815
4816    /// An orphan the fitted paragraph box surrounds (a short middle line the
4817    /// narrow model box missed while claiming the lines around it) is folded
4818    /// into the paragraph; an empty regular box goes away, a formula stays, a
4819    /// picture is never refitted, and a page with no cells is left untouched.
4820    #[test]
4821    fn fitting_folds_surrounded_orphans_and_drops_empty_regulars() {
4822        let wide = |text: &str, t: f32| cell(text, 60.0, t, 400.0, t + 11.0);
4823        let cells = vec![
4824            wide("first line of the paragraph", 100.0),
4825            cell("stray", 250.0, 112.0, 400.0, 123.0), // clear of the narrow box
4826            wide("third line of the paragraph", 124.0),
4827        ];
4828        let mut regions = vec![
4829            // Narrow box: claims the wide lines at 0.41, misses the short one.
4830            region("text", 0.9, 60.0, 98.0, 200.0, 136.0),
4831            region("section_header", 0.8, 60.0, 300.0, 200.0, 320.0), // no cells
4832            region("formula", 0.8, 60.0, 340.0, 200.0, 360.0),        // no cells, kept
4833            region("picture", 0.8, 0.0, 400.0, 500.0, 600.0),
4834        ];
4835        super::add_orphan_regions(&mut regions, &cells);
4836        assert_eq!(regions.len(), 5, "the short line became an orphan");
4837        super::fit_regions_to_cells(&mut regions, &cells);
4838        let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
4839        assert_eq!(labels, ["text", "formula", "picture"]);
4840        let para = &regions[0];
4841        assert_eq!(
4842            (para.l, para.t, para.r, para.b),
4843            (60.0, 100.0, 400.0, 135.0)
4844        );
4845        assert_eq!(
4846            super::region_texts_exclusive(&regions, &cells)[0],
4847            "first line of the paragraph stray third line of the paragraph"
4848        );
4849        assert_eq!(
4850            (regions[2].t, regions[2].b),
4851            (400.0, 600.0),
4852            "picture untouched"
4853        );
4854
4855        let mut untouched = vec![region("text", 0.9, 0.0, 0.0, 10.0, 10.0)];
4856        super::fit_regions_to_cells(&mut untouched, &[]);
4857        assert_eq!(untouched.len(), 1, "no cells yet: nothing dropped");
4858    }
4859}