Skip to main content

docling_pdf/
assemble.rs

1//! Layout-driven assembly: map detected [`Region`]s + text cells to a
2//! [`DoclingDocument`], mirroring docling's page-assembly + reading-order.
3//!
4//! Overlapping detections are resolved greedily by score, each text cell is
5//! assigned to its best-containing region, regions are ordered in reading order
6//! (two-column aware), and each becomes a typed node by its layout label.
7
8use docling_core::{CaptionParent, Node, PictureClass, PictureImage, Table};
9#[cfg(any(feature = "ml", feature = "ocr-prep"))]
10use image::RgbImage;
11
12use crate::layout::Region;
13use crate::pdfium_backend::{PdfPage, TextCell};
14
15fn area(l: f32, t: f32, r: f32, b: f32) -> f32 {
16    ((r - l).max(0.0)) * ((b - t).max(0.0))
17}
18
19/// Intersection area of two boxes.
20fn inter(a: &Region, l: f32, t: f32, r: f32, b: f32) -> f32 {
21    let il = a.l.max(l);
22    let it = a.t.max(t);
23    let ir = a.r.min(r);
24    let ib = a.b.min(b);
25    area(il, it, ir, ib)
26}
27
28/// Wrapper (structured-region) labels, ported from docling
29/// `LayoutPostprocessor.WRAPPER_TYPES`: a region that *contains* other regions
30/// and renders as a structured block (a table / table-of-contents index), not as
31/// its own flat text.
32fn is_wrapper(label: &str) -> bool {
33    matches!(
34        label,
35        "table" | "document_index" | "form" | "key_value_region"
36    )
37}
38
39/// Labels docling's table-structure (TableFormer) model runs on and that render
40/// as a Markdown table: a plain `table` and a `document_index` (a table of
41/// contents), which docling assembles as a `TableItem` too.
42pub fn is_table_like(label: &str) -> bool {
43    matches!(label, "table" | "document_index")
44}
45
46/// Greedily keep regions by descending score, dropping a region that is mostly
47/// covered by an already-kept one (RT-DETR emits overlapping duplicates).
48fn greedy(mut regions: Vec<Region>) -> Vec<Region> {
49    regions.sort_by(|a, b| b.score.total_cmp(&a.score));
50    let mut kept: Vec<Region> = Vec::new();
51    for r in regions {
52        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
53        let covered = kept.iter().any(|k| {
54            let i = inter(&r, k.l, k.t, k.r, k.b);
55            let ka = area(k.l, k.t, k.r, k.b).max(1.0);
56            // drop if most of r is inside k, or they strongly mutually overlap
57            i / ra > 0.7 || i / (ra + ka - i) > 0.5
58        });
59        if !covered {
60            kept.push(r);
61        }
62    }
63    kept
64}
65
66/// Resolve overlapping RT-DETR detections, ported from the bucket structure of
67/// docling's `LayoutPostprocessor`: regular, picture and wrapper clusters live in
68/// **separate** spatial indexes and are de-overlapped independently, so a
69/// high-score picture never suppresses a lower-score table or table-of-contents
70/// index (the redp5110 TOC that was otherwise replaced by a picture box). A
71/// cross-type pass first drops a picture that nearly coincides with a table
72/// (`_handle_cross_type_overlaps`), keeping the structured table.
73/// docling's `_remove_overlapping_clusters("picture")`: same-label picture
74/// detections whose boxes heavily overlap (IoU > 0.8, or either box > 80 %
75/// contained in the other) form one group, and a single survivor is kept per
76/// group. Survivor selection ports `_should_prefer_cluster` /
77/// `_select_best_cluster_from_group` with the picture params
78/// (`area_threshold` 2.0, `conf_threshold` 0.3): a candidate is rejected only
79/// when a rival is both comparable in size (candidate ≤ 2× its area) and
80/// clearly more confident (> 0.3); among the survivors the *larger* box wins
81/// unless it is > 0.3 less confident. Net effect on the corpus: a figure the
82/// detector proposes both whole and as its sub-panels (2206's four-thumbnail
83/// Figure 1) collapses to the whole-figure box, exactly like docling.
84pub(crate) fn dedup_pictures(regions: &mut Vec<Region>) {
85    remove_overlapping_specials(regions, |l| l == "picture", 2.0, 0.3);
86}
87
88/// docling's `_remove_overlapping_clusters` for one special bucket (the
89/// regions whose label satisfies `in_bucket`), with that bucket's
90/// `OVERLAP_PARAMS`: picture (2.0, 0.3) — see [`dedup_pictures`] — or wrapper
91/// (2.0, 0.2) for the table bucket in [`resolve`]. Regions outside the bucket
92/// are untouched; a bucket member overlapping nothing is always kept.
93fn remove_overlapping_specials(
94    regions: &mut Vec<Region>,
95    in_bucket: impl Fn(&str) -> bool,
96    area_threshold: f32,
97    conf_threshold: f32,
98) {
99    let idx: Vec<usize> = (0..regions.len())
100        .filter(|&i| in_bucket(regions[i].label))
101        .collect();
102    if idx.len() < 2 {
103        return;
104    }
105    // Union-find over the bucket.
106    let mut parent: Vec<usize> = (0..idx.len()).collect();
107    fn find(parent: &mut [usize], i: usize) -> usize {
108        let mut root = i;
109        while parent[root] != root {
110            root = parent[root];
111        }
112        let mut cur = i;
113        while parent[cur] != root {
114            let next = parent[cur];
115            parent[cur] = root;
116            cur = next;
117        }
118        root
119    }
120    let boxed = |r: &Region| (r.l, r.t, r.r, r.b);
121    for a in 0..idx.len() {
122        for b in (a + 1)..idx.len() {
123            let (ra, rb) = (&regions[idx[a]], &regions[idx[b]]);
124            let (al, at, ar, ab_) = boxed(ra);
125            let (bl, bt, br, bb) = boxed(rb);
126            let ix = (ar.min(br) - al.max(bl)).max(0.0);
127            let iy = (ab_.min(bb) - at.max(bt)).max(0.0);
128            let inter = ix * iy;
129            let aa = area(al, at, ar, ab_).max(f32::EPSILON);
130            let ba = area(bl, bt, br, bb).max(f32::EPSILON);
131            let iou = inter / (aa + ba - inter).max(f32::EPSILON);
132            if iou > 0.8 || inter / aa > 0.8 || inter / ba > 0.8 {
133                let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
134                if pa != pb {
135                    parent[pa] = pb;
136                }
137            }
138        }
139    }
140    // Per group, run docling's pairwise preference + larger-wins selection.
141    let mut groups: std::collections::HashMap<usize, Vec<usize>> = std::collections::HashMap::new();
142    for i in 0..idx.len() {
143        let root = find(&mut parent, i);
144        groups.entry(root).or_default().push(i);
145    }
146    let mut drop = vec![false; regions.len()];
147    for group in groups.values() {
148        if group.len() < 2 {
149            continue;
150        }
151        let area_of = |i: usize| {
152            let r = &regions[idx[i]];
153            area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
154        };
155        let mut best: Option<usize> = None;
156        for &cand in group {
157            let passes = group.iter().all(|&other| {
158                if other == cand {
159                    return true;
160                }
161                let area_ratio = area_of(cand) / area_of(other);
162                let conf_diff = regions[idx[other]].score - regions[idx[cand]].score;
163                !(area_ratio <= area_threshold && conf_diff > conf_threshold)
164            });
165            if passes {
166                best = Some(match best {
167                    None => cand,
168                    Some(cur) => {
169                        if area_of(cand) > area_of(cur)
170                            && regions[idx[cur]].score - regions[idx[cand]].score <= conf_threshold
171                        {
172                            cand
173                        } else {
174                            cur
175                        }
176                    }
177                });
178            }
179        }
180        // Every candidate rejected can't happen with docling's rule (rejection
181        // needs a strictly better rival); guard with highest score anyway.
182        let keep = best.unwrap_or_else(|| {
183            *group
184                .iter()
185                .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(&regions[idx[b]].score))
186                .expect("non-empty group")
187        });
188        for &i in group {
189            if i != keep {
190                drop[idx[i]] = true;
191            }
192        }
193    }
194    let mut keep_iter = drop.into_iter();
195    regions.retain(|_| !keep_iter.next().expect("aligned"));
196}
197
198/// docling's `_remove_overlapping_clusters("regular")` for what [`greedy`]
199/// leaves standing, run on OCR'd pages before the region-scoped OCR (found
200/// with #471's synthetic scan: every line of the page came out twice).
201///
202/// `greedy` keeps regions by descending score and drops a candidate mostly
203/// inside an already-kept one — so a *lower*-score block that contains
204/// several higher-score line boxes (RT-DETR's favourite reading of a sparse
205/// scanned page: every line, plus the paragraph) survives alongside them.
206/// On a digital page that is harmless: [`fit_regions_to_cells`] hands each
207/// text cell to one owner and drops the regions left empty. On a scanned
208/// page the cells don't exist yet — they come from OCR of *each* region's
209/// crop — so the block and its lines were each recognized, and the block,
210/// which then owned both cell sets, read every line twice.
211///
212/// Upstream never has this problem because its OCR runs over the bitmap
213/// before layout postprocessing and each cell is assigned once; the
214/// postprocessor's regular pass then groups clusters that overlap (IoU > 0.8,
215/// or either > 80 % contained in the other) with a union-find, keeps one
216/// survivor per group via `_should_prefer_cluster` /
217/// `_select_best_cluster_from_group` (`area_threshold` 1.3, `conf_threshold`
218/// 0.05; a LIST_ITEM beats a same-sized TEXT, a CODE box beats what it
219/// contains) and merges the losers' cells into it, whose box is then fitted
220/// to those cells. The equivalent for region-scoped OCR: one survivor per
221/// group, keeping its label and score, with the group's **union** box so the
222/// single crop still covers every merged line. Pictures and wrappers are not
223/// regulars and are left alone ([`dedup_pictures`], [`resolve`]).
224pub(crate) fn merge_overlapping_regulars(regions: &mut Vec<Region>) {
225    let idx: Vec<usize> = (0..regions.len())
226        .filter(|&i| regions[i].label != "picture" && !is_wrapper(regions[i].label))
227        .collect();
228    if idx.len() < 2 {
229        return;
230    }
231    let mut parent: Vec<usize> = (0..idx.len()).collect();
232    fn find(parent: &mut [usize], i: usize) -> usize {
233        let mut root = i;
234        while parent[root] != root {
235            root = parent[root];
236        }
237        let mut cur = i;
238        while parent[cur] != root {
239            let next = parent[cur];
240            parent[cur] = root;
241            cur = next;
242        }
243        root
244    }
245    for a in 0..idx.len() {
246        for b in (a + 1)..idx.len() {
247            let (ra, rb) = (&regions[idx[a]], &regions[idx[b]]);
248            let i = inter(ra, rb.l, rb.t, rb.r, rb.b);
249            let aa = area(ra.l, ra.t, ra.r, ra.b).max(f32::EPSILON);
250            let ba = area(rb.l, rb.t, rb.r, rb.b).max(f32::EPSILON);
251            if i / (aa + ba - i).max(f32::EPSILON) > 0.8 || i / aa > 0.8 || i / ba > 0.8 {
252                let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
253                if pa != pb {
254                    parent[pa] = pb;
255                }
256            }
257        }
258    }
259    let mut groups: std::collections::BTreeMap<usize, Vec<usize>> =
260        std::collections::BTreeMap::new();
261    for i in 0..idx.len() {
262        let root = find(&mut parent, i);
263        groups.entry(root).or_default().push(i);
264    }
265    const AREA_THRESHOLD: f32 = 1.3;
266    const CONF_THRESHOLD: f32 = 0.05;
267    let area_of = |i: usize| {
268        let r = &regions[idx[i]];
269        area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
270    };
271    // `_should_prefer_cluster(candidate, other)` with the regular params.
272    let prefer = |cand: usize, other: usize| -> bool {
273        let (c, o) = (&regions[idx[cand]], &regions[idx[other]]);
274        let area_ratio = area_of(cand) / area_of(other);
275        if c.label == "list_item" && o.label == "text" && (1.0 - area_ratio).abs() < 0.2 {
276            return true;
277        }
278        if c.label == "code" && inter(o, c.l, c.t, c.r, c.b) / area_of(other) > 0.8 {
279            return true;
280        }
281        !(area_ratio <= AREA_THRESHOLD && o.score - c.score > CONF_THRESHOLD)
282    };
283    let mut drop = vec![false; regions.len()];
284    let mut unions: Vec<(usize, (f32, f32, f32, f32))> = Vec::new();
285    for group in groups.values() {
286        if group.len() < 2 {
287            continue;
288        }
289        let mut best: Option<usize> = None;
290        for &cand in group {
291            if group
292                .iter()
293                .all(|&other| other == cand || prefer(cand, other))
294            {
295                best = Some(match best {
296                    None => cand,
297                    Some(cur)
298                        if area_of(cand) > area_of(cur)
299                            && regions[idx[cur]].score - regions[idx[cand]].score
300                                <= CONF_THRESHOLD =>
301                    {
302                        cand
303                    }
304                    Some(cur) => cur,
305                });
306            }
307        }
308        // docling falls back to the group's first cluster; the highest score
309        // is the deterministic equivalent for a set with no insertion order.
310        let keep = best.unwrap_or_else(|| {
311            *group
312                .iter()
313                .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(&regions[idx[b]].score))
314                .expect("non-empty group")
315        });
316        let mut u = (
317            f32::INFINITY,
318            f32::INFINITY,
319            f32::NEG_INFINITY,
320            f32::NEG_INFINITY,
321        );
322        for &i in group {
323            let r = &regions[idx[i]];
324            u = (u.0.min(r.l), u.1.min(r.t), u.2.max(r.r), u.3.max(r.b));
325            if i != keep {
326                drop[idx[i]] = true;
327            }
328        }
329        unions.push((idx[keep], u));
330    }
331    for (i, (l, t, r, b)) in unions {
332        let k = &mut regions[i];
333        (k.l, k.t, k.r, k.b) = (l, t, r, b);
334    }
335    let mut keep_iter = drop.into_iter();
336    regions.retain(|_| !keep_iter.next().expect("aligned"));
337}
338
339/// `intersection_over_union` of two regions.
340fn iou(a: &Region, b: &Region) -> f32 {
341    let i = inter(a, b.l, b.t, b.r, b.b);
342    let u = area(a.l, a.t, a.r, a.b) + area(b.l, b.t, b.r, b.b) - i;
343    if u > 0.0 {
344        i / u
345    } else {
346        0.0
347    }
348}
349
350/// docling's `_resolve_coincident_pairs` (#4059, 2.122): for every (loser,
351/// winner) pair at a near-identical box (IoU > 0.8) whose confidences are
352/// within 0.1 (`loser.score - winner.score < 0.1`), the loser label is dropped
353/// so the label with the richer downstream semantic survives. Nothing else —
354/// containment, area — is considered; a clearly more confident loser stays.
355fn coincident_losers(regions: &[Region], losers: &[usize], winners: &[usize]) -> Vec<usize> {
356    let mut out = Vec::new();
357    for &li in losers {
358        for &wi in winners {
359            if iou(&regions[li], &regions[wi]) > 0.8 && regions[li].score - regions[wi].score < 0.1
360            {
361                out.push(li);
362                break;
363            }
364        }
365    }
366    out
367}
368
369/// docling's `_handle_cross_type_overlaps` (2.122/2.123 shape): the layout
370/// model can emit one grounded region under several labels, and the picture /
371/// table / container buckets are de-overlapped independently, so such a region
372/// survives twice. Elect a winner for the near-identical pairs:
373///
374/// | pair                                  | loser     | winner              |
375/// |---------------------------------------|-----------|---------------------|
376/// | TABLE vs DOCUMENT_INDEX               | table     | document_index      |
377/// | PICTURE vs TABLE / DOCUMENT_INDEX     | picture   | the table-like      |
378/// | FORM / KEY_VALUE_REGION vs *surviving* TABLE / DOCUMENT_INDEX / PICTURE | container | structured element |
379///
380/// IoU (not containment) so a genuine small figure inside a large table region
381/// is not removed; the confidence tolerance keeps a clearly more confident
382/// loser (an earlier port dropped every coincident picture regardless).
383fn handle_cross_type_overlaps(regions: Vec<Region>) -> Vec<Region> {
384    let by = |pred: &dyn Fn(&str) -> bool| -> Vec<usize> {
385        (0..regions.len())
386            .filter(|&i| pred(regions[i].label))
387            .collect()
388    };
389    let tables = by(&|l| l == "table");
390    let doc_indices = by(&|l| l == "document_index");
391    let pictures = by(&|l| l == "picture");
392    let containers = by(&|l| matches!(l, "form" | "key_value_region"));
393    let mut drop = vec![false; regions.len()];
394    for i in coincident_losers(&regions, &tables, &doc_indices) {
395        drop[i] = true;
396    }
397    let table_like: Vec<usize> = tables.iter().chain(&doc_indices).copied().collect();
398    for i in coincident_losers(&regions, &pictures, &table_like) {
399        drop[i] = true;
400    }
401    let structured: Vec<usize> = table_like
402        .iter()
403        .chain(&pictures)
404        .copied()
405        .filter(|&i| !drop[i])
406        .collect();
407    for i in coincident_losers(&regions, &containers, &structured) {
408        drop[i] = true;
409    }
410    let mut drop = drop.into_iter();
411    let mut regions = regions;
412    regions.retain(|_| !drop.next().expect("aligned"));
413    regions
414}
415
416pub fn resolve(regions: Vec<Region>) -> Vec<Region> {
417    let regions = handle_cross_type_overlaps(regions);
418    // De-overlap each bucket on its own.
419    let pictures = greedy(
420        regions
421            .iter()
422            .filter(|r| r.label == "picture")
423            .cloned()
424            .collect(),
425    );
426    // Tables and containers are separate buckets since docling 2.123
427    // (`TABLE_TYPES` vs `CONTAINER_TYPES`, docling#4064): a form drawn around a
428    // table no longer competes with it for survival — the table nests inside
429    // the container instead (`order_with_containers`).
430    let mut tables = greedy(
431        regions
432            .iter()
433            .filter(|r| is_table_like(r.label))
434            .cloned()
435            .collect(),
436    );
437    // `greedy` only drops a table mostly inside a *more* confident one, so a
438    // low-score whole-page table proposed over the column tables it contains
439    // (a two-column glossary page: 0.53 over 0.71/0.67/0.66) survived next to them and every
440    // cell was emitted twice. docling's `_remove_overlapping_clusters(tables,
441    // "wrapper")` groups tables whose boxes overlap (IoU > 0.8, or either one
442    // > 80 % inside the other) and keeps one per group: run it on what
443    // `greedy` leaves, like `merge_overlapping_regulars` does for regulars.
444    remove_overlapping_specials(&mut tables, |_| true, 2.0, 0.2);
445    let containers = greedy(
446        regions
447            .iter()
448            .filter(|r| matches!(r.label, "form" | "key_value_region"))
449            .cloned()
450            .collect(),
451    );
452    let mut kept = greedy(
453        regions
454            .iter()
455            .filter(|r| r.label != "picture" && !is_wrapper(r.label))
456            .cloned()
457            .collect(),
458    );
459    dedup_nested_code(&mut kept);
460    kept.extend(pictures);
461    kept.extend(tables);
462    kept.extend(containers);
463    kept
464}
465
466/// Drop a regular region that is >80% contained in a surviving special region we
467/// render **as a single unit** — a table/table-of-contents index — ported from
468/// docling's "Remove regular clusters that are included in wrappers" step: the
469/// special absorbs it as a child (a table cell), so it must not also be emitted
470/// as its own paragraph/list-item. This stops the survey list-items from
471/// appearing both inside the detected table and again as bullets
472/// (`table_mislabeled_as_picture`).
473///
474/// `picture` regions are **not** in the swallow set: docling keeps a picture's
475/// contained clusters as the `PictureItem`'s *children* in the document JSON
476/// (`_set_cluster_children`, `ReadingOrderModel._add_child_elements`), while
477/// its `MarkdownPictureSerializer` prints only the caption and the image — the
478/// children never reach the Markdown (verified against the corpus groundtruth:
479/// `amt_handbook`'s in-figure callout labels are absent). So the regulars
480/// inside a picture stay in the region list, claim their cells and are fitted
481/// like any regular, and [`assemble_page`] lifts them out of the page's
482/// reading order into a [`Node::PictureChildren`] after their picture (see
483/// [`picture_parents`]). A line only partially under a figure box (straddling
484/// its border, ≤80 % contained) is no picture's child and is emitted, as
485/// docling emits it (#165).
486///
487/// `form` / `key_value_region` wrappers are deliberately **excluded**: this
488/// pipeline does not render them as a structured block (they are skipped), so
489/// their textual content comes precisely from the contained regular regions —
490/// A `page_footer` that is really the body of the last heading on the page
491/// becomes `text`.
492///
493/// The layout model labels the bottom margin by position as much as by
494/// content: a one-line paragraph that happens to sit where a running footer
495/// would — a CV's `Languages` line under its `## Languages` heading, the last
496/// entry of a section that runs to the page edge — comes back as
497/// `page_footer` (0.88 against 0.49 for `text` on the reporting file, with
498/// the fp32 model and either renderer alike), and Markdown drops furniture,
499/// so the document loses real content while the JSON keeps it under the
500/// wrong label. docling fails the same way. Deliberate deviation, gated so a
501/// real footer keeps its label: the nearest `section_header` above the footer
502/// must overlap it horizontally, end within 2.5 footer heights of its top,
503/// and be the **last body element** of its column — no other non-furniture
504/// region starts at or below the heading's bottom edge over the same span —
505/// and the footer must be a line of text, not a page number: at least 40 %
506/// of the page width. A running footer never follows a heading that has no
507/// body of its own, so the combination is the misread heading body.
508pub fn reclaim_heading_body_footers(regions: &mut [Region], page_w: f32) {
509    let n = regions.len();
510    let overlap_x = |a: &Region, b: &Region| a.r.min(b.r) - a.l.max(b.l) > 0.0;
511    for fi in 0..n {
512        let f = regions[fi].clone();
513        if f.label != "page_footer" || f.r - f.l < 0.4 * page_w {
514            continue;
515        }
516        let fh = (f.b - f.t).max(1.0);
517        // The nearest heading above the footer, over the footer's span.
518        let heading = (0..n)
519            .filter(|&j| {
520                let h = &regions[j];
521                j != fi && h.label == "section_header" && h.b <= f.t + 0.5 * fh && overlap_x(h, &f)
522            })
523            .min_by(|&a, &b| regions[b].b.total_cmp(&regions[a].b));
524        let Some(hi) = heading else {
525            continue;
526        };
527        let h = regions[hi].clone();
528        if f.t - h.b > 2.5 * fh {
529            continue;
530        }
531        // The heading must have no body of its own: nothing but the footer
532        // starts at or below its bottom edge over the heading's or footer's
533        // span (a heading whose paragraph follows is not this case, and a
534        // heading with the footer far below it was filtered above).
535        let has_body = (0..n).any(|j| {
536            let r = &regions[j];
537            j != fi
538                && j != hi
539                && !matches!(r.label, "page_footer" | "page_header")
540                && r.t >= h.b - 0.5 * fh
541                && (overlap_x(r, &h) || overlap_x(r, &f))
542        });
543        if has_body {
544            continue;
545        }
546        regions[fi].label = "text";
547    }
548}
549
550/// dropping those would erase the page (e.g. `right_to_left_03`'s form-heavy
551/// pages). Runs *after* [`drop_false_pictures`] so a phantom picture can't
552/// swallow real text on its way out.
553pub fn drop_contained_regulars(regions: &mut Vec<Region>) {
554    let specials: Vec<(f32, f32, f32, f32)> = regions
555        .iter()
556        .filter(|r| is_table_like(r.label))
557        .map(|r| (r.l, r.t, r.r, r.b))
558        .collect();
559    if specials.is_empty() {
560        return;
561    }
562    regions.retain(|r| {
563        if r.label == "picture" || is_wrapper(r.label) {
564            return true;
565        }
566        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
567        !specials
568            .iter()
569            .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.8)
570    });
571}
572
573/// docling's `_set_cluster_children` for pictures: a regular region (one that
574/// claims cells) whose box is > 80 % inside a picture's box
575/// (`intersection_over_self`) is that picture's child — not a page element
576/// (it leaves the reading order and the layout score), but an item under the
577/// `PictureItem` in the JSON. Returns, per region, the index of its parent
578/// picture: the smallest containing one when pictures nest (upstream would
579/// attach it to each).
580pub fn picture_parents(regions: &[Region]) -> Vec<Option<usize>> {
581    regions
582        .iter()
583        .map(|r| {
584            if !claims_cells(r) {
585                return None;
586            }
587            let ra = area(r.l, r.t, r.r, r.b).max(1.0);
588            regions
589                .iter()
590                .enumerate()
591                .filter(|(_, p)| p.label == "picture" && inter(r, p.l, p.t, p.r, p.b) / ra > 0.8)
592                .min_by(|(_, a), (_, b)| {
593                    area(a.l, a.t, a.r, a.b).total_cmp(&area(b.l, b.t, b.r, b.b))
594                })
595                .map(|(i, _)| i)
596        })
597        .collect()
598}
599
600/// True for a bare, single-token source-code language label (`XML`, `C#`, `JSON`,
601/// `bash`, …) — the little header the docs render above a code block. Matched
602/// case-insensitively; anything with whitespace or longer than a token is out.
603fn is_code_language(t: &str) -> bool {
604    let t = t.trim();
605    if t.is_empty() || t.chars().any(char::is_whitespace) || t.chars().count() > 12 {
606        return false;
607    }
608    const LANGS: &[&str] = &[
609        "xml",
610        "html",
611        "xhtml",
612        "json",
613        "jsonc",
614        "yaml",
615        "yml",
616        "toml",
617        "ini",
618        "c#",
619        "csharp",
620        "f#",
621        "fsharp",
622        "vb",
623        "c",
624        "c++",
625        "cpp",
626        "java",
627        "kotlin",
628        "scala",
629        "go",
630        "golang",
631        "rust",
632        "swift",
633        "javascript",
634        "js",
635        "typescript",
636        "ts",
637        "jsx",
638        "tsx",
639        "python",
640        "py",
641        "ruby",
642        "rb",
643        "php",
644        "perl",
645        "lua",
646        "r",
647        "dart",
648        "bash",
649        "sh",
650        "shell",
651        "powershell",
652        "zsh",
653        "batch",
654        "cmd",
655        "sql",
656        "tsql",
657        "plsql",
658        "graphql",
659        "dockerfile",
660        "makefile",
661        "css",
662        "scss",
663        "sass",
664        "less",
665        "markdown",
666        "md",
667        "tex",
668        "latex",
669        "diff",
670        "proto",
671        "razor",
672        "cshtml",
673        "xaml",
674        "aspx",
675        "http",
676    ];
677    let lower = t.to_ascii_lowercase();
678    LANGS.contains(&lower.as_str())
679}
680
681/// Mark the region indices that are a code block's **language label** — a bare
682/// `XML`/`C#`/… token sitting directly above a `code` region — so they are consumed
683/// rather than emitted as their own stray paragraph/heading. The label may also be
684/// captured inside a wider code box (rendered as the fence's first line); dropping
685/// the standalone copy just removes the duplicate.
686fn code_language_labels(regions: &[Region], cells: &[TextCell]) -> Vec<bool> {
687    let mut drop = vec![false; regions.len()];
688    for (i, r) in regions.iter().enumerate() {
689        if matches!(r.label, "code" | "picture" | "table") {
690            continue;
691        }
692        if !is_code_language(&region_text(r, cells)) {
693            continue;
694        }
695        // The label sits just above the code (a blank line's gap) or is swallowed
696        // into the top of a wider code box; either way it is that block's label.
697        // The window is generous because the label's own font is small, so a
698        // one-line gap is several times its height.
699        let line_h = (r.b - r.t).abs().max(1.0);
700        let window = (line_h * 4.0).max(28.0);
701        let labels_code = regions.iter().enumerate().any(|(j, c)| {
702            if j == i || c.label != "code" {
703                return false;
704            }
705            let gap = c.t - r.b; // >0 when the code is below the label
706            let h_overlap = (r.r.min(c.r) - r.l.max(c.l)).max(0.0);
707            gap > -line_h * 3.0 && gap < window && h_overlap > 0.0
708        });
709        if labels_code {
710            drop[i] = true;
711        }
712    }
713    drop
714}
715
716/// Collapse `code` regions where one is nested inside another, keeping the larger.
717///
718/// RT-DETR sometimes emits a tight code box *and* a wider near-duplicate that also
719/// captures the block's language label (`XML`, `C#`, …). When the tight box scores
720/// higher it is kept first, and the wider container — not "mostly inside" the tight
721/// box — survives [`resolve`]'s greedy pass, so the block is emitted twice. Keeping
722/// the **larger** box (rather than dropping it) collapses the pair without leaking
723/// the container's extra cells back out as orphan text, since the larger box still
724/// covers every cell. Restricted to `code` so genuinely distinct nested regions of
725/// other kinds are untouched.
726fn dedup_nested_code(kept: &mut Vec<Region>) {
727    let mut drop = vec![false; kept.len()];
728    for i in 0..kept.len() {
729        if kept[i].label != "code" {
730            continue;
731        }
732        let ai = area(kept[i].l, kept[i].t, kept[i].r, kept[i].b).max(1.0);
733        for j in 0..kept.len() {
734            if i == j || drop[j] || kept[j].label != "code" {
735                continue;
736            }
737            let aj = area(kept[j].l, kept[j].t, kept[j].r, kept[j].b).max(1.0);
738            // Drop i when it is mostly inside a strictly larger code box j.
739            let overlap = inter(&kept[i], kept[j].l, kept[j].t, kept[j].r, kept[j].b);
740            if aj > ai && overlap / ai > 0.7 {
741                drop[i] = true;
742                break;
743            }
744        }
745    }
746    let mut keep = drop.iter();
747    kept.retain(|_| !*keep.next().unwrap());
748}
749
750/// Fraction of the page's non-empty text cells that some detected region
751/// claims (>0.2 intersection-over-self, docling's assignment rule). 1.0 for a
752/// page without text cells.
753///
754/// The int8-layout guard keys off this: a dense digital page whose detections
755/// cover almost none of its text is the signature of quantized confidences
756/// flipping under the 0.5 label thresholds on this CPU's kernels — not of a
757/// genuinely empty layout — and is worth re-running on the fp32 graph.
758pub fn layout_cell_coverage(regions: &[Region], cells: &[TextCell]) -> f32 {
759    let mut total = 0usize;
760    let mut covered = 0usize;
761    for c in cells {
762        if c.text.trim().is_empty() {
763            continue;
764        }
765        total += 1;
766        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
767        if regions
768            .iter()
769            .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
770        {
771            covered += 1;
772        }
773    }
774    if total == 0 {
775        1.0
776    } else {
777        covered as f32 / total as f32
778    }
779}
780
781/// Append `text` regions for cells the layout left uncovered ("orphan cells"),
782/// the way docling's `LayoutPostprocessor` does (`create_orphan_clusters`): any
783/// non-empty cell that no kept region covers (>50% of the cell's area) becomes a
784/// text region of its own, so text the detector missed (a stray `.`, a small
785/// label) is still emitted instead of silently dropped. Adjacent orphan cells on a
786/// line are merged so a missed paragraph doesn't shatter into one block per line.
787pub fn add_orphan_regions(regions: &mut Vec<Region>, cells: &[TextCell]) {
788    // docling assigns each cell to its single best-overlapping cluster at
789    // intersection-over-self > 0.2 and serializes exactly the assigned cells —
790    // and since [`region_texts_exclusive`] now emits under that very rule, the
791    // claim test here matches it: any cell over 0.2 will actually render in
792    // its best region, everything else becomes an orphan. Completeness by
793    // construction, with no (0.2, 0.5] hole (the old > 0.5 serializer needed
794    // the claim test raised to > 0.5 to keep right_to_left_03's `20300` from
795    // vanishing; the exclusive port closes that structurally).
796    //
797    // Only *regular* clusters claim cells: docling's `_find_unassigned_cells`
798    // walks `regular_clusters` alone, so a cell under a `picture` or a wrapper
799    // (`table`/`document_index`/`form`/`key_value_region`) that no regular
800    // cluster covers still becomes an orphan text cluster (#165). The orphans
801    // that end up *fully* inside the special are re-dropped by
802    // [`drop_contained_regulars`] (docling's Markdown drops them the same way
803    // — a picture's children never reach its `MarkdownPictureSerializer`
804    // output, a table's text renders through the reconstructed grid). The
805    // observable fix is the border-straddlers: a line only partially under a
806    // figure box used to lose its cells to the picture's 0.2 claim and vanish
807    // — now it forms an orphan region and is emitted, as docling does.
808    let assigned = |c: &TextCell| {
809        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
810        regions
811            .iter()
812            .filter(|r| r.label != "picture" && !is_wrapper(r.label))
813            .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
814    };
815    // Collect orphan cells (non-empty, unassigned), in page order.
816    let mut orphans: Vec<&TextCell> = cells
817        .iter()
818        .filter(|c| !c.text.trim().is_empty() && !assigned(c))
819        .collect();
820    if orphans.is_empty() {
821        return;
822    }
823    orphans.sort_by(|a, b| a.t.total_cmp(&b.t).then(a.l.total_cmp(&b.l)));
824    // Merge cells that sit on the same line and nearly touch into one region, so a
825    // dropped multi-word line stays one block (docling's refinement merges these).
826    let mut merged: Vec<Region> = Vec::new();
827    for c in orphans {
828        let h = (c.b - c.t).abs().max(1.0);
829        if let Some(last) = merged.last_mut() {
830            let same_line = (last.t - c.t).abs() < h * 0.5;
831            let touching = c.l <= last.r + h && c.l >= last.l - h;
832            // Both tolerances scale with the cell's own height, so a run set
833            // vertically (the arXiv stamp up the margin, #528 — a cell a few
834            // points wide and hundreds tall) would read as "on the line" of
835            // whatever precedes it and glue a whole column into one region.
836            // Lines of one row differ by a drop cap's few multiples at most.
837            let lh = (last.b - last.t).abs().max(1.0);
838            let comparable = h <= 4.0 * lh && lh <= 4.0 * h;
839            if same_line && touching && comparable {
840                last.l = last.l.min(c.l);
841                last.r = last.r.max(c.r);
842                last.t = last.t.min(c.t);
843                last.b = last.b.max(c.b);
844                continue;
845            }
846        }
847        merged.push(Region {
848            label: "text",
849            score: 0.0,
850            l: c.l,
851            t: c.t,
852            r: c.r,
853            b: c.b,
854        });
855    }
856    regions.extend(merged);
857}
858
859/// Demote a `picture` region that is really a **text panel** — a paragraph block
860/// the layout model boxed as a figure because it is typeset on a colored
861/// background (terms-and-conditions callouts, quote boxes) — into ordinary
862/// `text` regions, one per paragraph, so its words are read instead of shipped
863/// as pixels. docling loses this text the same way (cells assigned to a picture
864/// cluster are never serialized); this is a deliberate improvement, not parity.
865///
866/// The gate is conservative so a genuine figure keeps its crop: the region must
867/// contain at least three text lines whose median width spans most of the panel
868/// (axis labels and chat bubbles are narrow and varied) and whose cells cover a
869/// substantial fraction of its area (a photo or chart with sparse labels does
870/// not). Paragraph boundaries are re-derived from the line pitch: a vertical gap
871/// clearly larger than the panel's own leading starts a new `text` region, so
872/// the panel doesn't collapse into one giant paragraph.
873///
874/// Works on any cell source — the digital text layer or OCR lines recognized
875/// from the picture crop — so the native and browser paths, with or without
876/// force-OCR, demote identically.
877pub fn recover_text_panels(regions: &mut Vec<Region>, cells: &[TextCell]) {
878    // A *captioned* picture is a genuine figure whatever it contains — the
879    // corpus is full of document screenshots ("Figure 3: …" above a page
880    // image) that are exactly as dense and wide as a text panel. Only an
881    // uncaptioned picture is a demotion candidate.
882    let captioned: Vec<bool> = regions
883        .iter()
884        .map(|r| {
885            r.label == "picture"
886                && regions.iter().any(|c| {
887                    c.label == "caption" && c.r.min(r.r) - c.l.max(r.l) > 0.0 && {
888                        let gap = if c.t >= r.b {
889                            c.t - r.b
890                        } else if r.t >= c.b {
891                            r.t - c.b
892                        } else {
893                            f32::MAX // vertically overlapping: not a caption
894                        };
895                        gap <= 25.0
896                    }
897                })
898        })
899        .collect();
900    let mut out: Vec<Region> = Vec::with_capacity(regions.len());
901    // Synthesized paragraphs and the demoted panels' boxes are kept separate
902    // from `out` until the end: the dedup filter below must not confuse a
903    // paragraph we just built with a pre-existing region inside the panel.
904    let mut demoted_paras: Vec<Region> = Vec::new();
905    let mut demoted_boxes: Vec<(f32, f32, f32, f32)> = Vec::new();
906    for (i, r) in regions.drain(..).enumerate() {
907        if r.label != "picture" || captioned[i] {
908            out.push(r);
909            continue;
910        }
911        let inside: Vec<&TextCell> = cells
912            .iter()
913            .filter(|c| {
914                !c.text.trim().is_empty() && {
915                    let ca = area(c.l, c.t, c.r, c.b).max(1.0);
916                    inter(&r, c.l, c.t, c.r, c.b) / ca > 0.5
917                }
918            })
919            .collect();
920        // Group the contained cells into lines by vertical overlap (the same
921        // rule region_text orders by), tracking each line's union box.
922        let mut lines: Vec<(f32, f32, f32, f32)> = Vec::new(); // (t, b, l, r)
923        for c in &inside {
924            let (ct, cb) = (c.t.min(c.b), c.t.max(c.b));
925            match lines.iter_mut().find(|(lt, lb, _, _)| {
926                let ov = cb.min(*lb) - ct.max(*lt);
927                ov > 0.5 * (cb - ct).min(*lb - *lt).max(1.0)
928            }) {
929                Some((lt, lb, ll, lr)) => {
930                    *lt = lt.min(ct);
931                    *lb = lb.max(cb);
932                    *ll = ll.min(c.l);
933                    *lr = lr.max(c.r);
934                }
935                None => lines.push((ct, cb, c.l, c.r)),
936            }
937        }
938        if lines.len() < 3 {
939            out.push(r);
940            continue;
941        }
942        let panel_w = (r.r - r.l).max(1.0);
943        let coverage = inside.iter().map(|c| area(c.l, c.t, c.r, c.b)).sum::<f32>()
944            / area(r.l, r.t, r.r, r.b).max(1.0);
945        let mut widths: Vec<f32> = lines.iter().map(|(_, _, l, rr)| rr - l).collect();
946        widths.sort_by(f32::total_cmp);
947        // A figure's text is ragged: a title line, small axis/tick labels, and
948        // OCR boxes over the plot area come out at wildly different heights,
949        // whereas a real text panel is set in one face with constant leading.
950        // Require near-uniform line heights (median absolute deviation ≤ 35%
951        // of the median) so an uncaptioned chart keeps its crop even when its
952        // labels are dense enough to pass the coverage gate (#173) — garbled
953        // OCR of its bars is not content.
954        let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
955        heights.sort_by(f32::total_cmp);
956        let h_med = heights[heights.len() / 2].max(1.0);
957        let mut devs: Vec<f32> = heights.iter().map(|h| (h - h_med).abs()).collect();
958        devs.sort_by(f32::total_cmp);
959        let uniform = devs[devs.len() / 2] <= 0.35 * h_med;
960        let text_panel = coverage >= 0.2 && widths[widths.len() / 2] >= 0.45 * panel_w && uniform;
961        if !text_panel {
962            out.push(r);
963            continue;
964        }
965        lines.sort_by(|a, b| a.0.total_cmp(&b.0));
966        let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
967        heights.sort_by(f32::total_cmp);
968        let h = heights[heights.len() / 2].max(1.0);
969        let mut gaps: Vec<f32> = lines
970            .windows(2)
971            .map(|w| (w[1].0 - w[0].1).max(0.0))
972            .collect();
973        gaps.sort_by(f32::total_cmp);
974        let leading = if gaps.is_empty() {
975            0.0
976        } else {
977            gaps[gaps.len() / 2]
978        };
979        let brk = (1.8 * leading).max(0.75 * h);
980        let mut para: Option<(f32, f32, f32, f32)> = None; // (l, t, r, b) union
981        for (t, b, l, rr) in &lines {
982            match &mut para {
983                Some((pl, _, pr, pb)) if *t - *pb <= brk => {
984                    *pl = pl.min(*l);
985                    *pr = pr.max(*rr);
986                    *pb = pb.max(*b);
987                }
988                _ => {
989                    if let Some((pl, pt, pr, pb)) = para.take() {
990                        demoted_paras.push(Region {
991                            label: "text",
992                            score: r.score,
993                            l: pl,
994                            t: pt,
995                            r: pr,
996                            b: pb,
997                        });
998                    }
999                    para = Some((*l, *t, *rr, *b));
1000                }
1001            }
1002        }
1003        if let Some((pl, pt, pr, pb)) = para {
1004            demoted_paras.push(Region {
1005                label: "text",
1006                score: r.score,
1007                l: pl,
1008                t: pt,
1009                r: pr,
1010                b: pb,
1011            });
1012        }
1013        demoted_boxes.push((r.l, r.t, r.r, r.b));
1014    }
1015    // The paragraphs are rebuilt from *all* of the panel's cells, so any
1016    // surviving text region inside a demoted panel (an orphan cluster or a
1017    // layout-detected fragment — pictures no longer swallow them, #165) would
1018    // say the same words twice. Consume those; wrappers and pictures stay.
1019    if !demoted_boxes.is_empty() {
1020        out.retain(|r| {
1021            r.label == "picture" || is_wrapper(r.label) || {
1022                let ra = area(r.l, r.t, r.r, r.b).max(1.0);
1023                !demoted_boxes
1024                    .iter()
1025                    .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.5)
1026            }
1027        });
1028    }
1029    // docling's "Remove regular clusters that are included in wrappers" (a
1030    // regular > 80 % inside a table is absorbed by it) already ran as
1031    // [`drop_contained_regulars`], but before this demotion created new
1032    // regulars. Apply it to them too: a panel that coincides with a table (a
1033    // dense data table detected as picture 0.80 and table 0.62 on one box;
1034    // `_handle_cross_type_overlaps` keeps both once the picture is ≥ 0.1 more
1035    // confident) rebuilds the table's words as a paragraph the grid already
1036    // renders. A panel inside another picture is left as it was.
1037    demoted_paras.retain(|p| {
1038        let pa = area(p.l, p.t, p.r, p.b).max(1.0);
1039        !out.iter()
1040            .any(|s| is_table_like(s.label) && inter(p, s.l, s.t, s.r, s.b) / pa > 0.8)
1041    });
1042    out.extend(demoted_paras);
1043    *regions = out;
1044}
1045
1046/// Drop a `picture` detection covering more than 90 % of the page — docling's
1047/// `LayoutPostprocessor._process_special_clusters` "Filter out full-page
1048/// pictures" (upstream since 2.15), applied to the thresholded detections
1049/// before overlap resolution. A box that big is the page itself, not a figure
1050/// on it: the layout model emits one for a whole-page diagram (a LaTeX figure
1051/// PDF cropped to its drawing), a plate, or a scan, and keeping it would swallow
1052/// every text cell on the page as picture children — the diagram's labels and
1053/// caption vanish behind a lone `<!-- image -->`, where docling reads them out as
1054/// text. `page_w`/`page_h` is the display-frame page box.
1055pub fn drop_full_page_pictures(regions: &mut Vec<Region>, page_w: f32, page_h: f32) {
1056    let page_area = (page_w * page_h).max(1.0);
1057    regions.retain(|r| r.label != "picture" || area(r.l, r.t, r.r, r.b) / page_area <= 0.90);
1058}
1059
1060/// Drop a `picture` detection that is a small, empty, low-confidence margin box on
1061/// a **text page** — a false positive the RT-DETR layout sometimes emits (e.g.
1062/// `right_to_left_02`'s phantom right-column picture, score 0.40); docling does not
1063/// emit it. The gate is deliberately narrow so a genuine figure is never dropped:
1064/// (1) only on pages with a digital text layer — image/scanned/figure pages have
1065/// no `cells` yet at this point (OCR runs later), so their pictures, which *are*
1066/// the content, are kept; (2) only a box covering < 25 % of the page (a margin
1067/// artifact, not a dominant figure); (3) only when it contains no text and scores
1068/// below 0.5 (real empty figures in the corpus all score ≥ 0.86).
1069pub fn drop_false_pictures(
1070    regions: &mut Vec<Region>,
1071    cells: &[TextCell],
1072    page_w: f32,
1073    page_h: f32,
1074) {
1075    if cells.iter().all(|c| c.text.trim().is_empty()) {
1076        return; // no digital text layer (image/scanned page) — keep all pictures
1077    }
1078    // A text-document page carries several text-bearing non-picture regions (so a
1079    // spurious margin picture is clearly extra). A slide / figure page has at most
1080    // one — there the picture is the content, so never drop it.
1081    let content_regions = regions
1082        .iter()
1083        .filter(|r| r.label != "picture" && !region_text(r, cells).trim().is_empty())
1084        .count();
1085    if content_regions < 2 {
1086        return;
1087    }
1088    let page_area = (page_w * page_h).max(1.0);
1089    regions.retain(|r| {
1090        if r.label != "picture" || r.score >= 0.5 {
1091            return true;
1092        }
1093        if area(r.l, r.t, r.r, r.b) / page_area >= 0.25 {
1094            return true; // a dominant figure, not a margin artifact
1095        }
1096        // Keep it if any text cell falls mostly inside (a real captioned/labelled
1097        // figure); drop only the genuinely empty low-confidence boxes.
1098        cells.iter().any(|c| {
1099            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1100            !c.text.trim().is_empty() && inter(r, c.l, c.t, c.r, c.b) / ca > 0.5
1101        })
1102    });
1103}
1104
1105/// A small digit-only region in the top/bottom margin: a page number. docling
1106/// emits `right_to_left_02`'s bottom `11` as the page's *first* text item (its
1107/// reading-order model floats the page number to the front), whereas our
1108/// position-based ordering would place a bottom region last.
1109fn is_page_number(region: &Region, cells: &[TextCell], page_h: f32) -> bool {
1110    let t = region_text(region, cells);
1111    let t = t.trim();
1112    !t.is_empty()
1113        && t.chars().all(|c| c.is_ascii_digit())
1114        && (region.b - region.t).abs() < 30.0
1115        && (region.t < page_h * 0.12 || region.b > page_h * 0.88)
1116}
1117
1118/// docling's `form` / `key_value_region` *containers* (2.123, docling#4064):
1119/// every region sitting > 0.8 inside one — text, list items, and since #4064
1120/// tables and pictures too — is that container's child. Children are
1121/// reading-ordered among themselves and emitted as one block where the
1122/// container falls in the page's top-level order (a `form_area` /
1123/// `key_value_area` group upstream), instead of interleaving with the text
1124/// around the form. A child inside several containers belongs to the smallest
1125/// (then most confident, then first); a container with children shrinks to
1126/// their union for the top-level ordering, like upstream's bbox adjustment.
1127///
1128/// The containers themselves are still not emitted (`is_skipped`), so the
1129/// Markdown is exactly upstream's — a group prints only its children.
1130///
1131/// `cids` are the items' positions in docling's assembly order
1132/// ([`cluster_cids`]) — the reading-order predictor's same-row rule (#424)
1133/// pairs consecutive ones, within the top level and within each container.
1134fn order_with_containers<T: Clone>(
1135    items: &mut Vec<T>,
1136    cids: &[usize],
1137    page_w: f32,
1138    page_h: f32,
1139    reg: impl Fn(&T) -> &Region,
1140) {
1141    let is_container = |r: &Region| matches!(r.label, "form" | "key_value_region");
1142    let containers: Vec<usize> = (0..items.len())
1143        .filter(|&i| is_container(reg(&items[i])))
1144        .collect();
1145    if containers.is_empty() {
1146        order_regions(items, cids, page_w, page_h, reg);
1147        return;
1148    }
1149    // Parent container per item (containers never nest in each other here —
1150    // upstream assigns regulars and tables/pictures only).
1151    let mut parent: Vec<Option<usize>> = vec![None; items.len()];
1152    for i in 0..items.len() {
1153        let r = reg(&items[i]);
1154        if is_container(r) {
1155            continue;
1156        }
1157        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
1158        let mut best: Option<(usize, f32, f32)> = None; // (idx, area, -score)
1159        for &c in &containers {
1160            let cr = reg(&items[c]);
1161            if inter(r, cr.l, cr.t, cr.r, cr.b) / ra > 0.8 {
1162                let key = (area(cr.l, cr.t, cr.r, cr.b), -cr.score);
1163                if best.is_none_or(|(_, a, s)| key.0 < a || (key.0 == a && key.1 < s)) {
1164                    best = Some((c, key.0, key.1));
1165                }
1166            }
1167        }
1168        parent[i] = best.map(|(c, _, _)| c);
1169    }
1170    // Top-level pass: non-children plus the containers, the latter shrunk to
1171    // their children's union.
1172    let mut top: Vec<(usize, Region)> = Vec::new();
1173    for i in 0..items.len() {
1174        if parent[i].is_some() {
1175            continue;
1176        }
1177        let mut r = reg(&items[i]).clone();
1178        if is_container(&r) {
1179            let kids: Vec<&Region> = (0..items.len())
1180                .filter(|&k| parent[k] == Some(i))
1181                .map(|k| reg(&items[k]))
1182                .collect();
1183            if !kids.is_empty() {
1184                r.l = kids.iter().map(|k| k.l).fold(f32::INFINITY, f32::min);
1185                r.t = kids.iter().map(|k| k.t).fold(f32::INFINITY, f32::min);
1186                r.r = kids.iter().map(|k| k.r).fold(f32::NEG_INFINITY, f32::max);
1187                r.b = kids.iter().map(|k| k.b).fold(f32::NEG_INFINITY, f32::max);
1188            }
1189        }
1190        top.push((i, r));
1191    }
1192    let top_cids: Vec<usize> = top.iter().map(|(i, _)| cids[*i]).collect();
1193    order_regions(&mut top, &top_cids, page_w, page_h, |it| &it.1);
1194    let mut out: Vec<T> = Vec::with_capacity(items.len());
1195    for (i, _) in top {
1196        if is_container(reg(&items[i])) {
1197            let kid_idx: Vec<usize> = (0..items.len()).filter(|&k| parent[k] == Some(i)).collect();
1198            let mut kids: Vec<T> = kid_idx.iter().map(|&k| items[k].clone()).collect();
1199            let kid_cids: Vec<usize> = kid_idx.iter().map(|&k| cids[k]).collect();
1200            order_regions(&mut kids, &kid_cids, page_w, page_h, &reg);
1201            out.push(items[i].clone());
1202            out.extend(kids);
1203        } else {
1204            out.push(items[i].clone());
1205        }
1206    }
1207    *items = out;
1208}
1209
1210/// Furniture / not-yet-emitted labels.
1211fn is_skipped(label: &str) -> bool {
1212    matches!(
1213        label,
1214        "page_header" | "page_footer" | "form" | "key_value_region"
1215    )
1216}
1217
1218/// Reading-order sort of a page's regions, via the ported rule-based
1219/// [`reading_order`](crate::reading_order) predictor (docling's
1220/// `ReadingOrderPredictor`): an up/down geometry graph with same-row links
1221/// between `cids`-consecutive elements (#424), horizontal dilation and a
1222/// depth-first traversal, with `page_header`/`page_footer` ordered as their own
1223/// groups (first/last) as docling does.
1224fn order_regions<T: Clone>(
1225    items: &mut Vec<T>,
1226    cids: &[usize],
1227    page_w: f32,
1228    page_h: f32,
1229    reg: impl Fn(&T) -> &Region,
1230) {
1231    let boxes: Vec<(f32, f32, f32, f32)> = items
1232        .iter()
1233        .map(|it| {
1234            let r = reg(it);
1235            (r.l, r.t, r.r, r.b)
1236        })
1237        .collect();
1238    let is_header: Vec<bool> = items
1239        .iter()
1240        .map(|it| reg(it).label == "page_header")
1241        .collect();
1242    let is_footer: Vec<bool> = items
1243        .iter()
1244        .map(|it| reg(it).label == "page_footer")
1245        .collect();
1246    let order =
1247        crate::reading_order::order_page(&boxes, cids, &is_header, &is_footer, page_w, page_h);
1248    *items = order.iter().map(|&i| items[i].clone()).collect();
1249}
1250
1251/// docling's assembly order of a page's clusters (`LayoutPostprocessor`'s
1252/// final `_sort_clusters(mode="id")`, #424): each region's rank when sorted by
1253/// its first source cell, then by top edge, then left edge; a region with no
1254/// cells sorts after every one that has some. docling numbers its page
1255/// elements (`cid`) in this order, and the reading-order predictor's same-row
1256/// rule pairs elements with consecutive numbers, so the ranks are what
1257/// [`order_with_containers`] hands the predictor.
1258///
1259/// A regular region's first cell is the smallest index among the cells it
1260/// claims. A table, picture or container has no cells of its own upstream
1261/// either — its cells are its *children's*: the regular clusters > 0.8 inside
1262/// it, and upstream every cell no regular cluster claimed is an orphan cluster
1263/// of its own, so a table's interior text (which no regular cluster claims)
1264/// reaches the table through those orphans. Here that is the cells > 0.8
1265/// inside the region plus the claimed cells of the regular regions > 0.8
1266/// inside it. Without the interior cells every table would sort last, and two
1267/// side-by-side tables would then be consecutive and row-linked — reading the
1268/// right table's caption ahead of the left column's headings (2206 page 8).
1269pub fn cluster_cids(regions: &[Region], cells: &[TextCell]) -> Vec<usize> {
1270    let owned = assign_cells(regions, cells);
1271    let first_cell: Vec<usize> = regions
1272        .iter()
1273        .enumerate()
1274        .map(|(i, r)| {
1275            if claims_cells(r) {
1276                return owned[i].iter().copied().min().unwrap_or(usize::MAX);
1277            }
1278            let interior = cells
1279                .iter()
1280                .enumerate()
1281                .filter(|(_, c)| {
1282                    !c.text.trim().is_empty()
1283                        && inter(r, c.l, c.t, c.r, c.b) / area(c.l, c.t, c.r, c.b).max(1.0) > 0.8
1284                })
1285                .map(|(ci, _)| ci)
1286                .min();
1287            let children = regions
1288                .iter()
1289                .enumerate()
1290                .filter(|(j, child)| {
1291                    *j != i && claims_cells(child) && {
1292                        let ca = area(child.l, child.t, child.r, child.b).max(1.0);
1293                        inter(r, child.l, child.t, child.r, child.b) / ca > 0.8
1294                    }
1295                })
1296                .filter_map(|(j, _)| owned[j].iter().copied().min())
1297                .min();
1298            interior
1299                .into_iter()
1300                .chain(children)
1301                .min()
1302                .unwrap_or(usize::MAX)
1303        })
1304        .collect();
1305    let mut by_source: Vec<usize> = (0..regions.len()).collect();
1306    // Stable, like Python's `sorted`: full ties keep the layout order.
1307    by_source.sort_by(|&a, &b| {
1308        first_cell[a]
1309            .cmp(&first_cell[b])
1310            .then(regions[a].t.total_cmp(&regions[b].t))
1311            .then(regions[a].l.total_cmp(&regions[b].l))
1312    });
1313    let mut cids = vec![0; regions.len()];
1314    for (rank, &i) in by_source.iter().enumerate() {
1315        cids[i] = rank;
1316    }
1317    cids
1318}
1319
1320/// Clean a region's assembled text: undo soft-hyphen line wraps, map curly
1321/// quotes and the ellipsis to ASCII (matching docling), and collapse runs of
1322/// whitespace. pdfium emits the line-wrap hyphen as U+0002 in this corpus
1323/// (U+00AD elsewhere), so `word\u{2} continuation` is one hyphenated word —
1324/// drop the hyphen + the joining space and merge (`com\u{2} pact` → `compact`,
1325/// `end-to\u{2} end` → `end-toend`), exactly as docling does.
1326///
1327/// Token spacing is otherwise left as the geometric join produced it. We do not
1328/// tighten punctuation spacing: docling preserves the PDF's own spaces (it keeps
1329/// `{ ahn }`, `Name 1 .`, `[ 9 ]`), and a geometric gap heuristic diverges from
1330/// it more than a plain single-space join does.
1331/// An ordered-list enumeration marker at the start of a list item: leading ASCII
1332/// digits followed by `.`, e.g. `1. Undo/Redo` → `(1, "Undo/Redo")`. Returns
1333/// `None` when the text doesn't start with `digits.`.
1334/// docling's `ListItemMarkerProcessor` bullet patterns
1335/// (`docling/models/postprocessing/list_marker_processor.py`), one glyph each.
1336const LIST_BULLET_MARKERS: &str = "\u{2022}\u{2023}\u{25E6}\u{2043}\u{204C}\u{204D}\u{2219}\u{25AA}\u{25AB}\u{25CF}\u{25CB}-*+•·‣⁃►▶▸➤➢✓✔✗✘";
1337
1338/// docling's numbered-marker patterns as byte-length scanners over the start
1339/// of the text, in its first-wins order (the compound ones first, as they are
1340/// the more specific). Each returns the marker's candidate lengths, longest
1341/// (greedy) first — the alternatives Python's regex would backtrack through
1342/// before the `\s(.+)` tail. Hand-rolled rather than `regex` because this file
1343/// is part of the `pdf-text` (wasm) build, where the `regex` crate is not.
1344/// `\d` is Unicode-aware in Python, hence `char::is_numeric`; the letters are
1345/// ASCII classes in both.
1346const LIST_NUMBERED_MARKERS: &[fn(&str) -> Vec<usize>] = &[
1347    // `\d+(?:\.\d+)+\.?` — 1.1  1.2.3  1.1.
1348    |s| {
1349        let mut i = digits(s, 0);
1350        if i == 0 {
1351            return Vec::new();
1352        }
1353        let mut groups = 0;
1354        while s[i..].starts_with('.') && digits(s, i + 1) > i + 1 {
1355            i = digits(s, i + 1);
1356            groups += 1;
1357        }
1358        if groups == 0 {
1359            return Vec::new();
1360        }
1361        if s[i..].starts_with('.') {
1362            vec![i + 1, i]
1363        } else {
1364            vec![i]
1365        }
1366    },
1367    // `\d+\.?[a-zA-Z]\.` — 9a. 3.a.
1368    |s| digits_dot_letter(s, 0, '.').into_iter().collect(),
1369    // `\d+\.?[a-zA-Z]\)` — 9a) 3.a)
1370    |s| digits_dot_letter(s, 0, ')').into_iter().collect(),
1371    // `\(\d+\.?[a-zA-Z]\)` — (9a) (3.a)
1372    |s| {
1373        if !s.starts_with('(') {
1374            return Vec::new();
1375        }
1376        digits_dot_letter(s, 1, ')').into_iter().collect()
1377    },
1378    // `\d+\.` — 1. 2. 3.
1379    |s| digits_then(s, 0, '.').into_iter().collect(),
1380    // `\d+\)` — 1) 2) 3)
1381    |s| digits_then(s, 0, ')').into_iter().collect(),
1382    // `\(\d+\)` — (1) (2) (3)
1383    |s| {
1384        if !s.starts_with('(') {
1385            return Vec::new();
1386        }
1387        digits_then(s, 1, ')').into_iter().collect()
1388    },
1389    // `\[\d+\]` — [1] [2] [3]
1390    |s| {
1391        if !s.starts_with('[') {
1392            return Vec::new();
1393        }
1394        digits_then(s, 1, ']').into_iter().collect()
1395    },
1396    // `[ivxlcdm]+\.` — i. ii. iii.
1397    |s| class_run_then(s, "ivxlcdm", '.').into_iter().collect(),
1398    // `[IVXLCDM]+\.` — I. II. III.
1399    |s| class_run_then(s, "IVXLCDM", '.').into_iter().collect(),
1400    // `[a-z]\.` / `[A-Z]\.` / `[a-z]\)` / `[A-Z]\)`
1401    |s| {
1402        letter_then(s, char::is_ascii_lowercase, '.')
1403            .into_iter()
1404            .collect()
1405    },
1406    |s| {
1407        letter_then(s, char::is_ascii_uppercase, '.')
1408            .into_iter()
1409            .collect()
1410    },
1411    |s| {
1412        letter_then(s, char::is_ascii_lowercase, ')')
1413            .into_iter()
1414            .collect()
1415    },
1416    |s| {
1417        letter_then(s, char::is_ascii_uppercase, ')')
1418            .into_iter()
1419            .collect()
1420    },
1421];
1422
1423/// Byte offset just past the run of `\d` characters starting at `from`
1424/// (`from` itself when there is none).
1425fn digits(s: &str, from: usize) -> usize {
1426    s[from..]
1427        .char_indices()
1428        .find(|(_, c)| !c.is_numeric())
1429        .map_or(s.len(), |(i, _)| from + i)
1430}
1431
1432/// `\d+<close>` from `from`: the length through `close`, if it matches.
1433fn digits_then(s: &str, from: usize, close: char) -> Option<usize> {
1434    let end = digits(s, from);
1435    (end > from && s[end..].starts_with(close)).then(|| end + close.len_utf8())
1436}
1437
1438/// `\d+\.?[a-zA-Z]<close>` from `from`.
1439fn digits_dot_letter(s: &str, from: usize, close: char) -> Option<usize> {
1440    let mut i = digits(s, from);
1441    if i == from {
1442        return None;
1443    }
1444    if s[i..].starts_with('.') {
1445        i += 1;
1446    }
1447    let letter = s[i..].chars().next().filter(char::is_ascii_alphabetic)?;
1448    i += letter.len_utf8();
1449    s[i..].starts_with(close).then(|| i + close.len_utf8())
1450}
1451
1452/// `[<class>]+<close>` at the start.
1453fn class_run_then(s: &str, class: &str, close: char) -> Option<usize> {
1454    let end = s
1455        .char_indices()
1456        .find(|(_, c)| !class.contains(*c))
1457        .map_or(s.len(), |(i, _)| i);
1458    (end > 0 && s[end..].starts_with(close)).then(|| end + close.len_utf8())
1459}
1460
1461/// `[<letter class>]<close>` at the start.
1462fn letter_then(s: &str, class: fn(&char) -> bool, close: char) -> Option<usize> {
1463    let letter = s.chars().next().filter(class)?;
1464    let i = letter.len_utf8();
1465    s[i..].starts_with(close).then(|| i + close.len_utf8())
1466}
1467
1468/// docling's `ListItemMarkerProcessor.process_list_item`: the item's original
1469/// text is matched against `^(marker)\s(.+)` (DOTALL) for the bullet patterns,
1470/// then the numbered ones in order; a hit splits it into
1471/// `(marker, text, enumerated)`. `\s` is one Unicode whitespace character and
1472/// `.+` everything after it, which must be non-empty.
1473fn split_list_marker(text: &str) -> Option<(&str, &str, bool)> {
1474    let tail_after = |len: usize| -> Option<&str> {
1475        let ws = text[len..].chars().next()?;
1476        if !ws.is_whitespace() {
1477            return None;
1478        }
1479        let rest = &text[len + ws.len_utf8()..];
1480        (!rest.is_empty()).then_some(rest)
1481    };
1482    let first = text.chars().next()?;
1483    if LIST_BULLET_MARKERS.contains(first) {
1484        if let Some(rest) = tail_after(first.len_utf8()) {
1485            return Some((&text[..first.len_utf8()], rest, false));
1486        }
1487    }
1488    for matcher in LIST_NUMBERED_MARKERS {
1489        for len in matcher(text) {
1490            if let Some(rest) = tail_after(len) {
1491                return Some((&text[..len], rest, true));
1492            }
1493        }
1494    }
1495    None
1496}
1497
1498/// A PDF `list_item` region as docling emits it: `ListItemMarkerProcessor`
1499/// splits the marker off (see [`split_list_marker`]), and docling-core's
1500/// Markdown list serializer (default `orig_list_item_marker_mode = AUTO`,
1501/// `ensure_valid_list_item_marker`) prints it as `N. text` for an `N.` marker
1502/// (`case_already_valid`: the marker verbatim); as `- text` for a bullet glyph
1503/// (no letter or digit in the marker: only the `-` the serializer adds); and
1504/// as `- a) text` / `- 1.2 text` / `- [3] text` for any other marker
1505/// (`case_auto`: the serializer's `-`, then the original marker) — spelled
1506/// here as a bullet item whose text carries the marker, the way the DOCX and
1507/// DOC backends already spell theirs. An item without a recognizable marker is
1508/// a plain bullet. The symbol-font bullets docling-parse filters out of its
1509/// cells (`•◦▪·*` glued to the text) are stripped before the match, as before.
1510fn list_item_node(text: &str, loc: [u16; 4], first_in_list: bool) -> Node {
1511    // docling's match runs on the text as docling-parse hands it over; the
1512    // glued symbol-font bullets it never sees are stripped only when the raw
1513    // text matches no marker (`•Text` → `Text`, but `• Text` → marker `•`).
1514    let stripped = text
1515        .trim_start_matches(['•', '◦', '▪', '·', '*'])
1516        .trim_start();
1517    let split = split_list_marker(text).or_else(|| split_list_marker(stripped));
1518    let bullet = |text: String, marker: &str| Node::ListItem {
1519        ordered: false,
1520        number: 0,
1521        first_in_list,
1522        text: md_escape(&text),
1523        level: 0,
1524        // docling keeps the marker as the DocLang list marker
1525        // (`<ldiv><marker>·</marker></ldiv>`); Markdown prints its own `-`.
1526        marker: Some(marker.to_string()),
1527        location: Some(loc),
1528        dclx: None,
1529        href: None,
1530        layer: None,
1531    };
1532    // docling-core's `case_already_valid` is a *full* `\d+\.` match: `3.a.`
1533    // and `1.2` are `case_auto` markers (`- 3.a. text`), not item numbers.
1534    let is_number_dot = |m: &str| {
1535        m.strip_suffix('.')
1536            .is_some_and(|d| !d.is_empty() && d.chars().all(char::is_numeric))
1537    };
1538    match split {
1539        Some((marker, body, true)) if is_number_dot(marker) => {
1540            let number = parse_ordered_marker(marker).map_or(0, |(n, _)| n);
1541            Node::ListItem {
1542                ordered: true,
1543                number,
1544                first_in_list,
1545                text: md_escape(body),
1546                level: 0,
1547                marker: Some(marker.to_string()),
1548                location: Some(loc),
1549                dclx: None,
1550                href: None,
1551                layer: None,
1552            }
1553        }
1554        // `case_auto`: a marker holding a letter or digit rides in the text.
1555        Some((marker, body, true)) => bullet(format!("{marker} {body}"), marker),
1556        Some((marker, body, false)) => bullet(body.to_string(), marker),
1557        None => bullet(stripped.to_string(), "·"),
1558    }
1559}
1560
1561fn parse_ordered_marker(s: &str) -> Option<(u64, String)> {
1562    let digits: String = s.chars().take_while(|c| c.is_ascii_digit()).collect();
1563    if digits.is_empty() {
1564        return None;
1565    }
1566    let rest = s[digits.len()..].strip_prefix('.')?;
1567    let number = digits.parse().ok()?;
1568    Some((number, rest.trim_start().to_string()))
1569}
1570
1571/// Escape markdown special characters the way docling-core's markdown serializer
1572/// does (`markdown.py` post_process): `_` → `\_`, then HTML-escape `&`, `<`, `>`
1573/// (quote=False, so quotes are left). Applied to prose (headings, list items,
1574/// paragraphs); code blocks, the formula placeholder, and table cells are left raw.
1575fn md_escape(text: &str) -> String {
1576    text.replace('_', "\\_")
1577        .replace('&', "&amp;")
1578        .replace('<', "&lt;")
1579        .replace('>', "&gt;")
1580}
1581
1582fn clean_text(text: &str) -> String {
1583    // Typographic-quote normalization follows docling-parse's sanitizer table
1584    // (`pdf_sanitators/constants.h`): every curly quote — single *and double* —
1585    // becomes the ASCII apostrophe `'`, and `‚` a comma. A `"` in docling's
1586    // output only ever comes from a literal `quotedbl` glyph, never from `“ ”`
1587    // (2206's `'text in the wild"` pairs a curly open with a literal-quote
1588    // close). This replaces an earlier Hangul-only special case that patched
1589    // one symptom of mapping `“ ”` to `"`.
1590    let replaced = text
1591        .replace("\u{2} ", "")
1592        .replace("\u{ad} ", "")
1593        .replace(['\u{2}', '\u{ad}'], "") // any stray wrap hyphens not at a join
1594        .replace(
1595            [
1596                '\u{2018}', '\u{2019}', '\u{201b}', '\u{201c}', '\u{201d}', '\u{201e}', '\u{201f}',
1597            ],
1598            "'",
1599        ) // ‘ ’ ‛ “ ” „ ‟ → '
1600        .replace('\u{201a}', ",") // ‚ → ,
1601        .replace(
1602            [
1603                '\u{2010}', '\u{2011}', '\u{2012}', '\u{2013}', '\u{2014}', '\u{2015}', '\u{2212}',
1604            ],
1605            "-",
1606        ) // hyphen/dash family → -
1607        .replace('\u{2044}', "/") // ⁄ fraction slash → /
1608        .replace('\u{2022}', "\u{b7}") // • → · (docling never emits •; inline CCS-concept separators)
1609        .replace('\u{2026}', "..."); // … → ...
1610                                     // The docling-parse sanitizer already placed the correct spacing (e.g.
1611                                     // justified double spaces); preserve internal runs of spaces, only
1612                                     // normalizing line breaks/tabs and trimming the ends.
1613    let out = replaced.replace(['\n', '\r', '\t'], " ").trim().to_string();
1614    fix_arabic_lam_alef(&out)
1615}
1616
1617/// pdfium decomposes the Arabic lam-alef ligature (لا / لإ / لأ / لآ) into its
1618/// glyph constituents in *visual* order — `alef-variant, lam` — but docling keeps
1619/// logical order, `lam, alef-variant`. Swap a mid-word `alef-variant + lam` back
1620/// to `lam + alef-variant`. "Mid-word" (the previous char is an Arabic letter)
1621/// distinguishes the ligature from the definite article `ال` (word-initial
1622/// `alef + lam`), which must stay. No-op for non-Arabic text.
1623fn fix_arabic_lam_alef(s: &str) -> String {
1624    let is_arabic_letter = |c: char| ('\u{0620}'..='\u{064A}').contains(&c);
1625    let chars: Vec<char> = s.chars().collect();
1626    if !chars.iter().any(|&c| is_arabic_letter(c)) {
1627        return s.to_string(); // no-op for non-Arabic text
1628    }
1629    // Pass 1: swap mid-word `alef-variant + lam` → `lam + alef-variant`. Only the
1630    // hamza/madda alef variants (إ أ آ) are safe: the definite article is always
1631    // plain `ا + ل`, so plain `alef + lam` is ambiguous (a legitimate `فعالة` vs a
1632    // reversed `لا` ligature look identical) — leaving plain alef alone avoids
1633    // corrupting legitimate words.
1634    let mut a: Vec<char> = Vec::with_capacity(chars.len());
1635    let mut i = 0;
1636    while i < chars.len() {
1637        let c = chars[i];
1638        if matches!(c, '\u{0622}' | '\u{0623}' | '\u{0625}')
1639            && chars.get(i + 1) == Some(&'\u{0644}')
1640            && i > 0
1641            && is_arabic_letter(chars[i - 1])
1642            // A preceding lam means this alef-variant is *already* the logical
1643            // `lam + alef` ligature; the following lam is the next syllable's
1644            // letter, not a reversed ligature — swapping it corrupts `لآل` → `للآ`
1645            // (e.g. التعلم الآلي → الآلي, not اللآي).
1646            && chars[i - 1] != '\u{0644}'
1647        {
1648            a.push('\u{0644}');
1649            a.push(c);
1650            i += 2;
1651            continue;
1652        }
1653        a.push(c);
1654        i += 1;
1655    }
1656    // Pass 2: insert a space at Arabic↔Latin boundaries (bidi script switch) that
1657    // pdfium runs together — docling separates the embedded Latin run (`وPython`
1658    // → `و Python`).
1659    let mut out: Vec<char> = Vec::with_capacity(a.len());
1660    for (j, &c) in a.iter().enumerate() {
1661        if j > 0 {
1662            let p = a[j - 1];
1663            if (is_arabic_letter(p) && c.is_ascii_alphabetic())
1664                || (p.is_ascii_alphabetic() && is_arabic_letter(c))
1665            {
1666                out.push(' ');
1667            }
1668        }
1669        out.push(c);
1670    }
1671    out.into_iter().collect()
1672}
1673
1674/// docling's `PageAssembleModel._match_hyperlink`: the URI whose link
1675/// annotations cover at least half of the region's box, or `None`. Coverage is
1676/// intersection-over-region-area, **accumulated per URI** — a URL that wraps
1677/// across lines carries several annotation rects that sum toward the same
1678/// target. Ties resolve to the first-seen URI (Python's `max` over dict
1679/// insertion order); the winner still needs `>= 0.5`
1680/// (`_HYPERLINK_COVERAGE_THRESHOLD`).
1681pub(crate) fn region_hyperlink(
1682    region: &Region,
1683    links: &[crate::pdfium_backend::LinkAnnot],
1684) -> Option<String> {
1685    if links.is_empty() {
1686        return None;
1687    }
1688    let area = (region.r - region.l).max(0.0) * (region.b - region.t).max(0.0);
1689    if area <= 0.0 {
1690        return None;
1691    }
1692    let mut coverage: Vec<(&str, f32)> = Vec::new();
1693    for link in links {
1694        let ix = (region.r.min(link.r) - region.l.max(link.l)).max(0.0);
1695        let iy = (region.b.min(link.b) - region.t.max(link.t)).max(0.0);
1696        let c = ix * iy / area;
1697        match coverage.iter_mut().find(|(uri, _)| *uri == link.uri) {
1698            Some((_, acc)) => *acc += c,
1699            None => coverage.push((&link.uri, c)),
1700        }
1701    }
1702    let mut best: Option<(&str, f32)> = None;
1703    for (uri, c) in coverage {
1704        // Strictly greater keeps the first-seen URI on ties, like Python's max.
1705        if best.is_none_or(|(_, bc)| c > bc) {
1706            best = Some((uri, c));
1707        }
1708    }
1709    let (uri, c) = best?;
1710    (c >= 0.5).then(|| normalize_uri(uri))
1711}
1712
1713/// The pydantic-`AnyUrl` normalization docling's hyperlink value passes
1714/// through on its way to the serializer: a URL with an authority but no path
1715/// gains a trailing `/` (`https://arxiv.org` → `https://arxiv.org/`). Other
1716/// AnyUrl canonicalizations (scheme/host lowercasing, percent-encoding) don't
1717/// occur in PDF link annotations in practice, so they are not reproduced.
1718fn normalize_uri(uri: &str) -> String {
1719    if let Some((_, rest)) = uri.split_once("://") {
1720        if !rest.is_empty() && !rest.contains(['/', '?', '#']) {
1721            return format!("{uri}/");
1722        }
1723    }
1724    uri.to_string()
1725}
1726
1727/// Resolve each page hyperlink to the visible text it covers, as `(anchor, uri)`
1728/// in reading order. The anchor is the cells whose centre falls in the link rect,
1729/// joined left-to-right and cleaned the same way prose is (so it matches the
1730/// serialized text), deduped against the immediately-preceding link so pdfium's
1731/// occasional duplicate annotation doesn't double-list. Empty anchors are dropped.
1732pub(crate) fn resolve_link_anchors(page: &PdfPage) -> Vec<(String, String)> {
1733    let mut out: Vec<(String, String)> = Vec::new();
1734    // Use per-word cells, not the line-merged `cells`: a link rect covers a few
1735    // words on a line, and a whole merged line cell would over-capture (its centre
1736    // lands in one link's rect, grabbing the entire line as that link's anchor).
1737    let words = if page.word_cells.is_empty() {
1738        &page.cells
1739    } else {
1740        &page.word_cells
1741    };
1742    for link in &page.links {
1743        // A cell participates when its centre row is inside the rect and it
1744        // overlaps the rect horizontally. A cell can be *wider* than the rect:
1745        // PDFs often draw a whole header line as one text run ("LinkedIn |
1746        // GitHub | Credly"), which docling-parse's word grouping keeps as one
1747        // cell even though each label carries its own link annotation —
1748        // centre-in-rect alone would hand the entire line to every link.
1749        // [`cell_text_in_rect`] clips such a cell to the tokens under the rect.
1750        let mut inside: Vec<(&TextCell, String)> = words
1751            .iter()
1752            .filter(|c| {
1753                let cy = (c.t + c.b) / 2.0;
1754                cy >= link.t && cy <= link.b && c.r.min(link.r) > c.l.max(link.l)
1755            })
1756            .filter_map(|c| {
1757                let text = cell_text_in_rect(c, link.l, link.r);
1758                (!text.is_empty()).then_some((c, text))
1759            })
1760            .collect();
1761        // Reading order: top band then left-to-right (link anchors are LTR).
1762        let band = inside
1763            .iter()
1764            .map(|(c, _)| (c.b - c.t).abs())
1765            .fold(0.0f32, f32::max)
1766            .max(1.0);
1767        inside.sort_by_key(|(c, _)| ((c.t / band).round() as i64, (c.l * 10.0) as i64));
1768        let anchor = clean_text(
1769            &inside
1770                .iter()
1771                .map(|(_, t)| t.trim())
1772                .filter(|t| !t.is_empty())
1773                .collect::<Vec<_>>()
1774                .join(" "),
1775        );
1776        if anchor.is_empty() {
1777            continue;
1778        }
1779        if out
1780            .last()
1781            .is_some_and(|(a, u)| a == &anchor && u == &link.uri)
1782        {
1783            continue;
1784        }
1785        out.push((anchor, link.uri.clone()));
1786    }
1787    out
1788}
1789
1790/// The part of a cell's text that lies under a link rect's x-range. A cell
1791/// fully inside the rect (by centre) returns its whole text. A wider cell is
1792/// split into whitespace tokens whose x-spans are estimated proportionally to
1793/// their character positions (kerning makes this approximate, so selection
1794/// snaps to whole tokens, never characters); tokens whose estimated centre
1795/// falls inside the rect are kept. Returns "" when nothing falls inside.
1796fn cell_text_in_rect(c: &TextCell, l: f32, r: f32) -> String {
1797    let cx = (c.l + c.r) / 2.0;
1798    if cx >= l && cx <= r && c.l >= l - (c.r - c.l) * 0.25 && c.r <= r + (c.r - c.l) * 0.25 {
1799        return c.text.trim().to_string();
1800    }
1801    let chars: Vec<char> = c.text.chars().collect();
1802    let n = chars.len();
1803    if n == 0 || c.r <= c.l {
1804        return String::new();
1805    }
1806    let per = (c.r - c.l) / n as f32;
1807    let mut out: Vec<String> = Vec::new();
1808    let mut token = String::new();
1809    let mut start = 0usize;
1810    // A trailing sentinel space flushes the last token.
1811    for (i, &ch) in chars.iter().enumerate().chain(std::iter::once((n, &' '))) {
1812        if ch.is_whitespace() {
1813            if !token.is_empty() {
1814                let mid = c.l + (start as f32 + (i - start) as f32 / 2.0) * per;
1815                if mid >= l && mid <= r {
1816                    out.push(std::mem::take(&mut token));
1817                } else {
1818                    token.clear();
1819                }
1820            }
1821        } else {
1822            if token.is_empty() {
1823                start = i;
1824            }
1825            token.push(ch);
1826        }
1827    }
1828    out.join(" ")
1829}
1830
1831/// Cells assigned to a region (best container), in reading order, joined.
1832fn region_text(region: &Region, cells: &[TextCell]) -> String {
1833    let inside: Vec<&TextCell> = cells
1834        .iter()
1835        .filter(|c| {
1836            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1837            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1838        })
1839        .collect();
1840    cells_text(inside)
1841}
1842
1843/// docling's exclusive cell assignment (`_assign_cells_to_clusters`): every
1844/// non-empty cell goes to the single best-overlapping *regular* region at
1845/// intersection-over-self > 0.2, and each region serializes exactly its
1846/// assigned cells. A cell under two overlapping boxes is emitted once (by the
1847/// better-covering one), and a cell only partially under its region — e.g.
1848/// normal_4pages' big section numeral, ~30 % inside the heading box — still
1849/// joins it (`## 들어가며 1`) instead of leaking as an orphan. Pictures and
1850/// wrappers never claim (docling walks regular clusters only); ties go to the
1851/// first region, like docling's strict `>` best-overlap scan.
1852pub fn region_texts_exclusive(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
1853    let owned = assign_cells(regions, cells);
1854    // Non-claimers (tables/wrappers/pictures) keep the inclusive > 0.5 text:
1855    // docling fills a special cluster's cells from its contained children, and
1856    // downstream table assembly gates on that text being non-empty.
1857    regions
1858        .iter()
1859        .zip(owned)
1860        .map(|(r, cs)| {
1861            if claims_cells(r) {
1862                cells_text(cs.iter().map(|&i| &cells[i]).collect())
1863            } else {
1864                region_text(r, cells)
1865            }
1866        })
1867        .collect()
1868}
1869
1870/// A *regular* region in docling's sense — one that claims cells. Pictures and
1871/// the wrappers (`table`, `document_index`, `form`, `key_value_region`) fill
1872/// their cells from contained children instead.
1873fn claims_cells(r: &Region) -> bool {
1874    r.label != "picture" && !is_wrapper(r.label)
1875}
1876
1877/// docling's `_assign_cells_to_clusters`: each non-empty cell's index goes to
1878/// the single best-overlapping regular region at intersection-over-self > 0.2
1879/// (ties to the first region, like docling's strict `>` scan). One entry per
1880/// region, in region order.
1881fn assign_cells(regions: &[Region], cells: &[TextCell]) -> Vec<Vec<usize>> {
1882    let mut owned: Vec<Vec<usize>> = vec![Vec::new(); regions.len()];
1883    for (ci, c) in cells.iter().enumerate() {
1884        if c.text.trim().is_empty() {
1885            continue;
1886        }
1887        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1888        let mut best: Option<(usize, f32)> = None;
1889        for (i, r) in regions.iter().enumerate() {
1890            if !claims_cells(r) {
1891                continue;
1892            }
1893            let ov = inter(r, c.l, c.t, c.r, c.b) / ca;
1894            if ov > 0.2 && best.is_none_or(|(_, b)| ov > b) {
1895                best = Some((i, ov));
1896            }
1897        }
1898        if let Some((i, _)) = best {
1899            owned[i].push(ci);
1900        }
1901    }
1902    owned
1903}
1904
1905/// The ballot-box glyph a checkbox line can open with (#609): `Some(checked)`.
1906/// Empty boxes — `☐`, the white squares `□ ▢ ◻` and the shadowed `❏ ❐ ❑ ❒`
1907/// Word's checkbox bullets use — are unchecked; `☑ ☒ ⊠ ⌧ ▣ 🗹 🗷 🗵` are
1908/// checked. A bare tick or cross (`✓ ✗`) is a list bullet
1909/// ([`LIST_BULLET_MARKERS`]), not a box. Symbol-font boxes that reach the text
1910/// layer as Private Use Area codes or as their ASCII byte (Wingdings `o`,
1911/// `þ`) are not recognized: without the font the code is ambiguous (`U+F06F`
1912/// is Symbol's omicron too).
1913pub(crate) fn checkbox_glyph(c: char) -> Option<bool> {
1914    match c {
1915        '☐' | '□' | '▢' | '◻' | '❏' | '❐' | '❑' | '❒' => Some(false),
1916        '☑' | '☒' | '⊠' | '⌧' | '▣' | '\u{1F5F9}' | '\u{1F5F7}' | '\u{1F5F5}' => {
1917            Some(true)
1918        }
1919        _ => None,
1920    }
1921}
1922
1923/// `text` without its leading ballot-box glyph and the space after it — a
1924/// checkbox item's label is the option text, the box is its state.
1925pub(crate) fn strip_checkbox_glyph(text: &str) -> &str {
1926    let t = text.trim_start();
1927    match t.chars().next() {
1928        Some(c) if checkbox_glyph(c).is_some() => t[c.len_utf8()..].trim_start(),
1929        _ => text,
1930    }
1931}
1932
1933/// Give every checklist line its own checkbox region (#609). The layout
1934/// model often reads a checklist as one `text` (or `list_item`) block — on
1935/// the reporter's ReportLab page all four options are one region, which then
1936/// printed as `First option Second option …`, and docling itself splits it
1937/// into two garbled paragraphs. What marks an item is on the page: a drawn
1938/// square just left of the line ([`crate::checkbox`]), or a ballot-box glyph
1939/// the line opens with ([`checkbox_glyph`]: `☐ Yes`, `☒ Done`). A region
1940/// whose lines carry either splits into a `checkbox_selected` /
1941/// `checkbox_unselected` region per marked line (its box spans the square
1942/// and the line; a following unmarked line — a wrapped label — stays with
1943/// it), so assembly emits [`Node::CheckboxItem`]s exactly as for the model's
1944/// own checkbox labels, the glyph stripped from the label. Lines above the
1945/// first mark stay one region of the original label. The first piece
1946/// replaces the region in place and the rest are appended, so per-region
1947/// slices indexed by the original order (table grids, enrichments — neither
1948/// applies to text) stay aligned.
1949///
1950/// A square counts for a line when its vertical centre lies on the line and
1951/// it sits just left of the line's first cell: no more than 1.5 sides (at
1952/// least 6 pt) of gap, at most 1 pt of overlap. Text set *inside* squares (a
1953/// comb field) never matches. A glyph counts when it opens the line and is
1954/// the line's only box glyph; a line with several (`☐ Yes ☐ No`) stays a text
1955/// piece of its own rather than become one item labelled `Yes ☐ No`. A
1956/// region the model already labelled a checkbox is not split, but its state
1957/// follows its first line's mark when it has one — the glyph or the square
1958/// is what the page says, the model's label a guess from the pixels (it read
1959/// a `☑ Bread` line as unselected). With neither on the page this is a no-op.
1960pub fn split_checkbox_lines(
1961    regions: &mut Vec<Region>,
1962    cells: &[TextCell],
1963    boxes: &[crate::checkbox::CheckBox],
1964) {
1965    let has_glyph = || {
1966        cells
1967            .iter()
1968            .any(|c| c.text.chars().any(|ch| checkbox_glyph(ch).is_some()))
1969    };
1970    if boxes.is_empty() && !has_glyph() {
1971        return;
1972    }
1973    let owned = assign_cells(regions, cells);
1974    let mut used = vec![false; boxes.len()];
1975    let mut appended: Vec<Region> = Vec::new();
1976    for (i, mine) in owned.iter().enumerate() {
1977        let region = regions[i].clone();
1978        let model_checkbox = matches!(region.label, "checkbox_selected" | "checkbox_unselected");
1979        if !(model_checkbox || matches!(region.label, "text" | "list_item")) || mine.is_empty() {
1980            continue;
1981        }
1982        // The region's lines, top down: cells sharing a vertical centre band,
1983        // with the row's text in left-to-right order (for the glyph check).
1984        let mut sorted: Vec<&TextCell> = mine.iter().map(|&c| &cells[c]).collect();
1985        sorted.sort_by(|a, b| a.t.total_cmp(&b.t).then(a.l.total_cmp(&b.l)));
1986        // (l, t, r, b) of each line, and its cells.
1987        type Row<'c> = ((f32, f32, f32, f32), Vec<&'c TextCell>);
1988        let mut rows: Vec<Row> = Vec::new();
1989        for c in sorted {
1990            let mid = (c.t + c.b) / 2.0;
1991            match rows.last_mut() {
1992                Some((row, members)) if mid >= row.1 && mid <= row.3 => {
1993                    *row = (
1994                        row.0.min(c.l),
1995                        row.1.min(c.t),
1996                        row.2.max(c.r),
1997                        row.3.max(c.b),
1998                    );
1999                    members.push(c);
2000                }
2001                _ => rows.push(((c.l, c.t, c.r, c.b), vec![c])),
2002            }
2003        }
2004        // What each row does to the grouping.
2005        enum Mark {
2006            /// Opens a checkbox item (`square`: the drawn one, if any).
2007            Item {
2008                square: Option<usize>,
2009                checked: bool,
2010            },
2011            /// Several box glyphs on one line (`☐ Yes ☐ No`): a text piece of
2012            /// its own, closing the open item.
2013            Plain,
2014            /// Unmarked: continues the open piece (a wrapped label).
2015            Continue,
2016        }
2017        let mut marks: Vec<Mark> = Vec::with_capacity(rows.len());
2018        for ((l, t, _, b), members) in &rows {
2019            let square = boxes
2020                .iter()
2021                .enumerate()
2022                .filter(|&(k, sq)| {
2023                    let side = sq.r - sq.l;
2024                    let mid = (sq.t + sq.b) / 2.0;
2025                    let gap = l - sq.r;
2026                    !used[k]
2027                        && mid >= t - 2.0
2028                        && mid <= b + 2.0
2029                        && gap >= -1.0
2030                        && gap <= (1.5 * side).max(6.0)
2031                })
2032                .min_by(|a, b| (l - a.1.r).total_cmp(&(l - b.1.r)))
2033                .map(|(k, _)| k);
2034            let mark = match square {
2035                Some(k) => {
2036                    used[k] = true;
2037                    Mark::Item {
2038                        square: Some(k),
2039                        checked: boxes[k].checked,
2040                    }
2041                }
2042                None => {
2043                    let mut members = members.clone();
2044                    members.sort_by(|a, b| a.l.total_cmp(&b.l));
2045                    let text: String = members.iter().map(|c| c.text.as_str()).collect();
2046                    let first = text.trim_start().chars().next().and_then(checkbox_glyph);
2047                    let glyphs = text
2048                        .chars()
2049                        .filter(|&c| checkbox_glyph(c).is_some())
2050                        .count();
2051                    match (first, glyphs) {
2052                        (Some(checked), 1) => Mark::Item {
2053                            square: None,
2054                            checked,
2055                        },
2056                        (_, 0) => Mark::Continue,
2057                        _ => Mark::Plain,
2058                    }
2059                }
2060            };
2061            marks.push(mark);
2062        }
2063        // A region the model already labelled a checkbox keeps its extent;
2064        // only its state follows the mark — a glyph or a drawn square is the
2065        // page's own record, where the model reads it off the pixels (on a
2066        // `☑ Bread` line Heron says unselected).
2067        if model_checkbox {
2068            if let Some(Mark::Item { checked, .. }) = marks.first() {
2069                regions[i].label = if *checked {
2070                    "checkbox_selected"
2071                } else {
2072                    "checkbox_unselected"
2073                };
2074            }
2075            continue;
2076        }
2077        if !marks.iter().any(|m| matches!(m, Mark::Item { .. })) {
2078            continue;
2079        }
2080        // Group rows: a marked row opens an item; an unmarked one continues
2081        // the open group (the leading group keeps the region's own label).
2082        let mut pieces: Vec<Region> = Vec::new();
2083        for ((row, _), mark) in rows.iter().zip(&marks) {
2084            match mark {
2085                Mark::Item { square, checked } => {
2086                    let (mut l, mut t, mut r, mut b) = *row;
2087                    if let Some(s) = square.map(|k| &boxes[k]) {
2088                        (l, t, r, b) = (l.min(s.l), t.min(s.t), r.max(s.r), b.max(s.b));
2089                    }
2090                    pieces.push(Region {
2091                        label: if *checked {
2092                            "checkbox_selected"
2093                        } else {
2094                            "checkbox_unselected"
2095                        },
2096                        score: region.score,
2097                        l,
2098                        t,
2099                        r,
2100                        b,
2101                    });
2102                }
2103                Mark::Plain => pieces.push(Region {
2104                    l: row.0,
2105                    t: row.1,
2106                    r: row.2,
2107                    b: row.3,
2108                    ..region.clone()
2109                }),
2110                Mark::Continue => match pieces.last_mut() {
2111                    Some(p) => {
2112                        p.l = p.l.min(row.0);
2113                        p.t = p.t.min(row.1);
2114                        p.r = p.r.max(row.2);
2115                        p.b = p.b.max(row.3);
2116                    }
2117                    None => pieces.push(Region {
2118                        l: row.0,
2119                        t: row.1,
2120                        r: row.2,
2121                        b: row.3,
2122                        ..region.clone()
2123                    }),
2124                },
2125            }
2126        }
2127        let mut pieces = pieces.into_iter();
2128        if let Some(first) = pieces.next() {
2129            regions[i] = first;
2130        }
2131        appended.extend(pieces);
2132    }
2133    regions.extend(appended);
2134}
2135
2136/// docling's regular-cluster refinement after cell assignment
2137/// (`LayoutPostprocessor._process_regular_clusters`, #419), run once the page's
2138/// cells are final and before reading order:
2139///
2140/// 1. every regular region's box becomes the union of the cells it claimed
2141///    (`_adjust_cluster_bboxes` — a regular cluster's bbox *is* its cells'
2142///    bbox; a table's is the union with the model box, and pictures keep
2143///    theirs, so neither is touched here);
2144/// 2. a regular region that claimed no cell is dropped (`keep_empty_clusters`
2145///    is off; a `formula` is kept, as upstream keeps it);
2146/// 3. an orphan text region (`score == 0.0`, from [`add_orphan_regions`]) that
2147///    now sits > 0.8 inside another regular region's fitted box is folded into
2148///    it (`_remove_overlapping_clusters` at containment 0.8, the larger box
2149///    winning the group) — up to three rounds, like upstream's loop.
2150///
2151/// Why it matters: the layout model's box can end partway through a line. That
2152/// line fails the 0.2 claim and becomes an orphan — recoverable — but the
2153/// *model* box still overlaps the orphan's line by a few points, so the
2154/// reading-order graph, which links only strictly-above pairs, gets no edge
2155/// between them and may emit the next paragraph first, stranding the line
2156/// after the paragraph it belongs in (1540 of 6050 text blocks on the #419
2157/// book began mid-sentence). Fitted to its cells, the box ends on a line
2158/// boundary and the orphan slots in between; an orphan the fitted box
2159/// swallows joins the paragraph outright. Cell assignment is untouched: a
2160/// region's fitted box contains every cell it claimed, so
2161/// [`region_texts_exclusive`] hands it the same cells afterwards.
2162///
2163/// A page with no cells yet (a scan before OCR) is left alone: dropping every
2164/// text region for want of cells would be wrong, and the OCR paths call this
2165/// again once the cells exist.
2166pub fn fit_regions_to_cells(regions: &mut Vec<Region>, cells: &[TextCell]) {
2167    if !cells.iter().any(|c| !c.text.trim().is_empty()) {
2168        return;
2169    }
2170    for _ in 0..3 {
2171        let owned = assign_cells(regions, cells);
2172        let mut fitted: Vec<Region> = Vec::with_capacity(regions.len());
2173        for (r, own) in regions.iter().zip(&owned) {
2174            if !claims_cells(r) {
2175                fitted.push(r.clone());
2176                continue;
2177            }
2178            if own.is_empty() {
2179                if r.label == "formula" {
2180                    fitted.push(r.clone());
2181                }
2182                continue;
2183            }
2184            let mut f = r.clone();
2185            f.l = own
2186                .iter()
2187                .map(|&i| cells[i].l)
2188                .fold(f32::INFINITY, f32::min);
2189            f.t = own
2190                .iter()
2191                .map(|&i| cells[i].t)
2192                .fold(f32::INFINITY, f32::min);
2193            f.r = own
2194                .iter()
2195                .map(|&i| cells[i].r)
2196                .fold(f32::NEG_INFINITY, f32::max);
2197            f.b = own
2198                .iter()
2199                .map(|&i| cells[i].b)
2200                .fold(f32::NEG_INFINITY, f32::max);
2201            fitted.push(f);
2202        }
2203        let mut changed = fitted.len() != regions.len();
2204        // Fold orphans into the regular region whose fitted box holds them.
2205        let mut drop = vec![false; fitted.len()];
2206        for i in 0..fitted.len() {
2207            let o = &fitted[i];
2208            if !(o.score == 0.0 && o.label == "text") {
2209                continue;
2210            }
2211            let oa = area(o.l, o.t, o.r, o.b).max(1.0);
2212            let mut best: Option<(usize, f32)> = None;
2213            for (j, r) in fitted.iter().enumerate() {
2214                if j == i || drop[j] || r.score == 0.0 || !claims_cells(r) {
2215                    continue;
2216                }
2217                let ov = inter(r, o.l, o.t, o.r, o.b) / oa;
2218                if ov > 0.8 && best.is_none_or(|(_, b)| ov > b) {
2219                    best = Some((j, ov));
2220                }
2221            }
2222            if let Some((j, _)) = best {
2223                let (l, t, r, b) = (o.l, o.t, o.r, o.b);
2224                let host = &mut fitted[j];
2225                host.l = host.l.min(l);
2226                host.t = host.t.min(t);
2227                host.r = host.r.max(r);
2228                host.b = host.b.max(b);
2229                drop[i] = true;
2230                changed = true;
2231            }
2232        }
2233        let mut drop = drop.into_iter();
2234        fitted.retain(|_| !drop.next().expect("aligned"));
2235        *regions = fitted;
2236        if !changed {
2237            break;
2238        }
2239    }
2240}
2241
2242/// Join a prefiltered cell list into the region's text (docling's
2243/// `sanitize_text` over the sanitizer's cell order).
2244fn cells_text(inside: Vec<&TextCell>) -> String {
2245    // docling orders a cluster's cells by their docling-parse cell index
2246    // alone (`LayoutPostprocessor._sort_cells`: `sorted(cells, key=c.index)`)
2247    // — the sanitizer's output order, which our `cells` slice already is.
2248    // No geometric re-sort: normal_4pages' big section numerals paint
2249    // *after* their heading text, and docling's `## 들어가며 1` (numeral
2250    // last) only falls out of pure index order — a band sort dragged the
2251    // numeral to the front. The overlap-grouped line restore this replaced
2252    // measured strictly worse on the corpus (it fixed nothing the index
2253    // order broke, and broke the numerals).
2254    let joined = {
2255        // docling's `PageAssembleModel.sanitize_text`, ported verbatim over the
2256        // parse-index-ordered lines: append a separating space to a line —
2257        // unless it ends with `-`. A dash-ending line whose last word and the
2258        // next line's first word are both alphanumeric is a wrapped word: the
2259        // dash is dropped and the lines fuse (`platforms-` + `reflects` →
2260        // `platformsreflects`, `pp. 545-` + `561` → `545561`). Any other
2261        // dash-ending line — e.g. the *bare* `-` cell a superscript ORCID or an
2262        // inline `–` bullet splits off (its word list is empty, so the fuse
2263        // test fails) — keeps its dash and still takes no trailing space:
2264        // `[0000` `-` `0002` joins as docling's `[0000 -0002`, and the OTSL
2265        // list's `-` + `"C" cell -` + `a new table cell` collapses to
2266        // `-"C" cell a new table cell`. Our cells still carry the raw dash
2267        // family (docling-parse normalizes to `-` before this; clean_text does
2268        // it after), so the endswith test matches them all.
2269        let texts: Vec<&str> = inside
2270            .iter()
2271            .map(|c| c.text.trim())
2272            // Skip whitespace-only cells (a justified line's trailing space
2273            // glyph): an empty line would double the separator.
2274            .filter(|t| !t.is_empty())
2275            .collect();
2276        let last_word_alnum = |s: &str| {
2277            s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
2278                .rfind(|w| !w.is_empty())
2279                .is_some_and(|w| w.chars().all(char::is_alphanumeric))
2280        };
2281        let first_word_alnum = |s: &str| {
2282            s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
2283                .find(|w| !w.is_empty())
2284                .is_some_and(|w| w.chars().all(char::is_alphanumeric))
2285        };
2286        let mut out = String::new();
2287        for (i, t) in texts.iter().enumerate() {
2288            if i > 0 {
2289                let prev = texts[i - 1];
2290                let dashish = matches!(
2291                    prev.chars().last(),
2292                    Some(
2293                        '-' | '\u{2010}'
2294                            | '\u{2011}'
2295                            | '\u{2012}'
2296                            | '\u{2013}'
2297                            | '\u{2014}'
2298                            | '\u{2015}'
2299                            | '\u{2212}'
2300                    )
2301                );
2302                // docling#4052 (2.122): a dash only splits a word when it is
2303                // *attached* to one — the character before it is alphanumeric.
2304                // A dash that follows whitespace (a separator dash, a bullet
2305                // marker, a wrapped `-prefixed` token, the bare `-` cell an
2306                // ORCID splits off) is a literal character: it is kept and the
2307                // lines join with the ordinary space.
2308                let attached = prev.chars().rev().nth(1).is_some_and(char::is_alphanumeric);
2309                if dashish && attached {
2310                    if last_word_alnum(prev) && first_word_alnum(t) {
2311                        out.pop(); // wrapped word: fuse without the dash
2312                    }
2313                    // an attached dash never takes a separating space
2314                } else {
2315                    out.push(' ');
2316                }
2317            }
2318            out.push_str(t);
2319        }
2320        out
2321    };
2322    clean_text(&joined)
2323}
2324
2325/// Tighten the spaces pdfium leaves around tight punctuation in a code line
2326/// (`console .log` → `console.log`, `add (3 , 5)` → `add(3, 5)`), matching
2327/// docling-parse's source spacing.
2328fn tighten_code_punct(s: &str) -> String {
2329    s.replace(" .", ".")
2330        .replace(" ,", ",")
2331        .replace(" ;", ";")
2332        .replace(" )", ")")
2333        .replace(" (", "(")
2334}
2335
2336/// Assemble a **code** region's text with its line structure preserved.
2337///
2338/// Unlike [`region_text`] — which joins every cell with a single space, the right
2339/// thing for prose reflow — a code block's line breaks and indentation are
2340/// significant. The `code_cells` are already one physical source line each
2341/// (grouped space-glyph-only, so monospace runs keep their spacing), so this:
2342///
2343/// 1. groups the cells into vertical line bands and orders them top→bottom,
2344///    left→right;
2345/// 2. joins the lines with `\n` (rather than spaces), keeping the carriage
2346///    returns; and
2347/// 3. reconstructs each line's leading indentation from its left offset, in units
2348///    of the block's estimated monospace character width, so nesting survives.
2349///
2350/// Typography is normalized per line via [`clean_text`] (smart quotes, dashes,
2351/// ellipsis), which never merges lines. Returns an empty string if the region has
2352/// no code cells (the caller falls back to the prose text).
2353fn code_region_text(region: &Region, cells: &[TextCell]) -> String {
2354    let mut inside: Vec<&TextCell> = cells
2355        .iter()
2356        .filter(|c| {
2357            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2358            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
2359        })
2360        .filter(|c| !c.text.trim().is_empty())
2361        .collect();
2362    if inside.is_empty() {
2363        return String::new();
2364    }
2365
2366    // Quantize the top edge into ~line bands (like `region_text`), then order the
2367    // cells by band (top→bottom) and, within a band, by left edge.
2368    let band = inside
2369        .iter()
2370        .map(|c| (c.b - c.t).abs())
2371        .fold(0.0f32, f32::max)
2372        .max(1.0);
2373    let line_of = |c: &TextCell| (c.t / band).round() as i64;
2374    inside.sort_by_key(|c| (line_of(c), (c.l * 10.0) as i64));
2375
2376    // Estimate one monospace character's width (total ink width / total glyphs) to
2377    // convert a line's left offset into a count of leading spaces. Measured over
2378    // all lines so a single short line can't skew it.
2379    let (mut total_w, mut total_chars) = (0.0f32, 0usize);
2380    for c in &inside {
2381        let n = c.text.trim().chars().count();
2382        if n > 0 {
2383            total_w += (c.r - c.l).max(0.0);
2384            total_chars += n;
2385        }
2386    }
2387    let char_w = if total_chars > 0 {
2388        (total_w / total_chars as f32).max(1.0)
2389    } else {
2390        1.0
2391    };
2392    // The block's own left margin is the zero-indent baseline.
2393    let base_l = inside.iter().map(|c| c.l).fold(f32::INFINITY, f32::min);
2394
2395    let mut lines: Vec<String> = Vec::new();
2396    let mut cur: Option<i64> = None;
2397    for c in &inside {
2398        // Tighten pdfium's spaced punctuation per line (on the trimmed content, so
2399        // the reconstructed leading indentation is never nibbled).
2400        let text = tighten_code_punct(&clean_text(c.text.trim()));
2401        if Some(line_of(c)) == cur {
2402            // A second cell sharing this band (rare — e.g. split columns): keep it
2403            // on the same source line, separated by a space.
2404            if let Some(last) = lines.last_mut() {
2405                last.push(' ');
2406                last.push_str(&text);
2407            }
2408            continue;
2409        }
2410        let indent = ((c.l - base_l) / char_w).round().max(0.0) as usize;
2411        lines.push(format!("{}{}", " ".repeat(indent), text));
2412        cur = Some(line_of(c));
2413    }
2414    lines.join("\n")
2415}
2416
2417/// Reconstruct a table's grid geometrically from the text cells inside its
2418/// region: cluster cells into rows (by vertical centre) and columns (by clustered
2419/// left edges), then place each cell. A model-free stand-in for TableFormer that
2420/// recovers grid-aligned tables from the precise PDF text layer (it does not
2421/// resolve row/column spans).
2422pub fn reconstruct_table(region: &Region, cells: &[TextCell]) -> Vec<Vec<String>> {
2423    let mut inside: Vec<&TextCell> = cells
2424        .iter()
2425        .filter(|c| {
2426            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2427            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
2428        })
2429        .collect();
2430    if inside.is_empty() {
2431        return Vec::new();
2432    }
2433    inside.sort_by(|a, b| a.t.total_cmp(&b.t));
2434
2435    // Rows: consecutive cells whose vertical centre is within ~0.7 line height.
2436    let mut rows: Vec<(f32, Vec<&TextCell>)> = Vec::new();
2437    for c in &inside {
2438        let cyc = (c.t + c.b) / 2.0;
2439        let lh = (c.b - c.t).abs().max(1.0);
2440        if let Some((ryc, row)) = rows.last_mut() {
2441            if (cyc - *ryc).abs() < lh * 0.7 {
2442                row.push(c);
2443                continue;
2444            }
2445        }
2446        rows.push((cyc, vec![c]));
2447    }
2448
2449    // Columns: cluster left edges (merge those within a tolerance).
2450    let tol = {
2451        let mut hs: Vec<f32> = inside.iter().map(|c| (c.b - c.t).abs()).collect();
2452        hs.sort_by(f32::total_cmp);
2453        hs[hs.len() / 2].max(4.0) * 1.5
2454    };
2455    let mut lefts: Vec<f32> = inside.iter().map(|c| c.l).collect();
2456    lefts.sort_by(f32::total_cmp);
2457    let mut col_starts: Vec<f32> = Vec::new();
2458    for l in lefts {
2459        if col_starts.last().is_none_or(|&last| l - last > tol) {
2460            col_starts.push(l);
2461        }
2462    }
2463    let ncols = col_starts.len().max(1);
2464    let col_of = |l: f32| -> usize {
2465        col_starts
2466            .iter()
2467            .rposition(|&s| l + tol * 0.5 >= s)
2468            .unwrap_or(0)
2469            .min(ncols - 1)
2470    };
2471
2472    let mut grid = Vec::with_capacity(rows.len());
2473    for (_, mut row) in rows {
2474        row.sort_by(|a, b| a.l.total_cmp(&b.l));
2475        let mut cols = vec![String::new(); ncols];
2476        for c in row {
2477            let ci = col_of(c.l);
2478            // Strip the wrap-hyphen control char so it never lands in a cell.
2479            let t = c.text.trim().replace(['\u{2}', '\u{ad}'], "");
2480            if cols[ci].is_empty() {
2481                cols[ci] = t;
2482            } else {
2483                cols[ci].push(' ');
2484                cols[ci].push_str(&t);
2485            }
2486        }
2487        grid.push(cols);
2488    }
2489    grid
2490}
2491
2492/// Does the geometric reconstruction of a table look trustworthy enough to use
2493/// as-is, instead of paying for TableFormer?
2494///
2495/// [`reconstruct_table`] derives columns by clustering cell **left edges**. On a
2496/// clean grid that is exact, but when a column's entries are not left-aligned
2497/// (or the OCR boxes wobble) the clustering splits one real column into several,
2498/// and the result is a wide, mostly-empty grid — the "spurious empty columns"
2499/// failure TableFormer exists to fix.
2500///
2501/// Two symptoms separate the two cases, and both are properties of the grid
2502/// alone (no model needed):
2503/// * **density** — a real table is mostly full; a split-up one is mostly holes;
2504/// * **thin columns** — a column carrying at most one entry across several rows
2505///   is almost always a split artefact rather than a real column.
2506///
2507/// Deliberately conservative: it answers `true` only for grids that are plainly
2508/// well-formed, so the expensive path stays the default whenever there is doubt.
2509/// A caller that skips TableFormer on `true` trades no quality for the time.
2510pub fn geometric_table_is_reliable(rows: &[Vec<String>]) -> bool {
2511    let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
2512    // Fewer than two columns is not a grid this heuristic can vouch for: it is
2513    // exactly the shape a collapsed table takes, and TableFormer may recover
2514    // real structure from it.
2515    if rows.len() < 2 || ncols < 2 {
2516        return false;
2517    }
2518    let filled = |c: &String| !c.trim().is_empty();
2519    let total = rows.len() * ncols;
2520    let full = rows.iter().flatten().filter(|c| filled(c)).count();
2521    if (full as f32) < MIN_TABLE_FILL * total as f32 {
2522        return false;
2523    }
2524    // A column used by at most one row, when there are rows enough to tell.
2525    if rows.len() >= 3 {
2526        for ci in 0..ncols {
2527            let used = rows
2528                .iter()
2529                .filter(|r| r.get(ci).is_some_and(filled))
2530                .count();
2531            if used <= 1 {
2532                return false;
2533            }
2534        }
2535    }
2536    true
2537}
2538
2539/// Share of a geometric grid's cells that must carry text for it to be trusted
2540/// without TableFormer. Chosen well above the density a left-edge split
2541/// produces (those land nearer a third) and below what a genuine table with a
2542/// few blank cells reaches.
2543const MIN_TABLE_FILL: f32 = 0.6;
2544
2545/// The union bbox of the text cells assigned to a region (same >50%-overlap
2546/// rule as [`region_text`]), or `None` when no cell lands in it. docling's
2547/// LayoutPostprocessor shrinks a regular cluster's bbox to its cells, and the
2548/// enrichment crops are taken from that cell-tight box — cropping the raw
2549/// detector box instead hands the VLM surrounding chrome (e.g. the `Listing N:`
2550/// caption under a code block) that changes its output.
2551pub fn region_cell_bbox(region: &Region, cells: &[TextCell]) -> Option<[f32; 4]> {
2552    let mut bbox: Option<[f32; 4]> = None;
2553    for c in cells {
2554        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2555        if inter(region, c.l, c.t, c.r, c.b) / ca <= 0.5 {
2556            continue;
2557        }
2558        bbox = Some(match bbox {
2559            None => [c.l, c.t, c.r, c.b],
2560            Some([l, t, r, b]) => [l.min(c.l), t.min(c.t), r.max(c.r), b.max(c.b)],
2561        });
2562    }
2563    bbox
2564}
2565
2566/// One region's enrichment-model result, produced by the pipeline's opt-in
2567/// passes (issue #76) and applied during assembly.
2568#[derive(Debug, Clone)]
2569pub enum Enrichment {
2570    /// DocumentPictureClassifier predictions, descending confidence.
2571    PictureClasses(Vec<PictureClass>),
2572    /// CodeFormulaV2 output for a `code` region: the rewritten source text and
2573    /// the `<_language_>` prefix (when the model emitted one).
2574    Code {
2575        language: Option<String>,
2576        text: String,
2577    },
2578    /// CodeFormulaV2 output for a `formula` region: the decoded LaTeX.
2579    Formula { latex: String },
2580}
2581
2582/// Crop a region (page points, already expanded by the caller if needed) from
2583/// the rendered page image and resize it to `target_scale` pixels per point —
2584/// the enrichment-model equivalent of docling's
2585/// `page.get_image(scale=…, cropbox=…)`, sourced from the existing
2586/// [`crate::pdfium_backend::RENDER_SCALE`] render instead of a fresh pdfium
2587/// pass (the page bitmap is already the exact docling render at scale 2).
2588#[cfg(feature = "ml")]
2589pub fn crop_region_scaled(page: &PdfPage, bbox: [f32; 4], target_scale: f32) -> Option<RgbImage> {
2590    let s = page.scale;
2591    let [l, t, r, b] = bbox;
2592    let (iw, ih) = (page.image.width(), page.image.height());
2593    let x = (l * s).max(0.0) as u32;
2594    let y = (t * s).max(0.0) as u32;
2595    if x >= iw || y >= ih {
2596        return None;
2597    }
2598    let w = (((r - l.max(0.0)) * s) as u32).min(iw - x);
2599    let h = (((b - t.max(0.0)) * s) as u32).min(ih - y);
2600    if w == 0 || h == 0 {
2601        return None;
2602    }
2603    let crop = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
2604    // docling renders the crop at `target_scale` directly; from the scale-2
2605    // page render that is a resize to the same pixel geometry
2606    // (`round(width_points * scale)`, PIL's BICUBIC ≙ CatmullRom).
2607    let tw = ((w as f32 / s) * target_scale).round().max(1.0) as u32;
2608    let th = ((h as f32 / s) * target_scale).round().max(1.0) as u32;
2609    if (tw, th) == (w, h) {
2610        return Some(crop);
2611    }
2612    Some(image::imageops::resize(
2613        &crop,
2614        tw,
2615        th,
2616        image::imageops::FilterType::CatmullRom,
2617    ))
2618}
2619
2620/// Resample `img` (rendered at `from` px/pt, covering `w_pt`×`h_pt` points)
2621/// to `to` px/pt — docling's `round(points * scale)` pixel geometry, PIL's
2622/// BICUBIC ≙ CatmullRom. Unchanged when the geometry already matches.
2623#[cfg(feature = "ocr-prep")]
2624fn rescale(img: RgbImage, w_pt: f32, h_pt: f32, to: f32) -> RgbImage {
2625    let tw = (w_pt * to).round().max(1.0) as u32;
2626    let th = (h_pt * to).round().max(1.0) as u32;
2627    if (tw, th) == img.dimensions() {
2628        return img;
2629    }
2630    image::imageops::resize(&img, tw, th, image::imageops::FilterType::CatmullRom)
2631}
2632
2633/// Encode `img` as a PNG [`PictureImage`] rendered at `scale` px/pt.
2634#[cfg(feature = "ocr-prep")]
2635fn png_image(img: &RgbImage, scale: f32) -> Option<PictureImage> {
2636    let mut buf = std::io::Cursor::new(Vec::new());
2637    img.write_to(&mut buf, image::ImageFormat::Png).ok()?;
2638    Some(PictureImage {
2639        mimetype: "image/png".into(),
2640        width: img.width(),
2641        height: img.height(),
2642        data: buf.into_inner(),
2643        dpi: PictureImage::dpi_for_scale(scale),
2644    })
2645}
2646
2647/// The whole page render as docling's `PageItem.image` (#520): at `scale`
2648/// px/pt (`None` = the render's own), `None` when the page has no bitmap.
2649#[cfg(feature = "ocr-prep")]
2650pub fn page_image(page: &PdfPage, scale: Option<f32>) -> Option<PictureImage> {
2651    if page.image.width() == 0 || page.image.height() == 0 || page.scale <= 0.0 {
2652        return None;
2653    }
2654    let scale = scale.unwrap_or(page.scale);
2655    let img = rescale(page.image.clone(), page.width, page.height, scale);
2656    png_image(&img, scale)
2657}
2658
2659/// Crop a layout region from the rendered page image and encode it as PNG (the
2660/// figure bytes docling stores on a `PictureItem`). Region coordinates are page
2661/// points; the image is rendered at `page.scale` and resampled to `scale`
2662/// px/pt when one is given (docling's `images_scale`, #520). The image's `dpi`
2663/// is 72·scale (#519).
2664#[cfg(feature = "ocr-prep")]
2665fn crop_region(page: &PdfPage, region: &Region, scale: Option<f32>) -> Option<PictureImage> {
2666    let s = page.scale;
2667    let (iw, ih) = (page.image.width(), page.image.height());
2668    let x = (region.l * s).max(0.0) as u32;
2669    let y = (region.t * s).max(0.0) as u32;
2670    if x >= iw || y >= ih {
2671        return None;
2672    }
2673    let w = (((region.r - region.l) * s) as u32).min(iw - x);
2674    let h = (((region.b - region.t) * s) as u32).min(ih - y);
2675    if w == 0 || h == 0 {
2676        return None;
2677    }
2678    let sub = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
2679    match scale {
2680        Some(to) if (to - s).abs() > f32::EPSILON => {
2681            // The crop's own point extent (the pixel box, back in points), so
2682            // the resampled geometry is `round(points * scale)`.
2683            let img = rescale(sub, w as f32 / s, h as f32 / s, to);
2684            png_image(&img, to)
2685        }
2686        _ => png_image(&sub, s),
2687    }
2688}
2689
2690/// For each `picture` region, find the `caption` region closest below it (and
2691/// horizontally overlapping); docling pairs them and emits the caption first.
2692/// Each caption is claimed by at most one picture.
2693fn pair_captions(regions: &[Region]) -> Vec<Option<usize>> {
2694    let mut pairs = vec![None; regions.len()];
2695    let mut taken = vec![false; regions.len()];
2696    for (pi, p) in regions.iter().enumerate() {
2697        if p.label != "picture" {
2698            continue;
2699        }
2700        let mut best: Option<(usize, f32)> = None;
2701        for (ci, c) in regions.iter().enumerate() {
2702            if c.label != "caption" || taken[ci] {
2703                continue;
2704            }
2705            let line_h = (c.b - c.t).abs().max(1.0);
2706            let gap = c.t - p.b; // caption sits below the picture
2707            let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
2708            if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
2709                let dist = gap.abs();
2710                if best.is_none_or(|(_, bd)| dist < bd) {
2711                    best = Some((ci, dist));
2712                }
2713            }
2714        }
2715        if let Some((ci, _)) = best {
2716            pairs[pi] = Some(ci);
2717            taken[ci] = true;
2718        }
2719    }
2720    pairs
2721}
2722
2723/// Pair each `code` region with the `caption` region just **above** it (a
2724/// `Listing N:` label). docling renders the code block first, then its caption,
2725/// so the caption is consumed from its own (earlier) reading-order slot and
2726/// re-emitted after the code.
2727fn pair_code_captions(regions: &[Region]) -> Vec<Option<usize>> {
2728    let mut pairs = vec![None; regions.len()];
2729    let mut taken = vec![false; regions.len()];
2730    for (pi, p) in regions.iter().enumerate() {
2731        if p.label != "code" {
2732            continue;
2733        }
2734        let mut best: Option<(usize, f32)> = None;
2735        for (ci, c) in regions.iter().enumerate() {
2736            if c.label != "caption" || taken[ci] {
2737                continue;
2738            }
2739            let line_h = (c.b - c.t).abs().max(1.0);
2740            let gap = p.t - c.b; // caption sits above the code
2741            let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
2742            if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
2743                let dist = gap.abs();
2744                if best.is_none_or(|(_, bd)| dist < bd) {
2745                    best = Some((ci, dist));
2746                }
2747            }
2748        }
2749        if let Some((ci, _)) = best {
2750            pairs[pi] = Some(ci);
2751            taken[ci] = true;
2752        }
2753    }
2754    pairs
2755}
2756
2757/// Pair each `table`/`document_index` region with its `caption` (#265) the way
2758/// docling's `ReadingOrderPredictor._find_to_captions` does: by **reading-order
2759/// adjacency**, not geometry. A caption claims the media element
2760/// (table/picture/code) immediately next to it in the ordered region sequence,
2761/// and only when exactly one side holds one — a caption sandwiched between two
2762/// media elements stays unattached, and a text paragraph between caption and
2763/// table breaks the bond. This is what lets a flush-left "Table 3: …" label
2764/// bind a centered grid it doesn't horizontally overlap, while a caption in
2765/// the neighbouring column of a two-column page — geometrically close — never
2766/// pairs across the gutter. Runs after the picture and code pairings (the
2767/// picture/code arms of the same upstream matcher), so a caption they claimed
2768/// stays claimed. docling attaches these as `TableItem.captions` refs; the
2769/// paired caption is consumed from its own reading-order slot and rides on the
2770/// table node instead.
2771fn pair_table_captions(regions: &[Region], taken: &mut [bool]) -> Vec<Option<usize>> {
2772    let is_media = |label: &str| is_table_like(label) || matches!(label, "picture" | "code");
2773    let mut pairs: Vec<Option<usize>> = vec![None; regions.len()];
2774    for ci in 0..regions.len() {
2775        if regions[ci].label != "caption" || taken[ci] {
2776            continue;
2777        }
2778        // Furniture (headers/footers, form chrome) is not part of docling's
2779        // body-element sequence, so it neither bonds nor blocks.
2780        let prev = regions[..ci].iter().rposition(|r| !is_skipped(r.label));
2781        let next = regions[ci + 1..]
2782            .iter()
2783            .position(|r| !is_skipped(r.label))
2784            .map(|off| ci + 1 + off);
2785        let prev_media = prev.is_some_and(|j| is_media(regions[j].label));
2786        let next_media = next.is_some_and(|j| is_media(regions[j].label));
2787        let target = match (prev_media, next_media) {
2788            (true, false) => prev,
2789            (false, true) => next,
2790            // Ambiguous (media on both sides) or no media at all: leave the
2791            // caption in its own reading-order slot, as docling does.
2792            _ => None,
2793        };
2794        if let Some(ti) = target {
2795            // A first claim wins (a table with captions above *and* below
2796            // keeps the earlier one — docling's nearest-first tiebreak).
2797            if is_table_like(regions[ti].label) && pairs[ti].is_none() {
2798                pairs[ti] = Some(ci);
2799                taken[ci] = true;
2800            }
2801        }
2802    }
2803    pairs
2804}
2805
2806/// Assemble one page from its (already overlap-resolved) layout regions and
2807/// text cells.
2808/// Normalize a layout region (page points, top-left origin) to DocLang's 0–511
2809/// location grid: `clamp(round(512 · coord / page_dim), 0, 511)`, per axis,
2810/// order `[x0, y0, x1, y1]`. Mirrors docling_core's
2811/// `_doclang_utils._create_location_tokens_for_bbox` (resolution 512) so the
2812/// emitted `<location>` tokens line up with the Python groundtruth. Our heron
2813/// cluster boxes match docling's to within ~1 grid unit; the residual (mainly
2814/// the aspect-ratio-stretch vs letterbox preprocessing difference) is absorbed
2815/// by the conformance harness's geometry tolerance.
2816fn norm_loc(region: &Region, page_w: f32, page_h: f32) -> [u16; 4] {
2817    let q = |v: f32, dim: f32| -> u16 {
2818        if dim <= 0.0 {
2819            return 0;
2820        }
2821        let g = (512.0 * (v as f64) / (dim as f64)).round() as i64;
2822        g.clamp(0, 511) as u16
2823    };
2824    [
2825        q(region.l, page_w),
2826        q(region.t, page_h),
2827        q(region.r, page_w),
2828        q(region.b, page_h),
2829    ]
2830}
2831
2832/// Wrap a node in its layout provenance so the DocLang serializer emits the four
2833/// `<location>` tokens as the element's head (Markdown/JSON render `inner`
2834/// unchanged).
2835fn located(loc: [u16; 4], inner: Node) -> Node {
2836    Node::Located {
2837        location: loc,
2838        inner: Box::new(inner),
2839    }
2840}
2841
2842/// Stamp the real 1-based page number onto a page's leading marker (see
2843/// [`assemble_page`], which emits it with `page_no: 0` because only the
2844/// document-level collector knows the true index — `--pages` windows shift it).
2845pub fn stamp_page_no(nodes: &mut [Node], page_no: usize) {
2846    if let Some(Node::PageInfo { page_no: p, .. }) = nodes.first_mut() {
2847        *p = page_no;
2848    }
2849}
2850
2851/// A dense table grid plus its first-class cells (#240): `rows` is the text
2852/// grid every serializer renders (spans replicate their anchor's text);
2853/// `cells` are the docling-parity per-cell records (text, page-point bbox,
2854/// span rectangle, OTSL header roles). Produced by the TableFormer paths
2855/// (`tf_core`); lives in this always-compiled module so the pure-text (wasm
2856/// `pdf-text`) build sees the type.
2857#[derive(Clone, Debug)]
2858pub struct TableGrid {
2859    pub rows: Vec<Vec<String>>,
2860    pub cells: Vec<docling_core::TableCell>,
2861}
2862
2863/// docling's `_RICH_CELL_PICTURE_COVERAGE_THRESHOLD`.
2864const RICH_CELL_PICTURE_COVERAGE: f32 = 0.8;
2865
2866/// docling `ReadingOrderModel._match_table_pictures` (#3906, 2.118.1): every
2867/// picture ≥ 80 % inside a TableFormer-structured table on the page is matched
2868/// to the cell covering it, and returned per table as `cell index → pictures`.
2869/// A picture that pairs with a caption stays a standalone figure (upstream
2870/// would nest it and lose the caption; keeping the caption is the better
2871/// failure). Tables without first-class cells (geometric fallback) have no cell
2872/// boxes to match against and nest nothing.
2873fn match_table_pictures(
2874    regions: &[Region],
2875    table_rows: &[Option<TableGrid>],
2876    caption_for: &[Option<usize>],
2877) -> std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> {
2878    let mut out: std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> =
2879        std::collections::HashMap::new();
2880    for (p, pic) in regions.iter().enumerate() {
2881        if pic.label != "picture" || caption_for.get(p).is_some_and(Option::is_some) {
2882            continue;
2883        }
2884        let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2885        let mut best: Option<(f32, usize, usize)> = None; // (coverage, table, cell)
2886        for (t, tbl) in regions.iter().enumerate() {
2887            if !is_table_like(tbl.label) {
2888                continue;
2889            }
2890            let Some(grid) = table_rows.get(t).and_then(Option::as_ref) else {
2891                continue;
2892            };
2893            if inter(pic, tbl.l, tbl.t, tbl.r, tbl.b) / pa < RICH_CELL_PICTURE_COVERAGE {
2894                continue;
2895            }
2896            if let Some((cov, cell)) = match_picture_to_cell(pic, &grid.cells) {
2897                if best.is_none_or(|(b, _, _)| cov > b) {
2898                    best = Some((cov, t, cell));
2899                }
2900            }
2901        }
2902        if let Some((_, t, cell)) = best {
2903            let entry = out.entry(t).or_default();
2904            match entry.iter_mut().find(|(c, _)| *c == cell) {
2905                Some((_, pics)) => pics.push(p),
2906                None => entry.push((cell, vec![p])),
2907            }
2908        }
2909    }
2910    out
2911}
2912
2913/// docling `_match_picture_to_table_cell`: among the cells covering ≥ 80 % of
2914/// the picture, prefer the one at the picture's inferred grid position (the
2915/// row / column whose median cell center is nearest the picture's center —
2916/// cell boxes can overlap across logical rows and columns), else the best
2917/// coverage. Returns `(coverage, cell index)`.
2918fn match_picture_to_cell(pic: &Region, cells: &[docling_core::TableCell]) -> Option<(f32, usize)> {
2919    let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2920    let cover = |b: &[f32; 4]| inter(pic, b[0], b[1], b[2], b[3]) / pa;
2921    let eligible: Vec<(f32, usize)> = cells
2922        .iter()
2923        .enumerate()
2924        .filter_map(|(i, c)| {
2925            let b = c.bbox.as_ref()?;
2926            let cov = cover(b);
2927            (cov >= RICH_CELL_PICTURE_COVERAGE).then_some((cov, i))
2928        })
2929        .collect();
2930    if eligible.is_empty() {
2931        return None;
2932    }
2933    let mut row_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2934    let mut col_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2935    for c in cells {
2936        let Some(b) = c.bbox.as_ref() else { continue };
2937        for r in c.start_row..c.start_row + c.row_span {
2938            row_centers.entry(r).or_default().push((b[1] + b[3]) / 2.0);
2939        }
2940        for k in c.start_col..c.start_col + c.col_span {
2941            col_centers.entry(k).or_default().push((b[0] + b[2]) / 2.0);
2942        }
2943    }
2944    let median = |v: &mut Vec<f32>| -> f32 {
2945        v.sort_by(f32::total_cmp);
2946        let n = v.len();
2947        if n % 2 == 1 {
2948            v[n / 2]
2949        } else {
2950            (v[n / 2 - 1] + v[n / 2]) / 2.0
2951        }
2952    };
2953    let (px, py) = ((pic.l + pic.r) / 2.0, (pic.t + pic.b) / 2.0);
2954    let nearest = |centers: &mut std::collections::BTreeMap<usize, Vec<f32>>, target: f32| {
2955        centers
2956            .iter_mut()
2957            .map(|(&i, v)| (i, (median(v) - target).abs()))
2958            .min_by(|a, b| a.1.total_cmp(&b.1))
2959            .map(|(i, _)| i)
2960    };
2961    let row = nearest(&mut row_centers, py);
2962    let col = nearest(&mut col_centers, px);
2963    let logical: Vec<(f32, usize)> = eligible
2964        .iter()
2965        .copied()
2966        .filter(|&(_, i)| {
2967            let c = &cells[i];
2968            row.is_some_and(|r| c.start_row <= r && r < c.start_row + c.row_span)
2969                && col.is_some_and(|k| c.start_col <= k && k < c.start_col + c.col_span)
2970        })
2971        .collect();
2972    let pool = if logical.is_empty() {
2973        &eligible
2974    } else {
2975        &logical
2976    };
2977    // Python's `max` over `(coverage, cell_index, cell)` tuples: highest
2978    // coverage, ties to the higher index.
2979    pool.iter()
2980        .copied()
2981        .max_by(|a, b| a.0.total_cmp(&b.0).then(a.1.cmp(&b.1)))
2982}
2983
2984/// The DocLang structure overlay derived from first-class cells: span
2985/// continuations (`lcel`/`ucel`/`xcel`) and per-cell header roles, so the
2986/// PDF path's DCLX carries real spans instead of a flat grid.
2987fn structure_from_cells(
2988    cells: &[docling_core::TableCell],
2989    nrows: usize,
2990    ncols: usize,
2991) -> docling_core::TableStructure {
2992    let grid = || vec![vec![false; ncols]; nrows];
2993    let mut col_cont = grid();
2994    let mut row_cont = grid();
2995    let mut row_header = grid();
2996    let mut col_header = grid();
2997    for c in cells {
2998        for r in c.start_row..(c.start_row + c.row_span).min(nrows) {
2999            for k in c.start_col..(c.start_col + c.col_span).min(ncols) {
3000                col_cont[r][k] = k > c.start_col;
3001                row_cont[r][k] = r > c.start_row;
3002                row_header[r][k] = c.row_header;
3003                col_header[r][k] = c.column_header;
3004            }
3005        }
3006    }
3007    docling_core::TableStructure {
3008        header_row: Vec::new(),
3009        col_continuation: col_cont,
3010        row_continuation: row_cont,
3011        row_header,
3012        col_header,
3013    }
3014}
3015
3016pub fn assemble_page(
3017    page: &PdfPage,
3018    mut regions: Vec<Region>,
3019    table_rows: &[Option<TableGrid>],
3020    enrichments: &[Option<Enrichment>],
3021    // Picture-crop scale in px/pt (docling's `images_scale`, #520); `None`
3022    // keeps the page render's own scale.
3023    picture_scale: Option<f32>,
3024) -> (Vec<Node>, Vec<(String, String)>) {
3025    // Without pixels (the text-layer-only wasm build) no picture is cropped.
3026    #[cfg(not(feature = "ocr-prep"))]
3027    let _ = picture_scale;
3028    let mut nodes: Vec<Node> = Vec::new();
3029    // Every page opens with an invisible page marker carrying its size in
3030    // points — what the JSON export needs to build docling's `pages` map and
3031    // denormalize the 0–511 `<location>` grid into point bboxes (#171). The
3032    // page *number* is stamped by the document-level collector (which knows
3033    // the real 1-based index, `--pages` windows included); every serializer
3034    // except JSON skips the marker, so Markdown/DocLang stay byte-identical.
3035    nodes.push(Node::PageInfo {
3036        page_no: 0,
3037        width: page.width,
3038        height: page.height,
3039    });
3040    // Recover this page's hyperlinks (anchor-precise pairs for strict
3041    // Markdown; whole-item docling-parity links are baked below and their
3042    // pairs dropped from this list so strict output doesn't double-wrap).
3043    let mut links = resolve_link_anchors(page);
3044    // Lines with a drawn checkbox square in front become checkbox items (#609).
3045    split_checkbox_lines(&mut regions, &page.cells, &page.checkboxes);
3046    // Pair each region with its precomputed TableFormer grid and enrichment
3047    // (indexed by original order) and order by reading order together, so they
3048    // stay aligned.
3049    // A picture's children (docling's `_set_cluster_children`: the regulars
3050    // > 80 % inside it) are not page elements — they leave the reading order
3051    // here and ride with their picture, to be written under it in the JSON.
3052    let parents = picture_parents(&regions);
3053    let mut kids: Vec<Vec<Region>> = vec![Vec::new(); regions.len()];
3054    let mut top: Vec<(usize, Region)> = Vec::with_capacity(regions.len());
3055    for (i, (r, parent)) in regions.into_iter().zip(parents).enumerate() {
3056        match parent {
3057            Some(p) => kids[p].push(r),
3058            None => top.push((i, r)),
3059        }
3060    }
3061    // docling's assembly order of the regions — what its reading-order
3062    // predictor knows as `cid` (#424) — before they are shuffled.
3063    let top_regions: Vec<Region> = top.iter().map(|(_, r)| r.clone()).collect();
3064    let cids = cluster_cids(&top_regions, &page.cells);
3065    type RegionItem = (Region, Option<TableGrid>, Option<Enrichment>, Vec<Region>);
3066    let mut items: Vec<RegionItem> = top
3067        .into_iter()
3068        .map(|(i, r)| {
3069            (
3070                r,
3071                table_rows.get(i).cloned().flatten(),
3072                enrichments.get(i).cloned().flatten(),
3073                std::mem::take(&mut kids[i]),
3074            )
3075        })
3076        .collect();
3077    order_with_containers(&mut items, &cids, page.width, page.height, |it| &it.0);
3078    // Float a margin page number to the front of reading order (docling parity:
3079    // right_to_left_02's bottom `11` is its first item). Stable, so everything
3080    // else keeps its order; no-op on pages without such a region.
3081    let page_h = page.height;
3082    items.sort_by_key(|(r, _, _, _)| !is_page_number(r, &page.cells, page_h));
3083    let table_rows: Vec<Option<TableGrid>> = items.iter().map(|(_, t, _, _)| t.clone()).collect();
3084    let enrichments: Vec<Option<Enrichment>> = items.iter().map(|(_, _, e, _)| e.clone()).collect();
3085    let mut picture_children: Vec<Vec<Region>> = items
3086        .iter_mut()
3087        .map(|it| std::mem::take(&mut it.3))
3088        .collect();
3089    let regions: Vec<Region> = items.into_iter().map(|(r, _, _, _)| r).collect();
3090    // Children in docling's `_sort_clusters(mode="id")` order: first source
3091    // cell, then top, then left.
3092    for kids in picture_children.iter_mut().filter(|k| k.len() > 1) {
3093        let rank = cluster_cids(kids, &page.cells);
3094        let mut ranked: Vec<(usize, Region)> = rank.into_iter().zip(kids.drain(..)).collect();
3095        ranked.sort_by_key(|(k, _)| *k);
3096        kids.extend(ranked.into_iter().map(|(_, r)| r));
3097    }
3098    // docling emits a figure's caption *before* the image marker. Pair each
3099    // picture with the caption region nearest below it and consume that caption,
3100    // so it isn't also emitted in its own (lower) reading-order position.
3101    let caption_for = pair_captions(&regions);
3102    let code_caption_for = pair_code_captions(&regions);
3103    let mut consumed = vec![false; regions.len()];
3104    for ci in caption_for.iter().flatten() {
3105        consumed[*ci] = true;
3106    }
3107    for ci in code_caption_for.iter().flatten() {
3108        consumed[*ci] = true;
3109    }
3110    // Table captions (#265) claim from what the picture/code pairings left.
3111    let mut caption_taken = consumed.clone();
3112    let table_caption_for = pair_table_captions(&regions, &mut caption_taken);
3113    for ci in table_caption_for.iter().flatten() {
3114        consumed[*ci] = true;
3115    }
3116    // Pictures inside a table become rich-cell content (docling#3906, 2.118.1):
3117    // the picture is nested in the cell it covers and not emitted standalone.
3118    let rich_cell_pictures = match_table_pictures(&regions, &table_rows, &caption_for);
3119    for (_, pics) in rich_cell_pictures.values().flatten() {
3120        for &p in pics {
3121            consumed[p] = true;
3122        }
3123    }
3124    // A code block's language label (`XML`, `C#`, …) is chrome, not content — the
3125    // detector emits it as its own region above the code; consume it.
3126    for (i, is_label) in code_language_labels(&regions, &page.cells)
3127        .into_iter()
3128        .enumerate()
3129    {
3130        if is_label {
3131            consumed[i] = true;
3132        }
3133    }
3134
3135    // docling `ReadingOrderPredictor.predict_merges`: join a text fragment with a
3136    // following text fragment strictly to its right (an author column that wraps
3137    // into the next, a paragraph continuing in the next column) into one block —
3138    // the intra-page half of docling's reading-order merges (cross-page/vertical
3139    // continuations stay with [`merge_continuations`]). Already-consumed regions
3140    // (paired captions, code labels) are excluded.
3141    // Exclusive docling cell assignment: computed once for the ordered region
3142    // list and reused for every serialization below, so a cell can never render
3143    // in two regions. The picture children take part (docling assigns cells to
3144    // every regular cluster before it nests any); their texts are split off.
3145    let with_children: Vec<Region> = regions
3146        .iter()
3147        .chain(picture_children.iter().flatten())
3148        .cloned()
3149        .collect();
3150    let mut region_texts: Vec<String> = region_texts_exclusive(&with_children, &page.cells);
3151    let mut kid_texts = region_texts.split_off(regions.len()).into_iter();
3152    let child_texts: Vec<Vec<String>> = picture_children
3153        .iter()
3154        .map(|k| kid_texts.by_ref().take(k.len()).collect())
3155        .collect();
3156    let is_text: Vec<bool> = regions
3157        .iter()
3158        .enumerate()
3159        .map(|(i, r)| r.label == "text" && !consumed[i])
3160        .collect();
3161    let is_skip: Vec<bool> = regions
3162        .iter()
3163        .enumerate()
3164        .map(|(i, r)| {
3165            consumed[i]
3166                || matches!(
3167                    r.label,
3168                    "page_header" | "page_footer" | "table" | "picture" | "caption" | "footnote"
3169                )
3170        })
3171        .collect();
3172    let boxes: Vec<(f32, f32, f32, f32)> = regions.iter().map(|r| (r.l, r.t, r.r, r.b)).collect();
3173    if docling_core::env::flag("DOCLING_RS_DEBUG_MERGES") {
3174        for (i, r) in regions.iter().enumerate() {
3175            eprintln!(
3176                "MRG {i:2} {} text={} skip={} [{:.0},{:.0},{:.0},{:.0}] {:?}",
3177                r.label,
3178                is_text[i],
3179                is_skip[i],
3180                r.l,
3181                r.t,
3182                r.r,
3183                r.b,
3184                region_texts[i].chars().take(40).collect::<String>()
3185            );
3186        }
3187    }
3188    let mut merge_suffix: Vec<String> = vec![String::new(); regions.len()];
3189    for (head, children) in
3190        crate::reading_order::predict_merges(&boxes, &region_texts, &is_text, &is_skip)
3191            .into_iter()
3192            .enumerate()
3193    {
3194        for c in children {
3195            let t = region_texts[c].trim();
3196            if !t.is_empty() {
3197                merge_suffix[head].push(' ');
3198                merge_suffix[head].push_str(t);
3199            }
3200            consumed[c] = true;
3201        }
3202    }
3203
3204    for (i, region) in regions.iter().enumerate() {
3205        if consumed[i] {
3206            continue;
3207        }
3208        // Page headers/footers: docling emits them as furniture blocks
3209        // (`<page_header>`/`<page_footer>` with a layer + location + text) at
3210        // their reading-order position, not as body — emit them, don't skip.
3211        if matches!(region.label, "page_header" | "page_footer") {
3212            let text = region_texts[i].clone();
3213            if !text.is_empty() {
3214                nodes.push(Node::PageFurniture {
3215                    footer: region.label == "page_footer",
3216                    location: norm_loc(region, page.width, page_h),
3217                    text: md_escape(&text),
3218                });
3219            }
3220            continue;
3221        }
3222        if is_skipped(region.label) {
3223            continue;
3224        }
3225        // Layout provenance for this region, normalized to docling's 0–511 grid.
3226        let loc = norm_loc(region, page.width, page_h);
3227        if region.label == "picture" {
3228            // The figure pixels are cropped from the page render for image export.
3229            // Captions are prose: markdown-escaped like a paragraph (the JSON
3230            // export unescapes back to the raw text, matching docling).
3231            let caption = caption_for[i]
3232                .map(|ci| md_escape(&region_texts[ci]))
3233                .filter(|t| !t.is_empty());
3234            // The caption's own region box (#609): docling gives every caption
3235            // the cluster it came from as its `prov`, not the picture's.
3236            let caption_location = caption_for[i]
3237                .filter(|_| caption.is_some())
3238                .map(|ci| norm_loc(&regions[ci], page.width, page_h));
3239            let classification = match &enrichments[i] {
3240                Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
3241                _ => None,
3242            };
3243            // Without the page render (text-layer-only build) a picture keeps
3244            // its caption/classification but carries no cropped pixels.
3245            #[cfg(feature = "ocr-prep")]
3246            let image =
3247                crate::timing::timed("crop_region", || crop_region(page, region, picture_scale));
3248            #[cfg(not(feature = "ocr-prep"))]
3249            let image: Option<PictureImage> = None;
3250            nodes.push(located(
3251                loc,
3252                Node::Picture {
3253                    caption,
3254                    caption_href: None,
3255                    image,
3256                    classification,
3257                    // docling's layout pipeline parents a figure's caption to
3258                    // the picture itself (#390) — the one backend that does.
3259                    caption_parent: CaptionParent::Item,
3260                    caption_location,
3261                },
3262            ));
3263            let children: Vec<Node> = picture_children[i]
3264                .iter()
3265                .zip(&child_texts[i])
3266                .filter_map(|(r, text)| {
3267                    picture_child_node(r, text, norm_loc(r, page.width, page_h))
3268                })
3269                .collect();
3270            if !children.is_empty() {
3271                nodes.push(Node::PictureChildren(children));
3272            }
3273            continue;
3274        }
3275        let mut text = region_texts[i].clone();
3276        text.push_str(&merge_suffix[i]);
3277        if text.is_empty() {
3278            continue;
3279        }
3280        match region.label {
3281            // docling assembles checkboxes as TEXT_ELEM items (the region's
3282            // cells are the option label, e.g. right_to_left_03's بلی/خير)
3283            // and its Markdown serializer renders them as task-list lines
3284            // (`- [x] …`) — mirrored by [`Node::CheckboxItem`].
3285            // Located like every other text item, so the JSON item carries
3286            // its page and box (#609) — a chunk of checkboxes has a page.
3287            // The label without a leading ballot-box glyph: the item's state
3288            // already says what `☐` / `☒` drew (#609).
3289            "checkbox_selected" | "checkbox_unselected" => nodes.push(located(
3290                loc,
3291                Node::CheckboxItem {
3292                    checked: region.label == "checkbox_selected",
3293                    text: md_escape(strip_checkbox_glyph(&text)),
3294                },
3295            )),
3296            // docling renders both the document title and section headers as
3297            // `##` (it never emits a top-level `#` for PDFs), so match that.
3298            "title" | "section_header" => nodes.push(located(
3299                loc,
3300                Node::Heading {
3301                    level: 2,
3302                    text: md_escape(&text),
3303                },
3304            )),
3305            // docling's `ListItemMarkerProcessor.process_list_item` runs on
3306            // every PDF list item: a leading bullet glyph or enumeration marker
3307            // followed by whitespace is split off into the item's `marker`, and
3308            // docling-core's Markdown then prints `- text` for a bullet, `N. text`
3309            // for an `N.` marker and `- a) text` for any other marker holding a
3310            // letter or digit (see [`list_item_node`]). The symbol-font bullets
3311            // docling-parse filters out of its cells are stripped first.
3312            "list_item" => nodes.push(list_item_node(&text, loc, false)),
3313            // TableFormer structure (cells + spans, text matched from word cells)
3314            // when available; otherwise geometric grid reconstruction; finally a
3315            // single cell.
3316            "table" | "document_index" => {
3317                // TableFormer grids carry first-class cells (#240: text +
3318                // page-point bbox + span rectangle + OTSL header roles) into
3319                // the public model, and the DocLang structure overlay derives
3320                // from them so DCLX emits real span/header tokens. The
3321                // geometric fallback has no per-cell records.
3322                let (mut rows, cells, structure) = match table_rows[i].clone() {
3323                    Some(grid) => {
3324                        let nrows = grid.rows.len();
3325                        let ncols = grid.rows.first().map_or(0, Vec::len);
3326                        let structure = structure_from_cells(&grid.cells, nrows, ncols);
3327                        (grid.rows, Some(grid.cells), Some(structure))
3328                    }
3329                    None => {
3330                        let rows = reconstruct_table(region, &page.cells);
3331                        let rows = if rows.iter().any(|r| r.len() > 1) {
3332                            rows
3333                        } else {
3334                            vec![vec![text.clone()]]
3335                        };
3336                        (rows, None, None)
3337                    }
3338                };
3339                // The paired caption (#265) rides on the table — docling's
3340                // TableItem.captions ref; Markdown prints it above the grid,
3341                // the JSON export emits the $ref, DocLang the <caption>.
3342                let caption = table_caption_for[i]
3343                    .map(|ci| md_escape(&region_texts[ci]))
3344                    .filter(|t| !t.is_empty());
3345                // Its own region box becomes the caption item's `prov` (#609).
3346                let caption_location = table_caption_for[i]
3347                    .filter(|_| caption.is_some())
3348                    .map(|ci| norm_loc(&regions[ci], page.width, page_h));
3349                // Rich cells (docling#3906): the covering cell's blocks are its
3350                // text followed by the nested picture(s). docling's Markdown
3351                // renders a `RichTableCell` through the serializer — the
3352                // group's children joined by blank lines, newlines flattened
3353                // to spaces — so the flat `rows` text becomes
3354                // `text  <!-- image -->`; the first-class `cells` (the JSON
3355                // `table_cells` / `grid`) keep the plain text, as upstream.
3356                let mut cell_blocks: Option<Vec<Vec<Vec<Node>>>> = None;
3357                if let (Some(by_cell), Some(fc)) = (rich_cell_pictures.get(&i), cells.as_ref()) {
3358                    let nrows = rows.len();
3359                    let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
3360                    let mut blocks = vec![vec![Vec::<Node>::new(); ncols]; nrows];
3361                    for (cell_idx, pics) in by_cell {
3362                        let cell = &fc[*cell_idx];
3363                        let (r, c) = (cell.start_row, cell.start_col);
3364                        if r >= nrows || c >= ncols {
3365                            continue;
3366                        }
3367                        let mut parts: Vec<String> = Vec::new();
3368                        let mut cell_nodes: Vec<Node> = Vec::new();
3369                        if !cell.text.trim().is_empty() {
3370                            parts.push(cell.text.clone());
3371                            cell_nodes.push(Node::Paragraph {
3372                                text: cell.text.clone(),
3373                            });
3374                        }
3375                        for &p in pics {
3376                            parts.push("<!-- image -->".to_string());
3377                            let classification = match &enrichments[p] {
3378                                Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
3379                                _ => None,
3380                            };
3381                            #[cfg(feature = "ocr-prep")]
3382                            let image = crop_region(page, &regions[p], picture_scale);
3383                            #[cfg(not(feature = "ocr-prep"))]
3384                            let image: Option<PictureImage> = None;
3385                            cell_nodes.push(located(
3386                                norm_loc(&regions[p], page.width, page_h),
3387                                Node::Picture {
3388                                    caption: None,
3389                                    caption_href: None,
3390                                    image,
3391                                    classification,
3392                                    caption_parent: Default::default(),
3393                                    caption_location: None,
3394                                },
3395                            ));
3396                        }
3397                        let rendered = parts.join("  ");
3398                        for row in rows.iter_mut().skip(r).take(cell.row_span) {
3399                            for slot in row.iter_mut().skip(c).take(cell.col_span) {
3400                                *slot = rendered.clone();
3401                            }
3402                        }
3403                        blocks[r][c] = cell_nodes;
3404                    }
3405                    cell_blocks = Some(blocks);
3406                }
3407                nodes.push(located(
3408                    loc,
3409                    Node::Table(Table {
3410                        rows,
3411                        location: None,
3412                        structure,
3413                        cell_blocks,
3414                        cells,
3415                        caption,
3416                        // As for pictures: the caption is the table's child.
3417                        caption_parent: CaptionParent::Item,
3418                        caption_location,
3419                    }),
3420                ));
3421            }
3422            // With formula enrichment the CodeFormula model decodes the region
3423            // to LaTeX; otherwise docling emits a placeholder comment rather
3424            // than the (garbled) raw glyph text.
3425            "formula" => match &enrichments[i] {
3426                Some(Enrichment::Formula { latex }) => nodes.push(Node::Formula {
3427                    latex: latex.clone(),
3428                    orig: text.clone(),
3429                    location: Some(loc),
3430                }),
3431                _ => nodes.push(Node::Paragraph {
3432                    text: "<!-- formula-not-decoded -->".into(),
3433                }),
3434            },
3435            // Code blocks: use the space-glyph-only grouping (monospace keeps its
3436            // source spacing) and emit a fenced block, preserving the line breaks
3437            // and indentation of the source (unlike prose, which reflows). pdfium
3438            // still inserts spaces around tight punctuation (`console .log`,
3439            // `add (3 , 5)`); tighten them to match docling-parse's source spacing.
3440            "code" => {
3441                // `code_region_text` preserves line breaks/indentation and tightens
3442                // each line itself; the fallback prose `text` is tightened here.
3443                let code = code_region_text(region, &page.code_cells);
3444                let code = if code.is_empty() {
3445                    tighten_code_punct(&text)
3446                } else {
3447                    code
3448                };
3449                // With code enrichment the CodeFormula model rewrites the block
3450                // (and names its language); `orig` keeps the raw extraction in
3451                // docling's shape — its parser has no line-preserving code
3452                // path, so its `orig` is the same code with the lines joined
3453                // by single spaces (indentation collapsed).
3454                // docling's parser has no line-preserving code path — its code
3455                // items carry the lines joined by single spaces. That flat
3456                // form is what every byte-conformance surface serializes
3457                // (legacy Markdown, JSON, DocLang); the line-preserving
3458                // extraction rides in `pretty` for strict Markdown only.
3459                let flat = code
3460                    .lines()
3461                    .map(str::trim)
3462                    .filter(|l| !l.is_empty())
3463                    .collect::<Vec<_>>()
3464                    .join(" ");
3465                let node = match &enrichments[i] {
3466                    Some(Enrichment::Code {
3467                        language,
3468                        text: enriched,
3469                    }) => Node::Code {
3470                        language: language.clone(),
3471                        text: enriched.clone(),
3472                        orig: Some(flat),
3473                        pretty: None,
3474                    },
3475                    _ => Node::Code {
3476                        language: None,
3477                        text: flat,
3478                        orig: None,
3479                        pretty: Some(code),
3480                    },
3481                };
3482                nodes.push(located(loc, node));
3483                // docling emits the `Listing N:` caption after the code block.
3484                if let Some(ci) = code_caption_for[i] {
3485                    let cap = md_escape(&region_texts[ci]);
3486                    if !cap.is_empty() {
3487                        // With its own region box, like every caption (#609).
3488                        nodes.push(located(
3489                            norm_loc(&regions[ci], page.width, page_h),
3490                            Node::Paragraph { text: cap },
3491                        ));
3492                    }
3493                }
3494            }
3495            // text, caption, footnote → paragraph
3496            _ => {
3497                // docling parity (`PageAssembleModel._match_hyperlink`): when
3498                // link annotations cover ≥ half of the region's box, the
3499                // hyperlink attaches to the item and the legacy Markdown
3500                // serializer wraps its full text — 2206.01062's footnote URLs
3501                // render as `[1 https://…](https://…)`. Sparse in-paragraph
3502                // citation links stay below the 0.5 coverage threshold and
3503                // remain plain text, exactly like docling.
3504                //
3505                // Scope: **footnote regions only.** Upstream's page_assemble
3506                // matches every TEXT_ELEM label, but published docling
3507                // observably carries the hyperlink into the document only for
3508                // footnote items — in both committed groundtruth generations
3509                // (docling-JSON and Markdown, independent runs) the fully
3510                // covered plain-text DOI line of 2206.01062 page 1 has
3511                // `hyperlink: None` while the equally covered footnotes carry
3512                // theirs. The corpus is the conformance reference, so match
3513                // the observed behavior; widen the label set if a future
3514                // groundtruth refresh starts linking plain text too.
3515                let escaped = md_escape(&text);
3516                let hyperlink = (region.label == "footnote")
3517                    .then(|| region_hyperlink(region, &page.links))
3518                    .flatten();
3519                if let Some(uri) = &hyperlink {
3520                    // The strict-mode anchor pairs this item covers are
3521                    // superseded by the whole-item link.
3522                    links.retain(|(anchor, href)| {
3523                        !(href == uri && region_texts[i].contains(anchor.as_str()))
3524                    });
3525                }
3526                // A footnote keeps docling's label and carries its link as
3527                // the item's `hyperlink` (the JSON's `text` stays the raw
3528                // footnote, not Markdown); every other serializer renders it
3529                // as the `[text](uri)` paragraph it was before.
3530                let node = if region.label == "footnote" {
3531                    Node::LabeledText {
3532                        label: "footnote".into(),
3533                        text: escaped,
3534                        href: hyperlink,
3535                    }
3536                } else {
3537                    Node::Paragraph { text: escaped }
3538                };
3539                nodes.push(located(loc, node))
3540            }
3541        }
3542    }
3543    // A `/Rotate`-normalized scanned page (see `pdfium_backend`) was assembled
3544    // in upright space; rotate the finished geometry back so locations and the
3545    // page size are display-space, like docling and every viewer report them.
3546    if page.rotation != 0 {
3547        rotate_nodes_to_display(&mut nodes, page.rotation);
3548    }
3549    (nodes, links)
3550}
3551
3552/// One child of a picture as docling's `ReadingOrderModel._add_child_elements`
3553/// writes it under the `PictureItem`: a heading for a `section_header` /
3554/// `title` (upstream remaps title to section header), a list item for a
3555/// `list_item`, a furniture-layer text for a page header/footer, a caption,
3556/// otherwise a text item. `None` for a child that claimed no text.
3557fn picture_child_node(region: &Region, text: &str, loc: [u16; 4]) -> Option<Node> {
3558    if text.is_empty() {
3559        return None;
3560    }
3561    Some(match region.label {
3562        "title" | "section_header" => located(
3563            loc,
3564            Node::Heading {
3565                level: 2,
3566                text: md_escape(text),
3567            },
3568        ),
3569        // docling-core's `add_list_item` under a non-list parent opens a
3570        // list group per item, so every child item starts its own list;
3571        // `_add_child_elements` runs the marker processor on it too.
3572        "list_item" => list_item_node(text, loc, true),
3573        "page_header" | "page_footer" => Node::PageFurniture {
3574            footer: region.label == "page_footer",
3575            location: loc,
3576            text: md_escape(text),
3577        },
3578        "caption" => located(
3579            loc,
3580            Node::Caption {
3581                text: md_escape(text),
3582                href: None,
3583            },
3584        ),
3585        _ => located(
3586            loc,
3587            Node::Paragraph {
3588                text: md_escape(text),
3589            },
3590        ),
3591    })
3592}
3593
3594/// Rotate one 0–511 location bbox 90° clockwise on the grid (top-left origin):
3595/// `(x, y) → (511 - y, x)`.
3596fn rot_loc_cw(l: [u16; 4]) -> [u16; 4] {
3597    [511 - l[3], l[0], 511 - l[1], l[2]]
3598}
3599
3600/// Map upright-space geometry back to display space for a page whose `/Rotate`
3601/// was normalized away before inference: every `<location>` rotates `rot`°
3602/// clockwise on the 0–511 grid (the grid is per-axis normalized, so no page
3603/// dims are needed), and the `PageInfo` size returns to the display box. Node
3604/// text and order are untouched — reading order was decided upright, which is
3605/// the whole point.
3606fn rotate_nodes_to_display(nodes: &mut [Node], rot: u16) {
3607    let quarter_turns = (rot / 90) as usize;
3608    let rot_loc = |l: &mut [u16; 4]| {
3609        for _ in 0..quarter_turns {
3610            *l = rot_loc_cw(*l);
3611        }
3612    };
3613    fn walk(node: &mut Node, rot_loc: &impl Fn(&mut [u16; 4]), swap_dims: bool) {
3614        match node {
3615            Node::PageInfo { width, height, .. } => {
3616                if swap_dims {
3617                    std::mem::swap(width, height);
3618                }
3619            }
3620            Node::Located { location, inner } => {
3621                rot_loc(location);
3622                walk(inner, rot_loc, swap_dims);
3623            }
3624            Node::Furniture { inner, .. } => walk(inner, rot_loc, swap_dims),
3625            Node::Group { children, .. } | Node::PictureChildren(children) => {
3626                for c in children {
3627                    walk(c, rot_loc, swap_dims);
3628                }
3629            }
3630            Node::ListItem { location, .. }
3631            | Node::Formula { location, .. }
3632            | Node::Chart { location, .. } => {
3633                if let Some(l) = location {
3634                    rot_loc(l);
3635                }
3636            }
3637            Node::PageFurniture { location, .. } => rot_loc(location),
3638            Node::Table(t) => {
3639                if let Some(l) = &mut t.location {
3640                    rot_loc(l);
3641                }
3642            }
3643            _ => {}
3644        }
3645    }
3646    let swap_dims = quarter_turns % 2 == 1;
3647    for node in nodes {
3648        walk(node, &rot_loc, swap_dims);
3649    }
3650}
3651
3652/// Merge paragraph fragments split across a column or page break. docling joins a
3653/// paragraph whose previous fragment ends mid-sentence (a letter, not sentence
3654/// punctuation) with a lowercase continuation: `…definition of` + `lists in…` →
3655/// `…definition of lists in…`. The fragments are consecutive paragraphs, or
3656/// separated only by figure(s) the text wraps around: a column whose body flows
3657/// past a figure resumes below it (`…The wing type that is` ⟶[figure]⟶ `the most
3658/// common…`), and docling emits the whole paragraph before the figure. A heading,
3659/// table, or list between them ends the paragraph (no merge).
3660/// A paragraph that is really a figure/table caption (`Fig. 1. …`, `Table 2 …`).
3661/// Used to skip an unpaired caption when stitching a paragraph that wraps around
3662/// a figure.
3663fn looks_like_caption(text: &str) -> bool {
3664    let head: String = text.trim_start().chars().take(14).collect();
3665    (head.starts_with("Fig") || head.starts_with("Table"))
3666        && head.contains(|c: char| c.is_ascii_digit())
3667}
3668
3669/// A paragraph fragment is "open" — i.e. it might continue into the next
3670/// paragraph — when it ends mid-word (a letter) or with a wrap hyphen/dash.
3671/// docling joins `vocab-` + `ulary` → `vocab- ulary`.
3672fn paragraph_is_open(text: &str) -> bool {
3673    // docling's merge head test (`.+([a-z,\-\u00AD])\s*`): at least two chars,
3674    // ending in an ASCII lowercase letter, a comma, a hyphen, or a soft
3675    // hyphen. The comma matters: 2206's "…In phase four," resumes across the
3676    // page break. Uppercase/non-Latin endings do not merge, exactly as
3677    // upstream (the dash family is already `-` here — clean_text normalized).
3678    let t = text.trim_end();
3679    t.chars().count() >= 2
3680        && t.chars()
3681            .next_back()
3682            .is_some_and(|c| matches!(c, 'a'..='z' | ',' | '-' | '\u{ad}'))
3683}
3684
3685/// The paragraph text inside a node, looking through a [`Node::Located`]
3686/// provenance wrapper (PDF body paragraphs are wrapped since they carry a
3687/// `<location>`). Returns `None` for non-paragraph nodes.
3688fn as_paragraph(n: &Node) -> Option<&str> {
3689    match n {
3690        Node::Paragraph { text } => Some(text),
3691        Node::Located { inner, .. } => match inner.as_ref() {
3692            Node::Paragraph { text } => Some(text),
3693            _ => None,
3694        },
3695        _ => None,
3696    }
3697}
3698
3699/// Whether a node is a picture, looking through a [`Node::Located`] wrapper.
3700fn is_picture_node(n: &Node) -> bool {
3701    match n {
3702        Node::Picture { .. } => true,
3703        Node::Located { inner, .. } => matches!(inner.as_ref(), Node::Picture { .. }),
3704        _ => false,
3705    }
3706}
3707
3708/// A node a forward paragraph merge looks straight past: a figure or *table*
3709/// the text wraps around, or a page header/footer that falls between the two
3710/// fragments of a paragraph continuing across a page break (docling's merge
3711/// skip-labels: page_header, page_footer, table, picture, caption, footnote —
3712/// 2206's "…In phase four," resumes after a full caption+table+figure block).
3713fn is_merge_trailer(n: &Node) -> bool {
3714    is_picture_node(n)
3715        || matches!(
3716            n,
3717            Node::PageFurniture { .. }
3718                | Node::PageInfo { .. }
3719                | Node::Table(_)
3720                | Node::PictureChildren(_)
3721        )
3722        || matches!(n, Node::Located { inner, .. } if matches!(inner.as_ref(), Node::Table(_)))
3723        || is_footnote_node(n)
3724        || as_paragraph(n).is_some_and(looks_like_caption)
3725}
3726
3727/// Whether a node is a footnote item (a [`Node::LabeledText`] labelled
3728/// `footnote`), looking through a [`Node::Located`] wrapper — one of
3729/// docling's merge skip-labels: a footnote between the two halves of a
3730/// paragraph is looked past, never merged into.
3731fn is_footnote_node(n: &Node) -> bool {
3732    let n = match n {
3733        Node::Located { inner, .. } => inner.as_ref(),
3734        other => other,
3735    };
3736    matches!(n, Node::LabeledText { label, .. } if label == "footnote")
3737}
3738
3739/// Rebuild node `i` as a paragraph with `text`, preserving its `<location>`
3740/// wrapper (and thus provenance) if it had one.
3741fn reparagraph(node: &Node, text: String) -> Node {
3742    match node {
3743        Node::Located { location, .. } => located(*location, Node::Paragraph { text }),
3744        _ => Node::Paragraph { text },
3745    }
3746}
3747
3748pub(crate) fn merge_continuations(nodes: &mut Vec<Node>) {
3749    let mut i = 0;
3750    while i + 1 < nodes.len() {
3751        let Some(a) = as_paragraph(&nodes[i]) else {
3752            i += 1;
3753            continue;
3754        };
3755        // A figure/table caption is a self-contained unit; body text resuming
3756        // after a figure is the continuation case, not the caption itself. Never
3757        // stitch *from* a caption — otherwise a caption that ends in a lone glyph
3758        // (`Fig. 5. … PubTabNet. μ`) would swallow a following stray figure label
3759        // (a standalone `μ`) into `… μ μ`.
3760        if looks_like_caption(a) {
3761            i += 1;
3762            continue;
3763        }
3764        if !paragraph_is_open(a) {
3765            i += 1;
3766            continue;
3767        }
3768        // The continuation is the next paragraph, looking past any figures the
3769        // text wraps around — and a figure/table caption that was emitted as its
3770        // own paragraph (an above-the-figure caption that didn't pair), since the
3771        // body text resumes after the whole figure+caption block.
3772        let mut j = i + 1;
3773        while nodes.get(j).is_some_and(is_merge_trailer) {
3774            j += 1;
3775        }
3776        // docling's continuation regex allows either case, but its merge runs
3777        // over the pre-assembly element stream; at node level an uppercase
3778        // start is overwhelmingly a new sentence/heading fragment (allowing it
3779        // swallowed 2305's formula blocks and redp's chapter openers), so the
3780        // continuation stays lowercase-start here.
3781        let cont = nodes.get(j).and_then(as_paragraph).is_some_and(|b| {
3782            b.trim_start()
3783                .chars()
3784                .next()
3785                .is_some_and(char::is_lowercase)
3786        });
3787        if cont {
3788            let a = as_paragraph(&nodes[i]).unwrap().trim_end().to_string();
3789            let b = as_paragraph(&nodes[j]).unwrap().trim_start().to_string();
3790            // A soft hyphen -- or a hard hyphen followed by a lowercase
3791            // continuation (guaranteed lowercase by the `cont` gate above) --
3792            // is a word split across the break: strip it and join without a
3793            // space, docling#3888 ("vocab-" + "ulary" -> "vocabulary");
3794            // docling's older serializer kept the artifact ("vocab- ulary").
3795            // Everything else joins with the space, as before.
3796            let merged = match a.strip_suffix('\u{ad}').or_else(|| a.strip_suffix('-')) {
3797                Some(stem) => format!("{stem}{b}"),
3798                None => format!("{a} {b}"),
3799            };
3800            // Keep node i's provenance wrapper; docling's merged paragraph keeps
3801            // the first fragment's geometry as its primary location.
3802            nodes[i] = reparagraph(&nodes[i], merged);
3803            nodes.remove(j);
3804            // Re-check i: the merged paragraph may continue further.
3805        } else {
3806            i += 1;
3807        }
3808    }
3809}
3810
3811/// How many leading nodes of `nodes` are safe to flush now — i.e. cannot be
3812/// rewritten by a future [`merge_continuations`] once more pages are appended.
3813///
3814/// A forward merge can only start from an "open" paragraph (ends mid-word) and
3815/// only reaches across trailing pictures and figure/table captions. So we scan
3816/// from the end past those skippable trailers: if the first non-skippable node is
3817/// an open paragraph, it (and the trailers after it) must be held; anything else —
3818/// a closed paragraph, a heading, a table, a list — blocks any forward merge, so
3819/// the whole buffer is safe to flush.
3820fn hold_start(nodes: &[Node]) -> usize {
3821    for k in (0..nodes.len()).rev() {
3822        // Skippable trailers (figures, page furniture, captions): a forward merge
3823        // looks straight past them.
3824        if is_merge_trailer(&nodes[k]) {
3825            continue;
3826        }
3827        match as_paragraph(&nodes[k]) {
3828            // An open body paragraph might still pull a continuation off the next
3829            // page — hold from here to the end.
3830            Some(text) if paragraph_is_open(text) => return k,
3831            // A closed paragraph, heading, table, list, etc. ends the paragraph:
3832            // nothing after it can merge backwards across it. Flush everything.
3833            _ => return nodes.len(),
3834        }
3835    }
3836    // Only skippable trailers (or empty) and no open paragraph to anchor a merge.
3837    nodes.len()
3838}
3839
3840/// Streaming counterpart of [`merge_continuations`]: feed per-page node batches in
3841/// document order and get back the prefix that is final (its cross-page merges are
3842/// resolved and no future page can change it), holding back only the small tail
3843/// that might still merge into the next page. Concatenating every flushed batch
3844/// (then [`finish`](Self::finish)) yields exactly the same nodes as running
3845/// [`merge_continuations`] once over the whole document.
3846pub(crate) struct StreamAssembler {
3847    pending: Vec<Node>,
3848}
3849
3850impl StreamAssembler {
3851    pub(crate) fn new() -> Self {
3852        Self {
3853            pending: Vec::new(),
3854        }
3855    }
3856
3857    /// Append one page's nodes, resolve merges within the buffer, and return the
3858    /// now-final prefix to emit (possibly empty).
3859    pub(crate) fn push(&mut self, mut nodes: Vec<Node>) -> Vec<Node> {
3860        self.pending.append(&mut nodes);
3861        merge_continuations(&mut self.pending);
3862        let cut = hold_start(&self.pending);
3863        let tail = self.pending.split_off(cut);
3864        std::mem::replace(&mut self.pending, tail)
3865    }
3866
3867    /// Flush whatever is left after the last page (the held tail is final once no
3868    /// more pages can follow).
3869    pub(crate) fn finish(self) -> Vec<Node> {
3870        self.pending
3871    }
3872}
3873
3874#[cfg(test)]
3875mod tests {
3876    use super::{cells_text, clean_text, merge_overlapping_regulars, reclaim_heading_body_footers};
3877
3878    /// docling drops a picture covering > 90 % of the page (its labels then
3879    /// read out as text); a dominant-but-not-full figure and any other label
3880    /// stay whatever their size.
3881    #[test]
3882    fn full_page_pictures_are_dropped_like_docling() {
3883        use super::drop_full_page_pictures;
3884        use crate::layout::Region;
3885        let region = |label: &'static str, l, t, r, b| Region {
3886            label,
3887            score: 0.99,
3888            l,
3889            t,
3890            r,
3891            b,
3892        };
3893        let mut regions = vec![
3894            region("picture", 0.0, 0.5, 478.9, 241.8),
3895            region("picture", 10.0, 10.0, 400.0, 200.0),
3896            region("table", 0.0, 0.0, 480.0, 243.0),
3897            region("text", 5.0, 5.0, 100.0, 20.0),
3898        ];
3899        drop_full_page_pictures(&mut regions, 480.75, 243.75);
3900        let labels: Vec<_> = regions.iter().map(|r| (r.label, r.l)).collect();
3901        assert_eq!(
3902            labels,
3903            vec![("picture", 10.0), ("table", 0.0), ("text", 5.0)]
3904        );
3905    }
3906    use super::{code_region_text, merge_continuations, resolve_link_anchors, StreamAssembler};
3907    use crate::layout::Region;
3908    use crate::pdfium_backend::{LinkAnnot, PdfPage, TextCell};
3909    use docling_core::Node;
3910
3911    /// The int8-layout guard's coverage metric: cells under detections count,
3912    /// cells outside don't, whitespace cells are ignored, and a cell-less page
3913    /// reads as fully covered (nothing to rescue).
3914    #[test]
3915    fn layout_cell_coverage_counts_claimed_text_cells() {
3916        let cell = |text: &str, l: f32, t: f32| TextCell {
3917            text: text.into(),
3918            l,
3919            t,
3920            r: l + 40.0,
3921            b: t + 10.0,
3922        };
3923        let region = Region {
3924            label: "text",
3925            score: 0.9,
3926            l: 0.0,
3927            t: 0.0,
3928            r: 100.0,
3929            b: 50.0,
3930        };
3931        let cells = vec![
3932            cell("inside", 10.0, 10.0),
3933            cell("also inside", 10.0, 30.0),
3934            cell("outside", 10.0, 200.0),
3935            cell("   ", 10.0, 210.0), // whitespace: not counted at all
3936        ];
3937        let cov = super::layout_cell_coverage(std::slice::from_ref(&region), &cells);
3938        assert!((cov - 2.0 / 3.0).abs() < 1e-6, "got {cov}");
3939        assert_eq!(super::layout_cell_coverage(&[], &[]), 1.0);
3940        assert_eq!(super::layout_cell_coverage(&[], &cells), 0.0);
3941    }
3942
3943    /// #165: a picture no longer claims cells at 0.2 intersection-over-self.
3944    /// A line straddling the figure border (≤80 % contained) becomes an orphan
3945    /// region and is emitted as page text — before the fix its cells were
3946    /// silently erased. A line fully inside the picture is the picture's child
3947    /// (docling's `_set_cluster_children`): it survives the containment drop,
3948    /// leaves the page's reading order, and is written only under the picture
3949    /// in the JSON — never in the Markdown, like docling's picture serializer.
3950    #[test]
3951    fn border_straddlers_are_page_text_picture_interior_is_a_picture_child() {
3952        let pic = Region {
3953            label: "picture",
3954            score: 0.9,
3955            l: 0.0,
3956            t: 0.0,
3957            r: 100.0,
3958            b: 100.0,
3959        };
3960        // ~35 % of this cell overlaps the picture (l=90..120 of 0..100): above
3961        // the old 0.2 claim (was swallowed), below full containment (survives).
3962        let straddler = TextCell {
3963            text: "axis label".into(),
3964            l: 90.0,
3965            t: 40.0,
3966            r: 120.0,
3967            b: 48.0,
3968        };
3969        let interior = TextCell {
3970            text: "in-figure callout".into(),
3971            l: 10.0,
3972            t: 10.0,
3973            r: 60.0,
3974            b: 18.0,
3975        };
3976        let cells = vec![straddler, interior];
3977        let mut regions = vec![pic];
3978        super::add_orphan_regions(&mut regions, &cells);
3979        super::drop_contained_regulars(&mut regions);
3980        assert_eq!(
3981            regions.iter().filter(|r| r.label == "text").count(),
3982            2,
3983            "both unclaimed lines become orphans, and a picture swallows neither"
3984        );
3985        let parents = super::picture_parents(&regions);
3986        let parent_of = |l: f32| {
3987            regions
3988                .iter()
3989                .zip(&parents)
3990                .find(|(r, _)| r.label == "text" && r.l == l)
3991                .and_then(|(_, p)| *p)
3992        };
3993        assert_eq!(
3994            parent_of(10.0),
3995            Some(0),
3996            "the callout is the picture's child"
3997        );
3998        assert_eq!(parent_of(90.0), None, "the straddler is a page element");
3999
4000        let page = PdfPage::from_cells(200.0, 200.0, 2.0, cells);
4001        let n = regions.len();
4002        let (nodes, _) = super::assemble_page(&page, regions, &vec![None; n], &vec![None; n], None);
4003        let children: Vec<&Node> = nodes
4004            .iter()
4005            .filter_map(|n| match n {
4006                Node::PictureChildren(c) => Some(c),
4007                _ => None,
4008            })
4009            .flatten()
4010            .collect();
4011        assert!(
4012            matches!(children.as_slice(), [Node::Located { inner, .. }]
4013                if matches!(inner.as_ref(), Node::Paragraph { text } if text == "in-figure callout")),
4014            "{children:?}"
4015        );
4016        let mut doc = docling_core::DoclingDocument::new("t");
4017        doc.nodes = nodes;
4018        let md = doc.export_to_markdown();
4019        assert!(md.contains("axis label"), "{md}");
4020        assert!(!md.contains("in-figure callout"), "{md}");
4021        let json = doc.export_to_json_value();
4022        let pic = &json["pictures"][0];
4023        let child = pic["children"][0]["$ref"].as_str().expect("a child ref");
4024        let idx: usize = child.rsplit('/').next().unwrap().parse().unwrap();
4025        assert_eq!(json["texts"][idx]["text"], "in-figure callout");
4026        assert_eq!(json["texts"][idx]["parent"]["$ref"], "#/pictures/0");
4027        assert_eq!(json["texts"][idx]["content_layer"], "body");
4028    }
4029
4030    /// docling#3906's concern, pinned on our side: a picture detected fully
4031    /// inside a table region must survive the containment drop (upstream now
4032    /// attaches it to the table's cell; we keep it as a body sibling — either
4033    /// way it must not vanish). The text region inside the same table is the
4034    /// control: regulars are the ones the drop swallows.
4035    #[test]
4036    fn picture_inside_a_table_region_survives_the_containment_drop() {
4037        let mut regions = vec![
4038            region("table", 0.9, 0.0, 0.0, 200.0, 200.0),
4039            region("picture", 0.9, 20.0, 20.0, 120.0, 120.0),
4040            region("text", 0.9, 20.0, 140.0, 180.0, 180.0),
4041        ];
4042        super::drop_contained_regulars(&mut regions);
4043        let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
4044        assert_eq!(
4045            labels,
4046            ["table", "picture"],
4047            "the in-table picture stays; the in-table regular is the special's child"
4048        );
4049    }
4050
4051    /// Table–caption pairing (#265) is reading-order adjacency, docling's
4052    /// `_find_to_captions`: a caption binds the table directly next to it in
4053    /// the region sequence — above-caption and below-caption both work, and
4054    /// geometry is irrelevant (a same-page caption in the other column of a
4055    /// two-column layout is *not* adjacent, however close its box is). A
4056    /// caption with media on both sides, or separated from the table by a
4057    /// text paragraph, stays unattached.
4058    #[test]
4059    fn table_captions_pair_by_reading_order_adjacency() {
4060        // caption → table (above-caption), then table → caption (below-caption),
4061        // then a caption fenced off by a paragraph, then one between two tables.
4062        let regions = vec![
4063            region("text", 0.9, 0.0, 0.0, 100.0, 10.0), // 0 body text
4064            region("caption", 0.9, 0.0, 12.0, 60.0, 20.0), // 1 above-caption
4065            region("table", 0.9, 20.0, 22.0, 90.0, 60.0), // 2 ← pairs with 1
4066            region("table", 0.9, 0.0, 70.0, 100.0, 110.0), // 3 ← pairs with 4
4067            region("caption", 0.9, 0.0, 112.0, 60.0, 120.0), // 4 below-caption
4068            region("text", 0.9, 0.0, 130.0, 100.0, 140.0), // 5 body text
4069            region("caption", 0.9, 0.0, 142.0, 60.0, 150.0), // 6 fenced by 5/7
4070            region("text", 0.9, 0.0, 152.0, 100.0, 162.0), // 7 body text
4071            region("table", 0.9, 0.0, 170.0, 100.0, 200.0), // 8 unpaired
4072            region("caption", 0.9, 0.0, 202.0, 60.0, 210.0), // 9 ambiguous
4073            region("table", 0.9, 0.0, 212.0, 100.0, 240.0), // 10 unpaired
4074        ];
4075        let mut taken = vec![false; regions.len()];
4076        let pairs = super::pair_table_captions(&regions, &mut taken);
4077        assert_eq!(pairs[2], Some(1), "caption directly above its table pairs");
4078        assert_eq!(pairs[3], Some(4), "caption directly below its table pairs");
4079        assert_eq!(
4080            pairs[8], None,
4081            "a text paragraph between caption and table breaks the bond"
4082        );
4083        assert_eq!(
4084            pairs[10], None,
4085            "a caption between two tables is ambiguous and stays loose"
4086        );
4087        assert!(taken[1] && taken[4] && !taken[6] && !taken[9]);
4088    }
4089
4090    /// A colored terms-and-conditions panel detected as `picture` demotes into
4091    /// per-paragraph `text` regions (the blank line between C.7 and C.8 splits
4092    /// them); a chart whose only text is a few narrow axis labels keeps its
4093    /// crop untouched.
4094    #[test]
4095    fn text_panels_demote_to_paragraphs_but_charts_keep_their_crop() {
4096        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
4097            text: text.to_string(),
4098            l,
4099            t,
4100            r,
4101            b,
4102        };
4103        let panel = Region {
4104            label: "picture",
4105            score: 0.9,
4106            l: 0.0,
4107            t: 0.0,
4108            r: 100.0,
4109            b: 100.0,
4110        };
4111        // Three tight lines, a blank-line gap, two more: two paragraphs.
4112        let cells = vec![
4113            cell(
4114                "C.7. Wenn Sie diesen Vertrag widerrufen,",
4115                5.0,
4116                10.0,
4117                95.0,
4118                18.0,
4119            ),
4120            cell(
4121                "haben wir Ihnen alle Zahlungen, die wir",
4122                5.0,
4123                20.0,
4124                95.0,
4125                28.0,
4126            ),
4127            cell(
4128                "von Ihnen erhalten haben, zurückzuzahlen.",
4129                5.0,
4130                30.0,
4131                90.0,
4132                38.0,
4133            ),
4134            cell(
4135                "C.8. Wir können die Rückzahlung verweigern,",
4136                5.0,
4137                52.0,
4138                95.0,
4139                60.0,
4140            ),
4141            cell(
4142                "bis wir die Waren wieder zurückerhalten haben.",
4143                5.0,
4144                62.0,
4145                92.0,
4146                70.0,
4147            ),
4148        ];
4149        let mut regions = vec![panel.clone()];
4150        super::recover_text_panels(&mut regions, &cells);
4151        assert_eq!(
4152            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
4153            ["text", "text"],
4154            "dense panel must demote into one text region per paragraph"
4155        );
4156        assert!(regions[0].b < regions[1].t, "paragraphs split at the gap");
4157        // Sparse narrow labels (a chart): picture survives.
4158        let labels = vec![
4159            cell("0", 5.0, 90.0, 8.0, 95.0),
4160            cell("50", 5.0, 50.0, 10.0, 55.0),
4161            cell("100", 5.0, 10.0, 12.0, 15.0),
4162            cell("t, s", 45.0, 96.0, 55.0, 100.0),
4163        ];
4164        let mut regions = vec![panel];
4165        super::recover_text_panels(&mut regions, &labels);
4166        assert_eq!(
4167            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
4168            ["picture"]
4169        );
4170    }
4171
4172    /// An uncaptioned chart on a scanned page whose title, axis labels, and
4173    /// OCR boxes over the plot area are dense and wide enough to pass the
4174    /// coverage/width gates still keeps its crop: its line heights are ragged
4175    /// (title face vs tick labels vs bar-area OCR), failing the uniform-leading
4176    /// gate — a real text panel is set with constant leading (#173).
4177    #[test]
4178    fn dense_titled_chart_keeps_its_crop() {
4179        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
4180            text: text.to_string(),
4181            l,
4182            t,
4183            r,
4184            b,
4185        };
4186        let chart = Region {
4187            label: "picture",
4188            score: 0.9,
4189            l: 0.0,
4190            t: 0.0,
4191            r: 100.0,
4192            b: 100.0,
4193        };
4194        // Five wide lines at wildly different heights: a 12-pt title, 20-pt OCR
4195        // boxes over the bars, 4–5-pt tick/axis labels. Coverage and median
4196        // width both clear the panel thresholds.
4197        let cells = vec![
4198            cell("Underground Water Storage", 10.0, 5.0, 90.0, 17.0),
4199            cell("aquifer recharge zone", 15.0, 30.0, 75.0, 50.0),
4200            cell("confined | unconfined | perched", 12.0, 55.0, 80.0, 59.0),
4201            cell("saturated thickness", 8.0, 70.0, 60.0, 90.0),
4202            cell("distance from well, km", 20.0, 92.0, 85.0, 97.0),
4203        ];
4204        let mut regions = vec![chart];
4205        super::recover_text_panels(&mut regions, &cells);
4206        assert_eq!(
4207            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
4208            ["picture"],
4209            "ragged line heights mark a figure, not a text panel"
4210        );
4211    }
4212
4213    /// docling serializes a cluster's cells in docling-parse index order
4214    /// (`_sort_cells`) and joins them with `PageAssembleModel.sanitize_text`:
4215    /// a space after every line except one ending in `-`, which either fuses a
4216    /// wrapped word (alnum on both sides — dash dropped) or glues verbatim (a
4217    /// bare `-` cell: `[0000` `-` `0002` → `[0000 -0002`, the 2305 ORCID line;
4218    /// `-` + `"C" cell -` + `a new table cell` → `-"C" cell a new table cell`,
4219    /// its OTSL list). Verified against the corpus: pure index order beats any
4220    /// geometric re-sort (normal_4pages' heading numerals paint after their
4221    /// text and belong last: `## 들어가며 1`).
4222    #[test]
4223    fn cells_join_in_index_order_with_sanitize_text_rules() {
4224        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
4225            text: text.to_string(),
4226            l,
4227            t,
4228            r,
4229            b,
4230        };
4231        let region = Region {
4232            label: "text",
4233            score: 1.0,
4234            l: 0.0,
4235            t: 95.0,
4236            r: 200.0,
4237            b: 130.0,
4238        };
4239        // ORCID superscript: a bare dash cell is a *detached* dash — kept, and
4240        // since docling#4052 (2.122) it joins with the ordinary space on both
4241        // sides (`[0000 -0002 -6960]` before that fix).
4242        let orcid = vec![
4243            cell("[0000", 10.0, 100.0, 30.0, 110.0),
4244            cell("−", 30.0, 100.0, 34.0, 110.0),
4245            cell("0002", 34.0, 100.0, 50.0, 110.0),
4246            cell("−", 50.0, 100.0, 54.0, 110.0),
4247            cell("6960]", 54.0, 100.0, 70.0, 110.0),
4248        ];
4249        assert_eq!(super::region_text(&region, &orcid), "[0000 - 0002 - 6960]");
4250        // Wrapped word: dash dropped, lines fused (both boundary words alnum).
4251        let wrapped = vec![
4252            cell("platforms-", 10.0, 100.0, 60.0, 110.0),
4253            cell("reflects the design", 10.0, 112.0, 90.0, 122.0),
4254        ];
4255        assert_eq!(
4256            super::region_text(&region, &wrapped),
4257            "platformsreflects the design"
4258        );
4259        // Dash-ending lines that are *detached* dashes (a bare bullet cell, a
4260        // `cell -` separator): the dash stays and the lines join with a space
4261        // — docling#4052; before it they glued (`-"C" cell a new table cell`,
4262        // 2305's OTSL list bullets).
4263        let otsl = vec![
4264            cell("–", 10.0, 100.0, 14.0, 110.0),
4265            cell("\"C\" cell -", 16.0, 100.0, 60.0, 110.0),
4266            cell("a new table cell", 10.0, 112.0, 80.0, 122.0),
4267        ];
4268        assert_eq!(
4269            super::region_text(&region, &otsl),
4270            "- \"C\" cell - a new table cell"
4271        );
4272        // Index order is authoritative — no geometric re-sort.
4273        let numeral = vec![
4274            cell("들어가며", 30.0, 100.0, 80.0, 110.0),
4275            cell("1", 10.0, 98.0, 25.0, 112.0), // big numeral painted last
4276        ];
4277        assert_eq!(super::region_text(&region, &numeral), "들어가며 1");
4278    }
4279
4280    /// The geometric-reliability gate, on the two shapes it has to tell apart.
4281    #[test]
4282    fn geometric_reliability_rejects_split_column_grids() {
4283        let g = |rows: &[&[&str]]| -> Vec<Vec<String>> {
4284            rows.iter()
4285                .map(|r| r.iter().map(|c| c.to_string()).collect())
4286                .collect()
4287        };
4288        // A genuine grid: dense, every column carrying entries. Nothing for
4289        // TableFormer to improve, so geometry is used as-is.
4290        assert!(super::geometric_table_is_reliable(&g(&[
4291            &["Datum", "Leistung", "Anzahl", "Kosten"],
4292            &["04.07", "Internet", "1", "40.30"],
4293            &["04.07", "Telefon", "2", "8.06"],
4294        ])));
4295        // The left-edge split artefact (the shape a scanned invoice produced):
4296        // one real label column plus values scattered across three sparse ones.
4297        assert!(!super::geometric_table_is_reliable(&g(&[
4298            &["www.magenta.at/faq", "", "", ""],
4299            &["Serviceteam", "", "", ""],
4300            &["Telefon", "0676/2000", "", ""],
4301            &["Kundennummer", "", "", "1.21699482"],
4302            &["Rechnungsnummer", "", "922769430725", ""],
4303            &["Rechnungsdatum", "", "", "04.07.2025"],
4304        ])));
4305        // A column only one row ever uses is a split artefact even when the
4306        // grid is otherwise dense.
4307        assert!(!super::geometric_table_is_reliable(&g(&[
4308            &["a", "b", ""],
4309            &["c", "d", ""],
4310            &["e", "f", "g"],
4311        ])));
4312        // Degenerate shapes are never vouched for — TableFormer may recover
4313        // structure a collapsed reconstruction lost.
4314        assert!(!super::geometric_table_is_reliable(&g(&[&[
4315            "only one column"
4316        ]])));
4317        assert!(!super::geometric_table_is_reliable(&[]));
4318    }
4319
4320    /// A `picture` region is cropped out of the rendered page, whatever built
4321    /// that page. The browser pipeline (#157) has no pdfium but does hand over
4322    /// the rasterized bitmap through `from_cells_with_image`, so it must get
4323    /// the same figure bytes the native path does — that is what makes
4324    /// `images = "embedded"` inline real pixels instead of a placeholder.
4325    #[cfg(feature = "ocr-prep")]
4326    #[test]
4327    fn picture_regions_are_cropped_from_a_host_supplied_page_image() {
4328        let mut img = image::RgbImage::new(200, 200);
4329        // Paint the figure area so the crop is distinguishable from the page.
4330        for y in 100..160 {
4331            for x in 20..120 {
4332                img.put_pixel(x, y, image::Rgb([255, 0, 0]));
4333            }
4334        }
4335        // scale 2.0: the region is in page points, the bitmap in pixels.
4336        let page = PdfPage::from_cells_with_image(100.0, 100.0, 2.0, Vec::new(), img);
4337        let region = Region {
4338            label: "picture",
4339            score: 0.9,
4340            l: 10.0,
4341            t: 50.0,
4342            r: 60.0,
4343            b: 80.0,
4344        };
4345        let (nodes, _) = super::assemble_page(&page, vec![region], &[None], &[None], None);
4346        // Layout-derived nodes carry provenance, so the picture arrives wrapped.
4347        let image = nodes
4348            .iter()
4349            .find_map(|n| match n {
4350                Node::Located { inner, .. } => match &**inner {
4351                    Node::Picture { image, .. } => image.as_ref(),
4352                    _ => None,
4353                },
4354                Node::Picture { image, .. } => image.as_ref(),
4355                _ => None,
4356            })
4357            .expect("a picture node with cropped pixels");
4358        assert_eq!(image.mimetype, "image/png");
4359        assert_eq!((image.width, image.height), (100, 60), "region × scale");
4360        assert!(!image.data.is_empty(), "PNG bytes were encoded");
4361    }
4362
4363    #[test]
4364    fn link_anchors_split_a_shared_word_cell_between_adjacent_links() {
4365        // A common header layout: one text run holds several pipe-separated
4366        // labels, each carrying its own link annotation. Every link must get
4367        // its own label as the anchor (and the "|" separators must belong to
4368        // none), not the whole run.
4369        let annot = |l: f32, r: f32, uri: &str| LinkAnnot {
4370            l,
4371            t: 100.0,
4372            r,
4373            b: 114.0,
4374            uri: uri.into(),
4375        };
4376        let page = PdfPage {
4377            width: 600.0,
4378            height: 800.0,
4379            scale: 2.0,
4380            cells: Vec::new(),
4381            code_cells: Vec::new(),
4382            checkboxes: Vec::new(),
4383            // "LinkedIn | GitHub | Credly" = 26 chars over x 100..360.
4384            word_cells: vec![cell(
4385                "LinkedIn | GitHub | Credly",
4386                100.0,
4387                100.0,
4388                360.0,
4389                114.0,
4390            )],
4391            image: image::RgbImage::new(1, 1),
4392            image_layout: None,
4393            links: vec![
4394                annot(100.0, 180.0, "https://l"),
4395                annot(200.0, 260.0, "https://g"),
4396                annot(290.0, 360.0, "https://c"),
4397            ],
4398            rotation: 0,
4399        };
4400        assert_eq!(
4401            resolve_link_anchors(&page),
4402            vec![
4403                ("LinkedIn".to_string(), "https://l".to_string()),
4404                ("GitHub".to_string(), "https://g".to_string()),
4405                ("Credly".to_string(), "https://c".to_string()),
4406            ]
4407        );
4408    }
4409
4410    /// A one-line code cell at `[l, r] × [t, b]` (top-left coords).
4411    fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
4412        TextCell {
4413            text: text.into(),
4414            l,
4415            t,
4416            r,
4417            b,
4418        }
4419    }
4420
4421    /// OCR-path grouping (docling's `_remove_overlapping_clusters("regular")`):
4422    /// the low-score paragraph box RT-DETR draws over its own high-score line
4423    /// boxes collapses to one region — the group's union, with the survivor's
4424    /// label and score — so region-scoped OCR reads each line once. Regions
4425    /// that merely sit near each other, and specials, are untouched.
4426    #[test]
4427    fn merge_overlapping_regulars_collapses_a_block_over_its_lines() {
4428        let mut regions = vec![
4429            region("text", 0.84, 60.0, 186.0, 270.0, 198.0),
4430            region("text", 0.80, 60.0, 160.0, 294.0, 172.0),
4431            region("text", 0.79, 59.0, 107.0, 272.0, 119.0),
4432            // The paragraph box, lower score, containing all three lines.
4433            region("text", 0.52, 59.0, 107.0, 295.0, 200.0),
4434            // Elsewhere on the page: stays as is.
4435            region("section_header", 0.77, 60.0, 71.0, 253.0, 86.0),
4436            // A picture the block overlaps is not a regular — never grouped.
4437            region("picture", 0.9, 50.0, 100.0, 300.0, 210.0),
4438        ];
4439        merge_overlapping_regulars(&mut regions);
4440        assert_eq!(regions.len(), 3, "{regions:?}");
4441        let block = regions
4442            .iter()
4443            .find(|r| r.label == "text")
4444            .expect("one text");
4445        // docling keeps the largest passing candidate unless a rival is both
4446        // comparable in size and > 0.05 more confident; the 16× larger block
4447        // passes, and a smaller line never replaces a larger current best.
4448        // Either way the survivor spans the whole group.
4449        assert_eq!(
4450            (block.l, block.t, block.r, block.b),
4451            (59.0, 107.0, 295.0, 200.0)
4452        );
4453        assert!(regions.iter().any(|r| r.label == "section_header"));
4454        assert!(regions.iter().any(|r| r.label == "picture"));
4455    }
4456
4457    /// The pairwise rules, each in the arrangement where it decides the
4458    /// outcome: docling seeds the survivor with the group's first passing
4459    /// cluster and a later one replaces it only when larger *and* within
4460    /// 0.05 confidence, so a rule that merely lets a cluster pass matters
4461    /// exactly when that cluster comes first — a same-sized list item ahead
4462    /// of a far more confident text box, a code box ahead of the text it
4463    /// contains. Without the rule either would be rejected outright (similar
4464    /// size, rival > 0.05 more confident) and the text box would win.
4465    #[test]
4466    fn merge_overlapping_regulars_follows_the_preference_rules() {
4467        let mut regions = vec![
4468            region("list_item", 0.6, 0.0, 0.0, 102.0, 20.0),
4469            region("text", 0.9, 0.0, 0.0, 100.0, 20.0),
4470        ];
4471        merge_overlapping_regulars(&mut regions);
4472        assert_eq!(regions.len(), 1);
4473        assert_eq!(regions[0].label, "list_item");
4474
4475        let mut regions = vec![
4476            region("code", 0.6, 0.0, 0.0, 100.0, 100.0),
4477            region("text", 0.9, 2.0, 2.0, 98.0, 98.0),
4478        ];
4479        merge_overlapping_regulars(&mut regions);
4480        assert_eq!(regions.len(), 1);
4481        assert_eq!(regions[0].label, "code");
4482
4483        // No rule applies: a near-identical rival that is > 0.05 more
4484        // confident rejects the candidate whatever the order.
4485        for order in [[0.9, 0.6], [0.6, 0.9]] {
4486            let mut regions = vec![
4487                region("text", order[0], 0.0, 0.0, 100.0, 20.0),
4488                region("text", order[1], 0.0, 0.0, 105.0, 21.0),
4489            ];
4490            merge_overlapping_regulars(&mut regions);
4491            assert_eq!(regions.len(), 1);
4492            assert_eq!(regions[0].score, 0.9, "the confident twin wins");
4493            assert_eq!(
4494                (regions[0].r, regions[0].b),
4495                (105.0, 21.0),
4496                "on the union box"
4497            );
4498        }
4499
4500        // Side by side (no containment, IoU 0): nothing to merge.
4501        let mut regions = vec![
4502            region("text", 0.9, 0.0, 0.0, 100.0, 20.0),
4503            region("text", 0.9, 0.0, 22.0, 100.0, 42.0),
4504        ];
4505        merge_overlapping_regulars(&mut regions);
4506        assert_eq!(regions.len(), 2);
4507    }
4508
4509    #[test]
4510    fn footer_under_a_body_less_heading_is_its_text() {
4511        // The reporting CV's last page: `## Languages` at t=773.7..783.4 and
4512        // its one-line body at 798.5..808.7, labelled page_footer (A4, 595 pt).
4513        let mut regions = vec![
4514            region("list_item", 0.9, 66.0, 754.0, 353.0, 765.0),
4515            region("section_header", 0.94, 43.0, 773.7, 89.5, 783.4),
4516            region("page_footer", 0.88, 43.0, 798.5, 527.8, 808.7),
4517        ];
4518        reclaim_heading_body_footers(&mut regions, 595.28);
4519        assert_eq!(regions[2].label, "text");
4520        assert_eq!(regions[1].label, "section_header");
4521
4522        // A heading with its own paragraph and a running footer below: kept.
4523        let mut regions = vec![
4524            region("section_header", 0.9, 43.0, 700.0, 120.0, 710.0),
4525            region("text", 0.9, 43.0, 714.0, 520.0, 780.0),
4526            region("page_footer", 0.9, 43.0, 798.0, 520.0, 808.0),
4527        ];
4528        reclaim_heading_body_footers(&mut regions, 595.28);
4529        assert_eq!(regions[2].label, "page_footer");
4530
4531        // A page number under a trailing heading is too narrow to be a body.
4532        let mut regions = vec![
4533            region("section_header", 0.9, 43.0, 773.0, 120.0, 783.0),
4534            region("page_footer", 0.9, 280.0, 798.0, 300.0, 808.0),
4535        ];
4536        reclaim_heading_body_footers(&mut regions, 595.28);
4537        assert_eq!(regions[1].label, "page_footer");
4538
4539        // Too far below the heading (a real footer after a heading that ends
4540        // the page): kept.
4541        let mut regions = vec![
4542            region("section_header", 0.9, 43.0, 700.0, 120.0, 710.0),
4543            region("page_footer", 0.9, 43.0, 798.0, 520.0, 808.0),
4544        ];
4545        reclaim_heading_body_footers(&mut regions, 595.28);
4546        assert_eq!(regions[1].label, "page_footer");
4547    }
4548
4549    fn region(label: &'static str, score: f32, l: f32, t: f32, r: f32, b: f32) -> Region {
4550        Region {
4551            label,
4552            score,
4553            l,
4554            t,
4555            r,
4556            b,
4557        }
4558    }
4559
4560    #[test]
4561    fn resolve_collapses_nested_code_keeping_the_larger_box() {
4562        // A tight high-score `code` box and a taller lower-score near-duplicate that
4563        // contains it must collapse to one — the *larger* box, so every cell stays
4564        // covered and nothing leaks out as orphan text.
4565        let tight = region("code", 0.95, 78.0, 292.0, 300.0, 330.0);
4566        let wide = region("code", 0.66, 63.0, 260.0, 320.0, 346.0);
4567        let kept = super::resolve(vec![tight, wide]);
4568        assert_eq!(kept.len(), 1, "nested code boxes must collapse to one");
4569        assert!(
4570            kept[0].l == 63.0 && kept[0].b == 346.0,
4571            "the larger containing box is kept"
4572        );
4573    }
4574
4575    #[test]
4576    fn resolve_keeps_distinct_and_differently_typed_regions() {
4577        // A text box fully inside a lower-score *table* must NOT be collapsed (the
4578        // code dedup is code-only), and two separate code blocks stay separate.
4579        let text = region("text", 0.95, 90.0, 210.0, 200.0, 230.0);
4580        let table = region("table", 0.60, 80.0, 200.0, 400.0, 500.0);
4581        assert_eq!(super::resolve(vec![text, table]).len(), 2);
4582
4583        let code_a = region("code", 0.9, 78.0, 100.0, 300.0, 140.0);
4584        let code_b = region("code", 0.9, 78.0, 300.0, 300.0, 360.0); // far below, no overlap
4585        assert_eq!(super::resolve(vec![code_a, code_b]).len(), 2);
4586    }
4587
4588    /// A two-column glossary page came out as three column
4589    /// tables *and* one low-score whole-page table over them. docling's wrapper
4590    /// `_remove_overlapping_clusters` keeps one table per overlapping group
4591    /// (here the whole-page one: > 2× every rival's area and ≤ 0.2 less
4592    /// confident than the running best); `greedy` alone kept all four and
4593    /// emitted every cell twice.
4594    #[test]
4595    fn resolve_keeps_one_table_per_nested_group() {
4596        let kept = super::resolve(vec![
4597            region("table", 0.71, 26.0, 203.0, 183.0, 558.0),
4598            region("table", 0.67, 26.0, 55.0, 183.0, 196.0),
4599            region("table", 0.66, 196.0, 56.0, 354.0, 561.0),
4600            region("table", 0.53, 25.0, 53.0, 354.0, 561.0),
4601        ]);
4602        assert_eq!(kept.len(), 1, "one survivor per overlapping group");
4603        assert_eq!((kept[0].l, kept[0].b), (25.0, 561.0));
4604        // Side-by-side tables that don't overlap stay separate.
4605        let kept = super::resolve(vec![
4606            region("table", 0.9, 26.0, 55.0, 183.0, 558.0),
4607            region("table", 0.9, 196.0, 56.0, 354.0, 561.0),
4608        ]);
4609        assert_eq!(kept.len(), 2);
4610    }
4611
4612    /// A dense data table detected as both picture (0.80) and table (0.62) on one box.
4613    /// The picture is ≥ 0.1 more confident, so `_handle_cross_type_overlaps`
4614    /// keeps both, and the dense table text passed the text-panel gates: the
4615    /// demoted paragraph repeated every cell the table grid renders. A
4616    /// paragraph > 80 % inside a surviving table is the table's child and is
4617    /// not emitted; a panel with no table under it still demotes.
4618    #[test]
4619    fn text_panel_over_a_table_does_not_repeat_its_cells() {
4620        let lines = |t0: f32| -> Vec<TextCell> {
4621            (0..4)
4622                .map(|i| {
4623                    let t = t0 + 10.0 * i as f32;
4624                    cell("1 2 3 4 5 6 7 8 9 10 11 12", 5.0, t, 95.0, t + 8.0)
4625                })
4626                .collect()
4627        };
4628        let mut cells = lines(0.0);
4629        cells.extend(lines(200.0));
4630        let mut regions = vec![
4631            region("picture", 0.80, 0.0, 0.0, 100.0, 45.0),
4632            region("table", 0.62, 0.0, 0.0, 100.0, 45.0),
4633            region("picture", 0.80, 0.0, 200.0, 100.0, 245.0),
4634        ];
4635        super::recover_text_panels(&mut regions, &cells);
4636        assert_eq!(
4637            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
4638            ["table", "text"]
4639        );
4640        assert_eq!(regions[1].t, 200.0, "the table-free panel still demotes");
4641    }
4642
4643    /// docling's `ListItemMarkerProcessor` + docling-core's Markdown list rules
4644    /// (see `list_item_node`): a bullet glyph or `-` is split off and the
4645    /// serializer's own `-` printed (2305's OTSL list reads `- "C" cell`, not
4646    /// `- - "C" cell`); `N.` is an ordered item; any other marker with a letter
4647    /// or digit stays in the text behind the bullet; no marker → plain bullet.
4648    #[test]
4649    fn list_item_markers_split_like_docling() {
4650        let text_of = |n: &Node| match n {
4651            Node::ListItem {
4652                ordered,
4653                number,
4654                text,
4655                marker,
4656                ..
4657            } => (*ordered, *number, text.clone(), marker.clone()),
4658            other => panic!("{other:?}"),
4659        };
4660        let loc = [0, 0, 100, 10];
4661        assert_eq!(
4662            text_of(&super::list_item_node(
4663                "- \"C\" cell - a new table cell",
4664                loc,
4665                false
4666            )),
4667            (
4668                false,
4669                0,
4670                "\"C\" cell - a new table cell".into(),
4671                Some("-".into())
4672            )
4673        );
4674        assert_eq!(
4675            text_of(&super::list_item_node("• Bullet text", loc, false)),
4676            (false, 0, "Bullet text".into(), Some("•".into()))
4677        );
4678        assert_eq!(
4679            text_of(&super::list_item_node("3. Third step", loc, false)),
4680            (true, 3, "Third step".into(), Some("3.".into()))
4681        );
4682        assert_eq!(
4683            text_of(&super::list_item_node("a) Option", loc, false)),
4684            (false, 0, "a) Option".into(), Some("a)".into()))
4685        );
4686        assert_eq!(
4687            text_of(&super::list_item_node("1.2 Nested outline", loc, false)),
4688            (false, 0, "1.2 Nested outline".into(), Some("1.2".into()))
4689        );
4690        // 2203's `3.a. If all IOU scores…`: a compound marker is not an item
4691        // number — docling prints `- 3.a. If all…`.
4692        assert_eq!(
4693            text_of(&super::list_item_node("3.a. If all IOU scores", loc, false)),
4694            (
4695                false,
4696                0,
4697                "3.a. If all IOU scores".into(),
4698                Some("3.a.".into())
4699            )
4700        );
4701        // A glued symbol-font bullet is stripped, a spaced one is the marker.
4702        assert_eq!(
4703            text_of(&super::list_item_node("•Glued", loc, false)),
4704            (false, 0, "Glued".into(), Some("·".into()))
4705        );
4706        // No whitespace after the glyph → not a marker (docling's `\s` is required).
4707        assert_eq!(
4708            text_of(&super::list_item_node("-5 degrees", loc, false)),
4709            (false, 0, "-5 degrees".into(), Some("·".into()))
4710        );
4711        // The remaining numbered shapes, first-wins like docling's list.
4712        for (input, marker, body) in [
4713            ("1.2.3. Deep", "1.2.3.", "Deep"),
4714            ("9a) Nine-a", "9a)", "Nine-a"),
4715            ("(3.a) Paren", "(3.a)", "Paren"),
4716            ("12) Twelve", "12)", "Twelve"),
4717            ("(4) Four", "(4)", "Four"),
4718            ("[7] Seven", "[7]", "Seven"),
4719            ("iv. Roman", "iv.", "Roman"),
4720            ("IX. Roman", "IX.", "Roman"),
4721            ("b. Letter", "b.", "Letter"),
4722            ("B) Letter", "B)", "Letter"),
4723        ] {
4724            assert_eq!(
4725                super::split_list_marker(input),
4726                Some((marker, body, true)),
4727                "{input}"
4728            );
4729        }
4730        // A `1.2.` whose optional dot would eat the separator backtracks like
4731        // Python's regex; a marker with nothing after the whitespace is none.
4732        assert_eq!(
4733            super::split_list_marker("1.2.\tx"),
4734            Some(("1.2.", "x", true))
4735        );
4736        assert_eq!(super::split_list_marker("1. "), None);
4737        assert_eq!(super::split_list_marker("• "), None);
4738        assert_eq!(
4739            text_of(&super::list_item_node("Plain item", loc, false)),
4740            (false, 0, "Plain item".into(), Some("·".into()))
4741        );
4742    }
4743
4744    #[test]
4745    fn code_language_label_above_code_is_detected() {
4746        // A bare "XML" token directly above a code box is a language label; a real
4747        // heading above the same code is not; a language word with no code below is
4748        // left alone.
4749        let label = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
4750        let code = region("code", 0.7, 77.0, 552.0, 290.0, 640.0);
4751        let heading = region("section_header", 0.9, 76.0, 500.0, 260.0, 512.0);
4752        let cells = vec![
4753            cell("XML", 78.0, 541.0, 94.0, 548.0),       // inside `label`
4754            cell("Overview", 78.0, 501.0, 250.0, 511.0), // inside `heading`
4755        ];
4756        let drop = super::code_language_labels(&[label, code, heading], &cells);
4757        assert_eq!(drop, vec![true, false, false], "only the label is consumed");
4758
4759        // Same label with no code region present → not consumed.
4760        let label2 = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
4761        let only = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
4762        assert_eq!(super::code_language_labels(&[label2], &only), vec![false]);
4763
4764        // A label swallowed into the top of a wider code box (negative gap) is still
4765        // recognized.
4766        let inside_lbl = region("text", 0.9, 76.0, 540.0, 96.0, 549.0);
4767        let wide_code = region("code", 0.7, 63.0, 531.0, 320.0, 654.0);
4768        let cells2 = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
4769        assert_eq!(
4770            super::code_language_labels(&[inside_lbl, wide_code], &cells2),
4771            vec![true, false]
4772        );
4773
4774        assert!(super::is_code_language("XML") && super::is_code_language("c#"));
4775        assert!(!super::is_code_language("Configure") && !super::is_code_language("XML schema"));
4776    }
4777
4778    #[test]
4779    fn code_region_text_keeps_lines_and_indentation() {
4780        // Three source lines; each glyph is 6 units wide (width / chars = 6), so the
4781        // `int X;` line indented to x=22 is (22-10)/6 = 2 spaces in.
4782        let region = Region {
4783            label: "code",
4784            score: 1.0,
4785            l: 0.0,
4786            t: -5.0,
4787            r: 100.0,
4788            b: 40.0,
4789        };
4790        let cells = vec![
4791            cell("struct P {", 10.0, 0.0, 70.0, 10.0),
4792            cell("int X;", 22.0, 12.0, 58.0, 22.0),
4793            cell("}", 10.0, 24.0, 16.0, 34.0),
4794        ];
4795        assert_eq!(code_region_text(&region, &cells), "struct P {\n  int X;\n}");
4796    }
4797
4798    #[test]
4799    fn code_region_text_tightens_punctuation_without_eating_indentation() {
4800        // A fluent `.Foo()` line at x=22 (2 chars in). Per-line tightening must not
4801        // consume the leading indent space by matching " ." across it.
4802        let region = Region {
4803            label: "code",
4804            score: 1.0,
4805            l: 0.0,
4806            t: -5.0,
4807            r: 100.0,
4808            b: 40.0,
4809        };
4810        let cells = vec![
4811            cell("builder", 10.0, 0.0, 52.0, 10.0),
4812            // pdfium spaced the call: ".Foo (x)" tightens to ".Foo(x)", still 2-indented.
4813            cell(".Foo (x)", 22.0, 12.0, 70.0, 22.0),
4814        ];
4815        assert_eq!(code_region_text(&region, &cells), "builder\n  .Foo(x)");
4816    }
4817
4818    #[test]
4819    fn code_region_text_orders_out_of_order_cells_and_ignores_blank_lines() {
4820        let region = Region {
4821            label: "code",
4822            score: 1.0,
4823            l: 0.0,
4824            t: -5.0,
4825            r: 100.0,
4826            b: 60.0,
4827        };
4828        // Fed bottom-up and with a whitespace-only cell; output is top-down, no blank.
4829        let cells = vec![
4830            cell("b();", 10.0, 24.0, 34.0, 34.0),
4831            cell("   ", 10.0, 12.0, 20.0, 22.0),
4832            cell("a();", 10.0, 0.0, 34.0, 10.0),
4833        ];
4834        assert_eq!(code_region_text(&region, &cells), "a();\nb();");
4835        // No code cells → empty, so the caller falls back to the prose text.
4836        assert_eq!(code_region_text(&region, &[]), "");
4837    }
4838
4839    fn para(text: &str) -> Node {
4840        Node::Paragraph { text: text.into() }
4841    }
4842
4843    /// Run a node sequence through [`StreamAssembler`] with the given page splits
4844    /// and assert the flushed result equals one-shot [`merge_continuations`].
4845    fn assert_stream_eq(nodes: &[Node], splits: &[usize]) {
4846        let mut want = nodes.to_vec();
4847        merge_continuations(&mut want);
4848
4849        let mut asm = StreamAssembler::new();
4850        let mut got = Vec::new();
4851        let mut start = 0;
4852        for &end in splits {
4853            got.extend(asm.push(nodes[start..end].to_vec()));
4854            start = end;
4855        }
4856        got.extend(asm.push(nodes[start..].to_vec()));
4857        got.extend(asm.finish());
4858        assert_eq!(got, want, "stream assembly diverged (splits={splits:?})");
4859    }
4860
4861    #[test]
4862    fn stream_assembler_matches_merge_continuations() {
4863        // Open fragment + lowercase continuation split across a page boundary.
4864        let cross = [para("the definition of"), para("lists in scope")];
4865        assert_stream_eq(&cross, &[1]);
4866        assert_stream_eq(&cross, &[]);
4867
4868        // Continuation that wraps around a figure (+ its caption) on the boundary.
4869        let wrap = [
4870            para("the wing type that is"),
4871            Node::Picture {
4872                caption: None,
4873                caption_href: None,
4874                image: None,
4875                classification: None,
4876                caption_parent: Default::default(),
4877                caption_location: None,
4878            },
4879            para("Fig. 1. a diagram"),
4880            para("the most common kind"),
4881        ];
4882        for splits in [&[][..], &[1][..], &[2][..], &[3][..], &[1, 3][..]] {
4883            assert_stream_eq(&wrap, splits);
4884        }
4885
4886        // A heading between fragments blocks the merge (must still flush correctly).
4887        let blocked = [
4888            para("ends mid word and"),
4889            Node::Heading {
4890                level: 2,
4891                text: "New Section".into(),
4892            },
4893            para("more body here"),
4894        ];
4895        for splits in [&[][..], &[1][..], &[2][..]] {
4896            assert_stream_eq(&blocked, splits);
4897        }
4898
4899        // A chain across three pages: each page is one open lowercase fragment.
4900        let chain = [
4901            para("alpha beta"),
4902            para("gamma delta"),
4903            para("epsilon zeta"),
4904        ];
4905        assert_stream_eq(&chain, &[1, 2]);
4906    }
4907
4908    #[test]
4909    fn clean_text_dehyphenates_and_normalizes_typography() {
4910        // U+0002 line-wrap hyphen + the join space → merged word (like docling).
4911        assert_eq!(clean_text("com\u{2} pact"), "compact");
4912        assert_eq!(clean_text("end-to\u{2} end deep"), "end-toend deep");
4913        // A stray wrap hyphen (no following join) is dropped.
4914        assert_eq!(clean_text("word\u{2}"), "word");
4915        // Typographic punctuation → ASCII: every curly quote becomes `'`
4916        // (docling-parse's sanitizer table), a literal `"` stays.
4917        assert_eq!(
4918            clean_text("Graph\u{2019}s \u{201c}x\u{201d} \"y\""),
4919            "Graph's 'x' \"y\""
4920        );
4921        assert_eq!(clean_text("a\u{2026}"), "a...");
4922        // The docling-parse sanitizer's internal spacing is preserved as
4923        // placed; line breaks/tabs normalize to a space, ends trim.
4924        assert_eq!(clean_text("a   b\nc"), "a   b c");
4925    }
4926
4927    /// docling#4064: a form's children are emitted together where the form
4928    /// sits in the top-level order, not interleaved with surrounding text.
4929    #[test]
4930    fn form_children_stay_together_in_reading_order() {
4931        let reg = |label: &'static str, l: f32, t: f32, r: f32, b: f32| Region {
4932            label,
4933            score: 0.9,
4934            l,
4935            t,
4936            r,
4937            b,
4938        };
4939        // Page: intro text, then a form spanning the left column with two
4940        // fields and a table inside, while a right-column paragraph sits
4941        // level with the form's first field (it would otherwise be read
4942        // between the form's children).
4943        let mut items = vec![
4944            reg("text", 50.0, 50.0, 550.0, 70.0),    // 0 intro
4945            reg("form", 50.0, 100.0, 300.0, 400.0),  // 1 container
4946            reg("text", 60.0, 110.0, 290.0, 130.0),  // 2 field A (child)
4947            reg("text", 320.0, 110.0, 550.0, 130.0), // 3 right column paragraph
4948            reg("table", 60.0, 150.0, 290.0, 300.0), // 4 table (child)
4949            reg("text", 60.0, 320.0, 290.0, 340.0),  // 5 field B (child)
4950            reg("text", 50.0, 450.0, 550.0, 470.0),  // 6 outro
4951        ];
4952        let cids = super::cluster_cids(&items, &[]);
4953        super::order_with_containers(&mut items, &cids, 600.0, 800.0, |r| r);
4954        let order: Vec<(&str, f32)> = items.iter().map(|r| (r.label, r.t)).collect();
4955        // The form block (container, then its children top-down) is one unit.
4956        let form_pos = order.iter().position(|(l, _)| *l == "form").unwrap();
4957        assert_eq!(
4958            &order[form_pos..form_pos + 4],
4959            &[
4960                ("form", 100.0),
4961                ("text", 110.0),
4962                ("table", 150.0),
4963                ("text", 320.0)
4964            ]
4965        );
4966        assert_eq!(order[0], ("text", 50.0));
4967        assert_eq!(order[order.len() - 1], ("text", 450.0));
4968        // Without a container the plain order interleaves by geometry.
4969        let mut flat: Vec<Region> = items
4970            .iter()
4971            .filter(|r| r.label != "form")
4972            .cloned()
4973            .collect();
4974        let cids = super::cluster_cids(&flat, &[]);
4975        super::order_regions(&mut flat, &cids, 600.0, 800.0, |r| r);
4976        assert_ne!(
4977            flat.iter().map(|r| r.t).collect::<Vec<_>>(),
4978            order
4979                .iter()
4980                .filter(|(l, _)| *l != "form")
4981                .map(|(_, t)| *t)
4982                .collect::<Vec<_>>()
4983        );
4984    }
4985
4986    /// docling#3906: a picture inside a table lands in the covering cell,
4987    /// chosen by the picture's inferred grid position when cell boxes overlap.
4988    #[test]
4989    fn picture_matches_the_cell_at_its_grid_position() {
4990        let cell = |r: usize, c: usize, bbox: [f32; 4]| docling_core::TableCell {
4991            text: format!("r{r}c{c}"),
4992            bbox: Some(bbox),
4993            start_row: r,
4994            start_col: c,
4995            row_span: 1,
4996            col_span: 1,
4997            column_header: false,
4998            row_header: false,
4999            row_section: false,
5000        };
5001        // 2×2 grid; the (1,0) cell box is generous and also covers the picture.
5002        let cells = vec![
5003            cell(0, 0, [0.0, 0.0, 100.0, 50.0]),
5004            cell(0, 1, [100.0, 0.0, 200.0, 50.0]),
5005            cell(1, 0, [0.0, 50.0, 100.0, 100.0]),
5006            cell(1, 1, [100.0, 50.0, 200.0, 100.0]),
5007        ];
5008        let pic = Region {
5009            label: "picture",
5010            score: 0.9,
5011            l: 110.0,
5012            t: 60.0,
5013            r: 190.0,
5014            b: 95.0,
5015        };
5016        assert_eq!(super::match_picture_to_cell(&pic, &cells), Some((1.0, 3)));
5017        // A picture only half inside any cell is not nested.
5018        let straddling = Region {
5019            label: "picture",
5020            score: 0.9,
5021            l: 60.0,
5022            t: 60.0,
5023            r: 160.0,
5024            b: 95.0,
5025        };
5026        assert_eq!(super::match_picture_to_cell(&straddling, &cells), None);
5027    }
5028
5029    /// docling#4052 (2.122): a line-final dash fuses the wrapped word only
5030    /// when attached to it; a detached dash is a literal and the lines join
5031    /// with a space.
5032    #[test]
5033    fn line_final_hyphen_fuses_only_when_attached_to_a_word() {
5034        let line = |text: &str, t: f32| TextCell {
5035            text: text.to_string(),
5036            l: 0.0,
5037            t,
5038            r: 100.0,
5039            b: t + 10.0,
5040        };
5041        // `algo-` / `rithms`: attached hyphen, alnum on both sides → fused.
5042        assert_eq!(
5043            cells_text(vec![&line("algo-", 0.0), &line("rithms", 12.0)]),
5044            "algorithms"
5045        );
5046        // `pp. 545-` / `561`: attached, digits count as alnum → `545561` (upstream).
5047        assert_eq!(
5048            cells_text(vec![&line("pp. 545-", 0.0), &line("561", 12.0)]),
5049            "pp. 545561"
5050        );
5051        // A dash after whitespace — a separator or a lone `-` cell — is kept and
5052        // the lines take the ordinary joining space.
5053        assert_eq!(
5054            cells_text(vec![&line("range -", 0.0), &line("wide", 12.0)]),
5055            "range - wide"
5056        );
5057        assert_eq!(
5058            cells_text(vec![&line("-", 0.0), &line("item", 12.0)]),
5059            "- item"
5060        );
5061        // Attached but the next line opens with no word (`x-` / `...`): dash
5062        // kept and, as before, no separating space.
5063        assert_eq!(
5064            cells_text(vec![&line("x-", 0.0), &line("...", 12.0)]),
5065            "x-..."
5066        );
5067    }
5068
5069    #[test]
5070    fn lam_alef_only_swaps_a_genuinely_reversed_ligature() {
5071        // A mid-word `alef-variant + lam` is pdfium's reversed lam-alef ligature and
5072        // is swapped back to logical `lam + alef-variant` (`ب أ ل` → `ب ل أ`).
5073        assert_eq!(
5074            clean_text("\u{0628}\u{0623}\u{0644}"),
5075            "\u{0628}\u{0644}\u{0623}"
5076        );
5077        // But when the alef-variant is *already* preceded by a lam it is the logical
5078        // ligature `لآ`; the following lam is the next syllable's letter and must not
5079        // move. `التعلم الآلي` must stay `الآلي`, not become `اللآي`.
5080        assert_eq!(
5081            clean_text("\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"),
5082            "\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"
5083        );
5084    }
5085
5086    /// The #419 page, in points: three layout boxes over one paragraph, two of
5087    /// them ending partway through a line. The sliced lines miss the 0.2 claim
5088    /// and become orphans; the third model box starts above the second orphan,
5089    /// so unfitted the reading order emits that box first and strands the line.
5090    fn sliced_paragraph() -> (Vec<Region>, Vec<TextCell>) {
5091        let line = |text: &str, t: f32, r: f32| cell(text, 60.0, t, r, t + 11.0);
5092        let cells = vec![
5093            line("The mission of this series is to improve", 135.0, 458.0),
5094            line("The books in this series are technical,", 147.0, 458.0),
5095            line("substantial. The authors are", 159.0, 458.0),
5096            line("highly experienced craftsmen and", 171.5, 458.0), // sliced: 1.5/11 under box A
5097            line("actually works in practice, as opposed", 185.0, 458.0),
5098            line("about what the author has done, not", 197.0, 458.0),
5099            line("about programming, there will be lots", 210.5, 458.0), // sliced: 1.5/11 under box B
5100            line("will be lots of case studies from real", 223.0, 206.0), // C's line
5101        ];
5102        let regions = vec![
5103            region("text", 0.9, 60.0, 132.0, 458.0, 173.0), // A: three lines + a sliver of the 4th
5104            region("text", 0.9, 60.0, 184.0, 458.0, 212.0), // B: two lines + a sliver of the 7th
5105            region("text", 0.9, 60.0, 216.0, 206.0, 227.0), // C: last line, box opening 5.5pt too early
5106        ];
5107        (regions, cells)
5108    }
5109
5110    fn ordered_texts(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
5111        let mut items: Vec<Region> = regions.to_vec();
5112        let cids = super::cluster_cids(&items, cells);
5113        super::order_regions(&mut items, &cids, 500.0, 700.0, |r| r);
5114        super::region_texts_exclusive(&items, cells)
5115            .into_iter()
5116            .map(|t| t.chars().take(9).collect())
5117            .collect()
5118    }
5119
5120    /// #419: fitted to its cells, a model box that cut a line in half no longer
5121    /// overlaps the orphan that line became, so the orphan orders where it
5122    /// reads; unfitted, the same page strands the line after the paragraph.
5123    #[test]
5124    fn fitting_boxes_to_cells_puts_a_sliced_line_back_in_order() {
5125        let (mut regions, cells) = sliced_paragraph();
5126        super::add_orphan_regions(&mut regions, &cells);
5127        assert_eq!(regions.len(), 5, "two orphan lines");
5128        // The defect, for the record: C (top 216) is not strictly below the
5129        // orphan at 210.5–221.5, so the graph orders C first.
5130        assert_eq!(
5131            ordered_texts(&regions, &cells).last().map(String::as_str),
5132            Some("about pro")
5133        );
5134
5135        super::fit_regions_to_cells(&mut regions, &cells);
5136        assert_eq!(regions.len(), 5);
5137        // A ends on its last claimed line, C starts on its only one.
5138        assert_eq!((regions[0].t, regions[0].b), (135.0, 170.0));
5139        assert_eq!((regions[2].t, regions[2].b), (223.0, 234.0));
5140        assert_eq!(
5141            ordered_texts(&regions, &cells),
5142            [
5143                "The missi",
5144                "highly ex",
5145                "actually ",
5146                "about pro",
5147                "will be l"
5148            ]
5149        );
5150    }
5151
5152    /// An orphan the fitted paragraph box surrounds (a short middle line the
5153    /// narrow model box missed while claiming the lines around it) is folded
5154    /// into the paragraph; an empty regular box goes away, a formula stays, a
5155    /// picture is never refitted, and a page with no cells is left untouched.
5156    #[test]
5157    fn fitting_folds_surrounded_orphans_and_drops_empty_regulars() {
5158        let wide = |text: &str, t: f32| cell(text, 60.0, t, 400.0, t + 11.0);
5159        let cells = vec![
5160            wide("first line of the paragraph", 100.0),
5161            cell("stray", 250.0, 112.0, 400.0, 123.0), // clear of the narrow box
5162            wide("third line of the paragraph", 124.0),
5163        ];
5164        let mut regions = vec![
5165            // Narrow box: claims the wide lines at 0.41, misses the short one.
5166            region("text", 0.9, 60.0, 98.0, 200.0, 136.0),
5167            region("section_header", 0.8, 60.0, 300.0, 200.0, 320.0), // no cells
5168            region("formula", 0.8, 60.0, 340.0, 200.0, 360.0),        // no cells, kept
5169            region("picture", 0.8, 0.0, 400.0, 500.0, 600.0),
5170        ];
5171        super::add_orphan_regions(&mut regions, &cells);
5172        assert_eq!(regions.len(), 5, "the short line became an orphan");
5173        super::fit_regions_to_cells(&mut regions, &cells);
5174        let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
5175        assert_eq!(labels, ["text", "formula", "picture"]);
5176        let para = &regions[0];
5177        assert_eq!(
5178            (para.l, para.t, para.r, para.b),
5179            (60.0, 100.0, 400.0, 135.0)
5180        );
5181        assert_eq!(
5182            super::region_texts_exclusive(&regions, &cells)[0],
5183            "first line of the paragraph stray third line of the paragraph"
5184        );
5185        assert_eq!(
5186            (regions[2].t, regions[2].b),
5187            (400.0, 600.0),
5188            "picture untouched"
5189        );
5190
5191        let mut untouched = vec![region("text", 0.9, 0.0, 0.0, 10.0, 10.0)];
5192        super::fit_regions_to_cells(&mut untouched, &[]);
5193        assert_eq!(untouched.len(), 1, "no cells yet: nothing dropped");
5194    }
5195
5196    /// #609: one `text` region over a checklist whose lines each have a drawn
5197    /// square in front splits into a checkbox region per boxed line — the
5198    /// wrapped second line of an option stays with it, the intro line above
5199    /// the first square keeps the region's label, and a ticked square gives
5200    /// `checkbox_selected`. Without squares nothing changes.
5201    #[test]
5202    fn checklist_lines_with_squares_split_into_checkbox_regions() {
5203        use crate::checkbox::CheckBox;
5204        let cells = vec![
5205            cell("Pick any:", 100.0, 90.0, 160.0, 100.0),
5206            cell("First option", 120.0, 110.0, 180.0, 120.0),
5207            cell("Second option that", 120.0, 129.0, 200.0, 139.0),
5208            cell("wraps onto a line", 120.0, 141.0, 196.0, 151.0),
5209            cell("Third option", 120.0, 160.0, 182.0, 170.0),
5210        ];
5211        let sq = |t: f32, checked| CheckBox {
5212            l: 100.0,
5213            t,
5214            r: 112.0,
5215            b: t + 12.0,
5216            checked,
5217        };
5218        let boxes = [sq(109.0, false), sq(128.0, true), sq(159.0, false)];
5219        let block = || vec![region("text", 0.6, 100.0, 90.0, 200.0, 170.0)];
5220
5221        let mut regions = block();
5222        super::split_checkbox_lines(&mut regions, &cells, &boxes);
5223        let texts = super::region_texts_exclusive(&regions, &cells);
5224        let got: Vec<(&str, &str)> = regions
5225            .iter()
5226            .zip(&texts)
5227            .map(|(r, t)| (r.label, t.as_str()))
5228            .collect();
5229        assert_eq!(
5230            got,
5231            [
5232                ("text", "Pick any:"),
5233                ("checkbox_unselected", "First option"),
5234                ("checkbox_selected", "Second option that wraps onto a line"),
5235                ("checkbox_unselected", "Third option"),
5236            ]
5237        );
5238        // A checkbox region spans its square and its line(s).
5239        assert_eq!((regions[2].l, regions[2].t), (100.0, 128.0));
5240        assert_eq!((regions[2].r, regions[2].b), (200.0, 151.0));
5241
5242        let shape = |rs: &[Region]| -> Vec<(&'static str, [f32; 4])> {
5243            rs.iter().map(|r| (r.label, [r.l, r.t, r.r, r.b])).collect()
5244        };
5245        let mut untouched = block();
5246        super::split_checkbox_lines(&mut untouched, &cells, &[]);
5247        assert_eq!(shape(&untouched), shape(&block()));
5248        // A square far left of the text (a margin mark), or one the text
5249        // sits inside (a comb field), does not make a checkbox.
5250        for far in [
5251            CheckBox {
5252                l: 40.0,
5253                r: 52.0,
5254                ..sq(109.0, false)
5255            },
5256            CheckBox {
5257                l: 115.0,
5258                r: 127.0,
5259                ..sq(109.0, false)
5260            },
5261        ] {
5262            let mut regions = block();
5263            super::split_checkbox_lines(&mut regions, &cells, &[far]);
5264            assert_eq!(shape(&regions), shape(&block()));
5265        }
5266    }
5267
5268    /// #609: a ballot-box glyph opening a line marks a checkbox item like a
5269    /// drawn square — `☐` unchecked, `☒` checked, the glyph stripped from the
5270    /// label at emission — while a line holding two boxes (`☐ Yes ☐ No`)
5271    /// stays text of its own.
5272    #[test]
5273    fn checklist_lines_opening_with_a_ballot_box_split_into_checkbox_regions() {
5274        let cells = vec![
5275            cell("Options:", 100.0, 90.0, 150.0, 100.0),
5276            cell("\u{2610} Tea", 100.0, 110.0, 140.0, 120.0),
5277            cell("\u{2612} Coffee", 100.0, 130.0, 150.0, 140.0),
5278            cell("\u{2610} Yes \u{2610} No", 100.0, 150.0, 170.0, 160.0),
5279        ];
5280        let mut regions = vec![region("text", 0.6, 100.0, 90.0, 170.0, 160.0)];
5281        super::split_checkbox_lines(&mut regions, &cells, &[]);
5282        let texts = super::region_texts_exclusive(&regions, &cells);
5283        let got: Vec<(&str, &str)> = regions
5284            .iter()
5285            .zip(&texts)
5286            .map(|(r, t)| match r.label {
5287                "text" => (r.label, t.as_str()),
5288                _ => (r.label, super::strip_checkbox_glyph(t)),
5289            })
5290            .collect();
5291        assert_eq!(
5292            got,
5293            [
5294                ("text", "Options:"),
5295                ("checkbox_unselected", "Tea"),
5296                ("checkbox_selected", "Coffee"),
5297                // Two boxes on one line: its own text, the item closed.
5298                ("text", "\u{2610} Yes \u{2610} No"),
5299            ]
5300        );
5301        assert_eq!(super::strip_checkbox_glyph("  \u{2611}  Done"), "Done");
5302        assert_eq!(super::strip_checkbox_glyph("No box"), "No box");
5303    }
5304}