Skip to main content

docling_pdf/
assemble.rs

1//! Layout-driven assembly: map detected [`Region`]s + text cells to a
2//! [`DoclingDocument`], mirroring docling's page-assembly + reading-order.
3//!
4//! Overlapping detections are resolved greedily by score, each text cell is
5//! assigned to its best-containing region, regions are ordered in reading order
6//! (two-column aware), and each becomes a typed node by its layout label.
7
8use docling_core::{CaptionParent, Node, PictureClass, PictureImage, Table};
9#[cfg(feature = "ml")]
10use image::RgbImage;
11
12use crate::layout::Region;
13use crate::pdfium_backend::{PdfPage, TextCell};
14
15fn area(l: f32, t: f32, r: f32, b: f32) -> f32 {
16    ((r - l).max(0.0)) * ((b - t).max(0.0))
17}
18
19/// Intersection area of two boxes.
20fn inter(a: &Region, l: f32, t: f32, r: f32, b: f32) -> f32 {
21    let il = a.l.max(l);
22    let it = a.t.max(t);
23    let ir = a.r.min(r);
24    let ib = a.b.min(b);
25    area(il, it, ir, ib)
26}
27
28/// Wrapper (structured-region) labels, ported from docling
29/// `LayoutPostprocessor.WRAPPER_TYPES`: a region that *contains* other regions
30/// and renders as a structured block (a table / table-of-contents index), not as
31/// its own flat text.
32fn is_wrapper(label: &str) -> bool {
33    matches!(
34        label,
35        "table" | "document_index" | "form" | "key_value_region"
36    )
37}
38
39/// Labels docling's table-structure (TableFormer) model runs on and that render
40/// as a Markdown table: a plain `table` and a `document_index` (a table of
41/// contents), which docling assembles as a `TableItem` too.
42pub fn is_table_like(label: &str) -> bool {
43    matches!(label, "table" | "document_index")
44}
45
46/// Greedily keep regions by descending score, dropping a region that is mostly
47/// covered by an already-kept one (RT-DETR emits overlapping duplicates).
48fn greedy(mut regions: Vec<Region>) -> Vec<Region> {
49    regions.sort_by(|a, b| b.score.total_cmp(&a.score));
50    let mut kept: Vec<Region> = Vec::new();
51    for r in regions {
52        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
53        let covered = kept.iter().any(|k| {
54            let i = inter(&r, k.l, k.t, k.r, k.b);
55            let ka = area(k.l, k.t, k.r, k.b).max(1.0);
56            // drop if most of r is inside k, or they strongly mutually overlap
57            i / ra > 0.7 || i / (ra + ka - i) > 0.5
58        });
59        if !covered {
60            kept.push(r);
61        }
62    }
63    kept
64}
65
66/// Resolve overlapping RT-DETR detections, ported from the bucket structure of
67/// docling's `LayoutPostprocessor`: regular, picture and wrapper clusters live in
68/// **separate** spatial indexes and are de-overlapped independently, so a
69/// high-score picture never suppresses a lower-score table or table-of-contents
70/// index (the redp5110 TOC that was otherwise replaced by a picture box). A
71/// cross-type pass first drops a picture that nearly coincides with a table
72/// (`_handle_cross_type_overlaps`), keeping the structured table.
73/// docling's `_remove_overlapping_clusters("picture")`: same-label picture
74/// detections whose boxes heavily overlap (IoU > 0.8, or either box > 80 %
75/// contained in the other) form one group, and a single survivor is kept per
76/// group. Survivor selection ports `_should_prefer_cluster` /
77/// `_select_best_cluster_from_group` with the picture params
78/// (`area_threshold` 2.0, `conf_threshold` 0.3): a candidate is rejected only
79/// when a rival is both comparable in size (candidate ≤ 2× its area) and
80/// clearly more confident (> 0.3); among the survivors the *larger* box wins
81/// unless it is > 0.3 less confident. Net effect on the corpus: a figure the
82/// detector proposes both whole and as its sub-panels (2206's four-thumbnail
83/// Figure 1) collapses to the whole-figure box, exactly like docling.
84pub(crate) fn dedup_pictures(regions: &mut Vec<Region>) {
85    let idx: Vec<usize> = (0..regions.len())
86        .filter(|&i| regions[i].label == "picture")
87        .collect();
88    if idx.len() < 2 {
89        return;
90    }
91    // Union-find over the picture subset.
92    let mut parent: Vec<usize> = (0..idx.len()).collect();
93    fn find(parent: &mut [usize], i: usize) -> usize {
94        let mut root = i;
95        while parent[root] != root {
96            root = parent[root];
97        }
98        let mut cur = i;
99        while parent[cur] != root {
100            let next = parent[cur];
101            parent[cur] = root;
102            cur = next;
103        }
104        root
105    }
106    let boxed = |r: &Region| (r.l, r.t, r.r, r.b);
107    for a in 0..idx.len() {
108        for b in (a + 1)..idx.len() {
109            let (ra, rb) = (&regions[idx[a]], &regions[idx[b]]);
110            let (al, at, ar, ab_) = boxed(ra);
111            let (bl, bt, br, bb) = boxed(rb);
112            let ix = (ar.min(br) - al.max(bl)).max(0.0);
113            let iy = (ab_.min(bb) - at.max(bt)).max(0.0);
114            let inter = ix * iy;
115            let aa = area(al, at, ar, ab_).max(f32::EPSILON);
116            let ba = area(bl, bt, br, bb).max(f32::EPSILON);
117            let iou = inter / (aa + ba - inter).max(f32::EPSILON);
118            if iou > 0.8 || inter / aa > 0.8 || inter / ba > 0.8 {
119                let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
120                if pa != pb {
121                    parent[pa] = pb;
122                }
123            }
124        }
125    }
126    // Per group, run docling's pairwise preference + larger-wins selection.
127    let mut groups: std::collections::HashMap<usize, Vec<usize>> = std::collections::HashMap::new();
128    for i in 0..idx.len() {
129        let root = find(&mut parent, i);
130        groups.entry(root).or_default().push(i);
131    }
132    let mut drop = vec![false; regions.len()];
133    for group in groups.values() {
134        if group.len() < 2 {
135            continue;
136        }
137        const AREA_THRESHOLD: f32 = 2.0;
138        const CONF_THRESHOLD: f32 = 0.3;
139        let area_of = |i: usize| {
140            let r = &regions[idx[i]];
141            area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
142        };
143        let mut best: Option<usize> = None;
144        for &cand in group {
145            let passes = group.iter().all(|&other| {
146                if other == cand {
147                    return true;
148                }
149                let area_ratio = area_of(cand) / area_of(other);
150                let conf_diff = regions[idx[other]].score - regions[idx[cand]].score;
151                !(area_ratio <= AREA_THRESHOLD && conf_diff > CONF_THRESHOLD)
152            });
153            if passes {
154                best = Some(match best {
155                    None => cand,
156                    Some(cur) => {
157                        if area_of(cand) > area_of(cur)
158                            && regions[idx[cur]].score - regions[idx[cand]].score <= CONF_THRESHOLD
159                        {
160                            cand
161                        } else {
162                            cur
163                        }
164                    }
165                });
166            }
167        }
168        // Every candidate rejected can't happen with docling's rule (rejection
169        // needs a strictly better rival); guard with highest score anyway.
170        let keep = best.unwrap_or_else(|| {
171            *group
172                .iter()
173                .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(&regions[idx[b]].score))
174                .expect("non-empty group")
175        });
176        for &i in group {
177            if i != keep {
178                drop[idx[i]] = true;
179            }
180        }
181    }
182    let mut keep_iter = drop.into_iter();
183    regions.retain(|_| !keep_iter.next().expect("aligned"));
184}
185
186/// `intersection_over_union` of two regions.
187fn iou(a: &Region, b: &Region) -> f32 {
188    let i = inter(a, b.l, b.t, b.r, b.b);
189    let u = area(a.l, a.t, a.r, a.b) + area(b.l, b.t, b.r, b.b) - i;
190    if u > 0.0 {
191        i / u
192    } else {
193        0.0
194    }
195}
196
197/// docling's `_resolve_coincident_pairs` (#4059, 2.122): for every (loser,
198/// winner) pair at a near-identical box (IoU > 0.8) whose confidences are
199/// within 0.1 (`loser.score - winner.score < 0.1`), the loser label is dropped
200/// so the label with the richer downstream semantic survives. Nothing else —
201/// containment, area — is considered; a clearly more confident loser stays.
202fn coincident_losers(regions: &[Region], losers: &[usize], winners: &[usize]) -> Vec<usize> {
203    let mut out = Vec::new();
204    for &li in losers {
205        for &wi in winners {
206            if iou(&regions[li], &regions[wi]) > 0.8 && regions[li].score - regions[wi].score < 0.1
207            {
208                out.push(li);
209                break;
210            }
211        }
212    }
213    out
214}
215
216/// docling's `_handle_cross_type_overlaps` (2.122/2.123 shape): the layout
217/// model can emit one grounded region under several labels, and the picture /
218/// table / container buckets are de-overlapped independently, so such a region
219/// survives twice. Elect a winner for the near-identical pairs:
220///
221/// | pair                                  | loser     | winner              |
222/// |---------------------------------------|-----------|---------------------|
223/// | TABLE vs DOCUMENT_INDEX               | table     | document_index      |
224/// | PICTURE vs TABLE / DOCUMENT_INDEX     | picture   | the table-like      |
225/// | FORM / KEY_VALUE_REGION vs *surviving* TABLE / DOCUMENT_INDEX / PICTURE | container | structured element |
226///
227/// IoU (not containment) so a genuine small figure inside a large table region
228/// is not removed; the confidence tolerance keeps a clearly more confident
229/// loser (an earlier port dropped every coincident picture regardless).
230fn handle_cross_type_overlaps(regions: Vec<Region>) -> Vec<Region> {
231    let by = |pred: &dyn Fn(&str) -> bool| -> Vec<usize> {
232        (0..regions.len())
233            .filter(|&i| pred(regions[i].label))
234            .collect()
235    };
236    let tables = by(&|l| l == "table");
237    let doc_indices = by(&|l| l == "document_index");
238    let pictures = by(&|l| l == "picture");
239    let containers = by(&|l| matches!(l, "form" | "key_value_region"));
240    let mut drop = vec![false; regions.len()];
241    for i in coincident_losers(&regions, &tables, &doc_indices) {
242        drop[i] = true;
243    }
244    let table_like: Vec<usize> = tables.iter().chain(&doc_indices).copied().collect();
245    for i in coincident_losers(&regions, &pictures, &table_like) {
246        drop[i] = true;
247    }
248    let structured: Vec<usize> = table_like
249        .iter()
250        .chain(&pictures)
251        .copied()
252        .filter(|&i| !drop[i])
253        .collect();
254    for i in coincident_losers(&regions, &containers, &structured) {
255        drop[i] = true;
256    }
257    let mut drop = drop.into_iter();
258    let mut regions = regions;
259    regions.retain(|_| !drop.next().expect("aligned"));
260    regions
261}
262
263pub fn resolve(regions: Vec<Region>) -> Vec<Region> {
264    let regions = handle_cross_type_overlaps(regions);
265    // De-overlap each bucket on its own.
266    let pictures = greedy(
267        regions
268            .iter()
269            .filter(|r| r.label == "picture")
270            .cloned()
271            .collect(),
272    );
273    // Tables and containers are separate buckets since docling 2.123
274    // (`TABLE_TYPES` vs `CONTAINER_TYPES`, docling#4064): a form drawn around a
275    // table no longer competes with it for survival — the table nests inside
276    // the container instead (`order_with_containers`).
277    let tables = greedy(
278        regions
279            .iter()
280            .filter(|r| is_table_like(r.label))
281            .cloned()
282            .collect(),
283    );
284    let containers = greedy(
285        regions
286            .iter()
287            .filter(|r| matches!(r.label, "form" | "key_value_region"))
288            .cloned()
289            .collect(),
290    );
291    let mut kept = greedy(
292        regions
293            .iter()
294            .filter(|r| r.label != "picture" && !is_wrapper(r.label))
295            .cloned()
296            .collect(),
297    );
298    dedup_nested_code(&mut kept);
299    kept.extend(pictures);
300    kept.extend(tables);
301    kept.extend(containers);
302    kept
303}
304
305/// Drop a regular region that is >80% contained in a surviving special region we
306/// render **as a single unit** — a table/table-of-contents index — ported from
307/// docling's "Remove regular clusters that are included in wrappers" step: the
308/// special absorbs it as a child (a table cell), so it must not also be emitted
309/// as its own paragraph/list-item. This stops the survey list-items from
310/// appearing both inside the detected table and again as bullets
311/// (`table_mislabeled_as_picture`).
312///
313/// `picture` regions stay in the swallow set even after #165: docling keeps a
314/// picture's contained clusters as the `PictureItem`'s *children* in the
315/// document JSON (`ReadingOrderModel._add_child_elements`), but its
316/// `MarkdownPictureSerializer` prints only the caption and the image — the
317/// children never reach the Markdown (verified against the corpus groundtruth:
318/// `amt_handbook`'s in-figure callout labels are absent). Dropping the
319/// fully-contained regulars here reproduces exactly that. What #165 *does*
320/// change is upstream, in [`add_orphan_regions`]: pictures no longer claim
321/// cells, so a line only partially under a figure box (straddling its border,
322/// ≤80 % contained) now forms an orphan region that survives this drop — those
323/// words were silently erased before, and docling emits them.
324///
325/// `form` / `key_value_region` wrappers are deliberately **excluded**: this
326/// pipeline does not render them as a structured block (they are skipped), so
327/// their textual content comes precisely from the contained regular regions —
328/// dropping those would erase the page (e.g. `right_to_left_03`'s form-heavy
329/// pages). Runs *after* [`drop_false_pictures`] so a phantom picture can't
330/// swallow real text on its way out.
331pub fn drop_contained_regulars(regions: &mut Vec<Region>) {
332    let specials: Vec<(f32, f32, f32, f32)> = regions
333        .iter()
334        .filter(|r| r.label == "picture" || is_table_like(r.label))
335        .map(|r| (r.l, r.t, r.r, r.b))
336        .collect();
337    if specials.is_empty() {
338        return;
339    }
340    regions.retain(|r| {
341        if r.label == "picture" || is_wrapper(r.label) {
342            return true;
343        }
344        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
345        !specials
346            .iter()
347            .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.8)
348    });
349}
350
351/// True for a bare, single-token source-code language label (`XML`, `C#`, `JSON`,
352/// `bash`, …) — the little header the docs render above a code block. Matched
353/// case-insensitively; anything with whitespace or longer than a token is out.
354fn is_code_language(t: &str) -> bool {
355    let t = t.trim();
356    if t.is_empty() || t.chars().any(char::is_whitespace) || t.chars().count() > 12 {
357        return false;
358    }
359    const LANGS: &[&str] = &[
360        "xml",
361        "html",
362        "xhtml",
363        "json",
364        "jsonc",
365        "yaml",
366        "yml",
367        "toml",
368        "ini",
369        "c#",
370        "csharp",
371        "f#",
372        "fsharp",
373        "vb",
374        "c",
375        "c++",
376        "cpp",
377        "java",
378        "kotlin",
379        "scala",
380        "go",
381        "golang",
382        "rust",
383        "swift",
384        "javascript",
385        "js",
386        "typescript",
387        "ts",
388        "jsx",
389        "tsx",
390        "python",
391        "py",
392        "ruby",
393        "rb",
394        "php",
395        "perl",
396        "lua",
397        "r",
398        "dart",
399        "bash",
400        "sh",
401        "shell",
402        "powershell",
403        "zsh",
404        "batch",
405        "cmd",
406        "sql",
407        "tsql",
408        "plsql",
409        "graphql",
410        "dockerfile",
411        "makefile",
412        "css",
413        "scss",
414        "sass",
415        "less",
416        "markdown",
417        "md",
418        "tex",
419        "latex",
420        "diff",
421        "proto",
422        "razor",
423        "cshtml",
424        "xaml",
425        "aspx",
426        "http",
427    ];
428    let lower = t.to_ascii_lowercase();
429    LANGS.contains(&lower.as_str())
430}
431
432/// Mark the region indices that are a code block's **language label** — a bare
433/// `XML`/`C#`/… token sitting directly above a `code` region — so they are consumed
434/// rather than emitted as their own stray paragraph/heading. The label may also be
435/// captured inside a wider code box (rendered as the fence's first line); dropping
436/// the standalone copy just removes the duplicate.
437fn code_language_labels(regions: &[Region], cells: &[TextCell]) -> Vec<bool> {
438    let mut drop = vec![false; regions.len()];
439    for (i, r) in regions.iter().enumerate() {
440        if matches!(r.label, "code" | "picture" | "table") {
441            continue;
442        }
443        if !is_code_language(&region_text(r, cells)) {
444            continue;
445        }
446        // The label sits just above the code (a blank line's gap) or is swallowed
447        // into the top of a wider code box; either way it is that block's label.
448        // The window is generous because the label's own font is small, so a
449        // one-line gap is several times its height.
450        let line_h = (r.b - r.t).abs().max(1.0);
451        let window = (line_h * 4.0).max(28.0);
452        let labels_code = regions.iter().enumerate().any(|(j, c)| {
453            if j == i || c.label != "code" {
454                return false;
455            }
456            let gap = c.t - r.b; // >0 when the code is below the label
457            let h_overlap = (r.r.min(c.r) - r.l.max(c.l)).max(0.0);
458            gap > -line_h * 3.0 && gap < window && h_overlap > 0.0
459        });
460        if labels_code {
461            drop[i] = true;
462        }
463    }
464    drop
465}
466
467/// Collapse `code` regions where one is nested inside another, keeping the larger.
468///
469/// RT-DETR sometimes emits a tight code box *and* a wider near-duplicate that also
470/// captures the block's language label (`XML`, `C#`, …). When the tight box scores
471/// higher it is kept first, and the wider container — not "mostly inside" the tight
472/// box — survives [`resolve`]'s greedy pass, so the block is emitted twice. Keeping
473/// the **larger** box (rather than dropping it) collapses the pair without leaking
474/// the container's extra cells back out as orphan text, since the larger box still
475/// covers every cell. Restricted to `code` so genuinely distinct nested regions of
476/// other kinds are untouched.
477fn dedup_nested_code(kept: &mut Vec<Region>) {
478    let mut drop = vec![false; kept.len()];
479    for i in 0..kept.len() {
480        if kept[i].label != "code" {
481            continue;
482        }
483        let ai = area(kept[i].l, kept[i].t, kept[i].r, kept[i].b).max(1.0);
484        for j in 0..kept.len() {
485            if i == j || drop[j] || kept[j].label != "code" {
486                continue;
487            }
488            let aj = area(kept[j].l, kept[j].t, kept[j].r, kept[j].b).max(1.0);
489            // Drop i when it is mostly inside a strictly larger code box j.
490            let overlap = inter(&kept[i], kept[j].l, kept[j].t, kept[j].r, kept[j].b);
491            if aj > ai && overlap / ai > 0.7 {
492                drop[i] = true;
493                break;
494            }
495        }
496    }
497    let mut keep = drop.iter();
498    kept.retain(|_| !*keep.next().unwrap());
499}
500
501/// Fraction of the page's non-empty text cells that some detected region
502/// claims (>0.2 intersection-over-self, docling's assignment rule). 1.0 for a
503/// page without text cells.
504///
505/// The int8-layout guard keys off this: a dense digital page whose detections
506/// cover almost none of its text is the signature of quantized confidences
507/// flipping under the 0.5 label thresholds on this CPU's kernels — not of a
508/// genuinely empty layout — and is worth re-running on the fp32 graph.
509pub fn layout_cell_coverage(regions: &[Region], cells: &[TextCell]) -> f32 {
510    let mut total = 0usize;
511    let mut covered = 0usize;
512    for c in cells {
513        if c.text.trim().is_empty() {
514            continue;
515        }
516        total += 1;
517        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
518        if regions
519            .iter()
520            .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
521        {
522            covered += 1;
523        }
524    }
525    if total == 0 {
526        1.0
527    } else {
528        covered as f32 / total as f32
529    }
530}
531
532/// Append `text` regions for cells the layout left uncovered ("orphan cells"),
533/// the way docling's `LayoutPostprocessor` does (`create_orphan_clusters`): any
534/// non-empty cell that no kept region covers (>50% of the cell's area) becomes a
535/// text region of its own, so text the detector missed (a stray `.`, a small
536/// label) is still emitted instead of silently dropped. Adjacent orphan cells on a
537/// line are merged so a missed paragraph doesn't shatter into one block per line.
538pub fn add_orphan_regions(regions: &mut Vec<Region>, cells: &[TextCell]) {
539    // docling assigns each cell to its single best-overlapping cluster at
540    // intersection-over-self > 0.2 and serializes exactly the assigned cells —
541    // and since [`region_texts_exclusive`] now emits under that very rule, the
542    // claim test here matches it: any cell over 0.2 will actually render in
543    // its best region, everything else becomes an orphan. Completeness by
544    // construction, with no (0.2, 0.5] hole (the old > 0.5 serializer needed
545    // the claim test raised to > 0.5 to keep right_to_left_03's `20300` from
546    // vanishing; the exclusive port closes that structurally).
547    //
548    // Only *regular* clusters claim cells: docling's `_find_unassigned_cells`
549    // walks `regular_clusters` alone, so a cell under a `picture` or a wrapper
550    // (`table`/`document_index`/`form`/`key_value_region`) that no regular
551    // cluster covers still becomes an orphan text cluster (#165). The orphans
552    // that end up *fully* inside the special are re-dropped by
553    // [`drop_contained_regulars`] (docling's Markdown drops them the same way
554    // — a picture's children never reach its `MarkdownPictureSerializer`
555    // output, a table's text renders through the reconstructed grid). The
556    // observable fix is the border-straddlers: a line only partially under a
557    // figure box used to lose its cells to the picture's 0.2 claim and vanish
558    // — now it forms an orphan region and is emitted, as docling does.
559    let assigned = |c: &TextCell| {
560        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
561        regions
562            .iter()
563            .filter(|r| r.label != "picture" && !is_wrapper(r.label))
564            .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
565    };
566    // Collect orphan cells (non-empty, unassigned), in page order.
567    let mut orphans: Vec<&TextCell> = cells
568        .iter()
569        .filter(|c| !c.text.trim().is_empty() && !assigned(c))
570        .collect();
571    if orphans.is_empty() {
572        return;
573    }
574    orphans.sort_by(|a, b| a.t.total_cmp(&b.t).then(a.l.total_cmp(&b.l)));
575    // Merge cells that sit on the same line and nearly touch into one region, so a
576    // dropped multi-word line stays one block (docling's refinement merges these).
577    let mut merged: Vec<Region> = Vec::new();
578    for c in orphans {
579        let h = (c.b - c.t).abs().max(1.0);
580        if let Some(last) = merged.last_mut() {
581            let same_line = (last.t - c.t).abs() < h * 0.5;
582            let touching = c.l <= last.r + h && c.l >= last.l - h;
583            if same_line && touching {
584                last.l = last.l.min(c.l);
585                last.r = last.r.max(c.r);
586                last.t = last.t.min(c.t);
587                last.b = last.b.max(c.b);
588                continue;
589            }
590        }
591        merged.push(Region {
592            label: "text",
593            score: 0.0,
594            l: c.l,
595            t: c.t,
596            r: c.r,
597            b: c.b,
598        });
599    }
600    regions.extend(merged);
601}
602
603/// Demote a `picture` region that is really a **text panel** — a paragraph block
604/// the layout model boxed as a figure because it is typeset on a colored
605/// background (terms-and-conditions callouts, quote boxes) — into ordinary
606/// `text` regions, one per paragraph, so its words are read instead of shipped
607/// as pixels. docling loses this text the same way (cells assigned to a picture
608/// cluster are never serialized); this is a deliberate improvement, not parity.
609///
610/// The gate is conservative so a genuine figure keeps its crop: the region must
611/// contain at least three text lines whose median width spans most of the panel
612/// (axis labels and chat bubbles are narrow and varied) and whose cells cover a
613/// substantial fraction of its area (a photo or chart with sparse labels does
614/// not). Paragraph boundaries are re-derived from the line pitch: a vertical gap
615/// clearly larger than the panel's own leading starts a new `text` region, so
616/// the panel doesn't collapse into one giant paragraph.
617///
618/// Works on any cell source — the digital text layer or OCR lines recognized
619/// from the picture crop — so the native and browser paths, with or without
620/// force-OCR, demote identically.
621pub fn recover_text_panels(regions: &mut Vec<Region>, cells: &[TextCell]) {
622    // A *captioned* picture is a genuine figure whatever it contains — the
623    // corpus is full of document screenshots ("Figure 3: …" above a page
624    // image) that are exactly as dense and wide as a text panel. Only an
625    // uncaptioned picture is a demotion candidate.
626    let captioned: Vec<bool> = regions
627        .iter()
628        .map(|r| {
629            r.label == "picture"
630                && regions.iter().any(|c| {
631                    c.label == "caption" && c.r.min(r.r) - c.l.max(r.l) > 0.0 && {
632                        let gap = if c.t >= r.b {
633                            c.t - r.b
634                        } else if r.t >= c.b {
635                            r.t - c.b
636                        } else {
637                            f32::MAX // vertically overlapping: not a caption
638                        };
639                        gap <= 25.0
640                    }
641                })
642        })
643        .collect();
644    let mut out: Vec<Region> = Vec::with_capacity(regions.len());
645    // Synthesized paragraphs and the demoted panels' boxes are kept separate
646    // from `out` until the end: the dedup filter below must not confuse a
647    // paragraph we just built with a pre-existing region inside the panel.
648    let mut demoted_paras: Vec<Region> = Vec::new();
649    let mut demoted_boxes: Vec<(f32, f32, f32, f32)> = Vec::new();
650    for (i, r) in regions.drain(..).enumerate() {
651        if r.label != "picture" || captioned[i] {
652            out.push(r);
653            continue;
654        }
655        let inside: Vec<&TextCell> = cells
656            .iter()
657            .filter(|c| {
658                !c.text.trim().is_empty() && {
659                    let ca = area(c.l, c.t, c.r, c.b).max(1.0);
660                    inter(&r, c.l, c.t, c.r, c.b) / ca > 0.5
661                }
662            })
663            .collect();
664        // Group the contained cells into lines by vertical overlap (the same
665        // rule region_text orders by), tracking each line's union box.
666        let mut lines: Vec<(f32, f32, f32, f32)> = Vec::new(); // (t, b, l, r)
667        for c in &inside {
668            let (ct, cb) = (c.t.min(c.b), c.t.max(c.b));
669            match lines.iter_mut().find(|(lt, lb, _, _)| {
670                let ov = cb.min(*lb) - ct.max(*lt);
671                ov > 0.5 * (cb - ct).min(*lb - *lt).max(1.0)
672            }) {
673                Some((lt, lb, ll, lr)) => {
674                    *lt = lt.min(ct);
675                    *lb = lb.max(cb);
676                    *ll = ll.min(c.l);
677                    *lr = lr.max(c.r);
678                }
679                None => lines.push((ct, cb, c.l, c.r)),
680            }
681        }
682        if lines.len() < 3 {
683            out.push(r);
684            continue;
685        }
686        let panel_w = (r.r - r.l).max(1.0);
687        let coverage = inside.iter().map(|c| area(c.l, c.t, c.r, c.b)).sum::<f32>()
688            / area(r.l, r.t, r.r, r.b).max(1.0);
689        let mut widths: Vec<f32> = lines.iter().map(|(_, _, l, rr)| rr - l).collect();
690        widths.sort_by(f32::total_cmp);
691        // A figure's text is ragged: a title line, small axis/tick labels, and
692        // OCR boxes over the plot area come out at wildly different heights,
693        // whereas a real text panel is set in one face with constant leading.
694        // Require near-uniform line heights (median absolute deviation ≤ 35%
695        // of the median) so an uncaptioned chart keeps its crop even when its
696        // labels are dense enough to pass the coverage gate (#173) — garbled
697        // OCR of its bars is not content.
698        let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
699        heights.sort_by(f32::total_cmp);
700        let h_med = heights[heights.len() / 2].max(1.0);
701        let mut devs: Vec<f32> = heights.iter().map(|h| (h - h_med).abs()).collect();
702        devs.sort_by(f32::total_cmp);
703        let uniform = devs[devs.len() / 2] <= 0.35 * h_med;
704        let text_panel = coverage >= 0.2 && widths[widths.len() / 2] >= 0.45 * panel_w && uniform;
705        if !text_panel {
706            out.push(r);
707            continue;
708        }
709        lines.sort_by(|a, b| a.0.total_cmp(&b.0));
710        let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
711        heights.sort_by(f32::total_cmp);
712        let h = heights[heights.len() / 2].max(1.0);
713        let mut gaps: Vec<f32> = lines
714            .windows(2)
715            .map(|w| (w[1].0 - w[0].1).max(0.0))
716            .collect();
717        gaps.sort_by(f32::total_cmp);
718        let leading = if gaps.is_empty() {
719            0.0
720        } else {
721            gaps[gaps.len() / 2]
722        };
723        let brk = (1.8 * leading).max(0.75 * h);
724        let mut para: Option<(f32, f32, f32, f32)> = None; // (l, t, r, b) union
725        for (t, b, l, rr) in &lines {
726            match &mut para {
727                Some((pl, _, pr, pb)) if *t - *pb <= brk => {
728                    *pl = pl.min(*l);
729                    *pr = pr.max(*rr);
730                    *pb = pb.max(*b);
731                }
732                _ => {
733                    if let Some((pl, pt, pr, pb)) = para.take() {
734                        demoted_paras.push(Region {
735                            label: "text",
736                            score: r.score,
737                            l: pl,
738                            t: pt,
739                            r: pr,
740                            b: pb,
741                        });
742                    }
743                    para = Some((*l, *t, *rr, *b));
744                }
745            }
746        }
747        if let Some((pl, pt, pr, pb)) = para {
748            demoted_paras.push(Region {
749                label: "text",
750                score: r.score,
751                l: pl,
752                t: pt,
753                r: pr,
754                b: pb,
755            });
756        }
757        demoted_boxes.push((r.l, r.t, r.r, r.b));
758    }
759    // The paragraphs are rebuilt from *all* of the panel's cells, so any
760    // surviving text region inside a demoted panel (an orphan cluster or a
761    // layout-detected fragment — pictures no longer swallow them, #165) would
762    // say the same words twice. Consume those; wrappers and pictures stay.
763    if !demoted_boxes.is_empty() {
764        out.retain(|r| {
765            r.label == "picture" || is_wrapper(r.label) || {
766                let ra = area(r.l, r.t, r.r, r.b).max(1.0);
767                !demoted_boxes
768                    .iter()
769                    .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.5)
770            }
771        });
772    }
773    out.extend(demoted_paras);
774    *regions = out;
775}
776
777/// Drop a `picture` detection that is a small, empty, low-confidence margin box on
778/// a **text page** — a false positive the RT-DETR layout sometimes emits (e.g.
779/// `right_to_left_02`'s phantom right-column picture, score 0.40); docling does not
780/// emit it. The gate is deliberately narrow so a genuine figure is never dropped:
781/// (1) only on pages with a digital text layer — image/scanned/figure pages have
782/// no `cells` yet at this point (OCR runs later), so their pictures, which *are*
783/// the content, are kept; (2) only a box covering < 25 % of the page (a margin
784/// artifact, not a dominant figure); (3) only when it contains no text and scores
785/// below 0.5 (real empty figures in the corpus all score ≥ 0.86).
786pub fn drop_false_pictures(
787    regions: &mut Vec<Region>,
788    cells: &[TextCell],
789    page_w: f32,
790    page_h: f32,
791) {
792    if cells.iter().all(|c| c.text.trim().is_empty()) {
793        return; // no digital text layer (image/scanned page) — keep all pictures
794    }
795    // A text-document page carries several text-bearing non-picture regions (so a
796    // spurious margin picture is clearly extra). A slide / figure page has at most
797    // one — there the picture is the content, so never drop it.
798    let content_regions = regions
799        .iter()
800        .filter(|r| r.label != "picture" && !region_text(r, cells).trim().is_empty())
801        .count();
802    if content_regions < 2 {
803        return;
804    }
805    let page_area = (page_w * page_h).max(1.0);
806    regions.retain(|r| {
807        if r.label != "picture" || r.score >= 0.5 {
808            return true;
809        }
810        if area(r.l, r.t, r.r, r.b) / page_area >= 0.25 {
811            return true; // a dominant figure, not a margin artifact
812        }
813        // Keep it if any text cell falls mostly inside (a real captioned/labelled
814        // figure); drop only the genuinely empty low-confidence boxes.
815        cells.iter().any(|c| {
816            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
817            !c.text.trim().is_empty() && inter(r, c.l, c.t, c.r, c.b) / ca > 0.5
818        })
819    });
820}
821
822/// A small digit-only region in the top/bottom margin: a page number. docling
823/// emits `right_to_left_02`'s bottom `11` as the page's *first* text item (its
824/// reading-order model floats the page number to the front), whereas our
825/// position-based ordering would place a bottom region last.
826fn is_page_number(region: &Region, cells: &[TextCell], page_h: f32) -> bool {
827    let t = region_text(region, cells);
828    let t = t.trim();
829    !t.is_empty()
830        && t.chars().all(|c| c.is_ascii_digit())
831        && (region.b - region.t).abs() < 30.0
832        && (region.t < page_h * 0.12 || region.b > page_h * 0.88)
833}
834
835/// docling's `form` / `key_value_region` *containers* (2.123, docling#4064):
836/// every region sitting > 0.8 inside one — text, list items, and since #4064
837/// tables and pictures too — is that container's child. Children are
838/// reading-ordered among themselves and emitted as one block where the
839/// container falls in the page's top-level order (a `form_area` /
840/// `key_value_area` group upstream), instead of interleaving with the text
841/// around the form. A child inside several containers belongs to the smallest
842/// (then most confident, then first); a container with children shrinks to
843/// their union for the top-level ordering, like upstream's bbox adjustment.
844///
845/// The containers themselves are still not emitted (`is_skipped`), so the
846/// Markdown is exactly upstream's — a group prints only its children.
847fn order_with_containers<T: Clone>(
848    items: &mut Vec<T>,
849    page_w: f32,
850    page_h: f32,
851    reg: impl Fn(&T) -> &Region,
852) {
853    let is_container = |r: &Region| matches!(r.label, "form" | "key_value_region");
854    let containers: Vec<usize> = (0..items.len())
855        .filter(|&i| is_container(reg(&items[i])))
856        .collect();
857    if containers.is_empty() {
858        order_regions(items, page_w, page_h, reg);
859        return;
860    }
861    // Parent container per item (containers never nest in each other here —
862    // upstream assigns regulars and tables/pictures only).
863    let mut parent: Vec<Option<usize>> = vec![None; items.len()];
864    for i in 0..items.len() {
865        let r = reg(&items[i]);
866        if is_container(r) {
867            continue;
868        }
869        let ra = area(r.l, r.t, r.r, r.b).max(1.0);
870        let mut best: Option<(usize, f32, f32)> = None; // (idx, area, -score)
871        for &c in &containers {
872            let cr = reg(&items[c]);
873            if inter(r, cr.l, cr.t, cr.r, cr.b) / ra > 0.8 {
874                let key = (area(cr.l, cr.t, cr.r, cr.b), -cr.score);
875                if best.is_none_or(|(_, a, s)| key.0 < a || (key.0 == a && key.1 < s)) {
876                    best = Some((c, key.0, key.1));
877                }
878            }
879        }
880        parent[i] = best.map(|(c, _, _)| c);
881    }
882    // Top-level pass: non-children plus the containers, the latter shrunk to
883    // their children's union.
884    let mut top: Vec<(usize, Region)> = Vec::new();
885    for i in 0..items.len() {
886        if parent[i].is_some() {
887            continue;
888        }
889        let mut r = reg(&items[i]).clone();
890        if is_container(&r) {
891            let kids: Vec<&Region> = (0..items.len())
892                .filter(|&k| parent[k] == Some(i))
893                .map(|k| reg(&items[k]))
894                .collect();
895            if !kids.is_empty() {
896                r.l = kids.iter().map(|k| k.l).fold(f32::INFINITY, f32::min);
897                r.t = kids.iter().map(|k| k.t).fold(f32::INFINITY, f32::min);
898                r.r = kids.iter().map(|k| k.r).fold(f32::NEG_INFINITY, f32::max);
899                r.b = kids.iter().map(|k| k.b).fold(f32::NEG_INFINITY, f32::max);
900            }
901        }
902        top.push((i, r));
903    }
904    order_regions(&mut top, page_w, page_h, |it| &it.1);
905    let mut out: Vec<T> = Vec::with_capacity(items.len());
906    for (i, _) in top {
907        if is_container(reg(&items[i])) {
908            let mut kids: Vec<T> = (0..items.len())
909                .filter(|&k| parent[k] == Some(i))
910                .map(|k| items[k].clone())
911                .collect();
912            order_regions(&mut kids, page_w, page_h, &reg);
913            out.push(items[i].clone());
914            out.extend(kids);
915        } else {
916            out.push(items[i].clone());
917        }
918    }
919    *items = out;
920}
921
922/// Furniture / not-yet-emitted labels.
923fn is_skipped(label: &str) -> bool {
924    matches!(
925        label,
926        "page_header" | "page_footer" | "form" | "key_value_region"
927    )
928}
929
930/// Reading-order sort of a page's regions, via the ported rule-based
931/// [`reading_order`](crate::reading_order) predictor (docling's
932/// `ReadingOrderPredictor`): an up/down geometry graph, horizontal dilation and a
933/// depth-first traversal, with `page_header`/`page_footer` ordered as their own
934/// groups (first/last) as docling does.
935fn order_regions<T: Clone>(
936    items: &mut Vec<T>,
937    page_w: f32,
938    page_h: f32,
939    reg: impl Fn(&T) -> &Region,
940) {
941    let boxes: Vec<(f32, f32, f32, f32)> = items
942        .iter()
943        .map(|it| {
944            let r = reg(it);
945            (r.l, r.t, r.r, r.b)
946        })
947        .collect();
948    let is_header: Vec<bool> = items
949        .iter()
950        .map(|it| reg(it).label == "page_header")
951        .collect();
952    let is_footer: Vec<bool> = items
953        .iter()
954        .map(|it| reg(it).label == "page_footer")
955        .collect();
956    let order = crate::reading_order::order_page(&boxes, &is_header, &is_footer, page_w, page_h);
957    *items = order.iter().map(|&i| items[i].clone()).collect();
958}
959
960/// Clean a region's assembled text: undo soft-hyphen line wraps, map curly
961/// quotes and the ellipsis to ASCII (matching docling), and collapse runs of
962/// whitespace. pdfium emits the line-wrap hyphen as U+0002 in this corpus
963/// (U+00AD elsewhere), so `word\u{2} continuation` is one hyphenated word —
964/// drop the hyphen + the joining space and merge (`com\u{2} pact` → `compact`,
965/// `end-to\u{2} end` → `end-toend`), exactly as docling does.
966///
967/// Token spacing is otherwise left as the geometric join produced it. We do not
968/// tighten punctuation spacing: docling preserves the PDF's own spaces (it keeps
969/// `{ ahn }`, `Name 1 .`, `[ 9 ]`), and a geometric gap heuristic diverges from
970/// it more than a plain single-space join does.
971/// An ordered-list enumeration marker at the start of a list item: leading ASCII
972/// digits followed by `.`, e.g. `1. Undo/Redo` → `(1, "Undo/Redo")`. Returns
973/// `None` when the text doesn't start with `digits.`.
974fn parse_ordered_marker(s: &str) -> Option<(u64, String)> {
975    let digits: String = s.chars().take_while(|c| c.is_ascii_digit()).collect();
976    if digits.is_empty() {
977        return None;
978    }
979    let rest = s[digits.len()..].strip_prefix('.')?;
980    let number = digits.parse().ok()?;
981    Some((number, rest.trim_start().to_string()))
982}
983
984/// Escape markdown special characters the way docling-core's markdown serializer
985/// does (`markdown.py` post_process): `_` → `\_`, then HTML-escape `&`, `<`, `>`
986/// (quote=False, so quotes are left). Applied to prose (headings, list items,
987/// paragraphs); code blocks, the formula placeholder, and table cells are left raw.
988fn md_escape(text: &str) -> String {
989    text.replace('_', "\\_")
990        .replace('&', "&amp;")
991        .replace('<', "&lt;")
992        .replace('>', "&gt;")
993}
994
995fn clean_text(text: &str) -> String {
996    // Typographic-quote normalization follows docling-parse's sanitizer table
997    // (`pdf_sanitators/constants.h`): every curly quote — single *and double* —
998    // becomes the ASCII apostrophe `'`, and `‚` a comma. A `"` in docling's
999    // output only ever comes from a literal `quotedbl` glyph, never from `“ ”`
1000    // (2206's `'text in the wild"` pairs a curly open with a literal-quote
1001    // close). This replaces an earlier Hangul-only special case that patched
1002    // one symptom of mapping `“ ”` to `"`.
1003    let replaced = text
1004        .replace("\u{2} ", "")
1005        .replace("\u{ad} ", "")
1006        .replace(['\u{2}', '\u{ad}'], "") // any stray wrap hyphens not at a join
1007        .replace(
1008            [
1009                '\u{2018}', '\u{2019}', '\u{201b}', '\u{201c}', '\u{201d}', '\u{201e}', '\u{201f}',
1010            ],
1011            "'",
1012        ) // ‘ ’ ‛ “ ” „ ‟ → '
1013        .replace('\u{201a}', ",") // ‚ → ,
1014        .replace(
1015            [
1016                '\u{2010}', '\u{2011}', '\u{2012}', '\u{2013}', '\u{2014}', '\u{2015}', '\u{2212}',
1017            ],
1018            "-",
1019        ) // hyphen/dash family → -
1020        .replace('\u{2044}', "/") // ⁄ fraction slash → /
1021        .replace('\u{2022}', "\u{b7}") // • → · (docling never emits •; inline CCS-concept separators)
1022        .replace('\u{2026}', "..."); // … → ...
1023    let out = if crate::pdfium_backend::use_dp_lines() {
1024        // The docling-parse sanitizer already placed the correct spacing (e.g.
1025        // justified double spaces); preserve internal runs of spaces, only
1026        // normalizing line breaks/tabs and trimming the ends.
1027        replaced.replace(['\n', '\r', '\t'], " ").trim().to_string()
1028    } else {
1029        // Legacy: collapse all whitespace runs to single spaces.
1030        replaced.split_whitespace().collect::<Vec<_>>().join(" ")
1031    };
1032    fix_arabic_lam_alef(&out)
1033}
1034
1035/// pdfium decomposes the Arabic lam-alef ligature (لا / لإ / لأ / لآ) into its
1036/// glyph constituents in *visual* order — `alef-variant, lam` — but docling keeps
1037/// logical order, `lam, alef-variant`. Swap a mid-word `alef-variant + lam` back
1038/// to `lam + alef-variant`. "Mid-word" (the previous char is an Arabic letter)
1039/// distinguishes the ligature from the definite article `ال` (word-initial
1040/// `alef + lam`), which must stay. No-op for non-Arabic text.
1041fn fix_arabic_lam_alef(s: &str) -> String {
1042    let is_arabic_letter = |c: char| ('\u{0620}'..='\u{064A}').contains(&c);
1043    let chars: Vec<char> = s.chars().collect();
1044    if !chars.iter().any(|&c| is_arabic_letter(c)) {
1045        return s.to_string(); // no-op for non-Arabic text
1046    }
1047    // Pass 1: swap mid-word `alef-variant + lam` → `lam + alef-variant`. Only the
1048    // hamza/madda alef variants (إ أ آ) are safe: the definite article is always
1049    // plain `ا + ل`, so plain `alef + lam` is ambiguous (a legitimate `فعالة` vs a
1050    // reversed `لا` ligature look identical) — leaving plain alef alone avoids
1051    // corrupting legitimate words.
1052    let mut a: Vec<char> = Vec::with_capacity(chars.len());
1053    let mut i = 0;
1054    while i < chars.len() {
1055        let c = chars[i];
1056        if matches!(c, '\u{0622}' | '\u{0623}' | '\u{0625}')
1057            && chars.get(i + 1) == Some(&'\u{0644}')
1058            && i > 0
1059            && is_arabic_letter(chars[i - 1])
1060            // A preceding lam means this alef-variant is *already* the logical
1061            // `lam + alef` ligature; the following lam is the next syllable's
1062            // letter, not a reversed ligature — swapping it corrupts `لآل` → `للآ`
1063            // (e.g. التعلم الآلي → الآلي, not اللآي).
1064            && chars[i - 1] != '\u{0644}'
1065        {
1066            a.push('\u{0644}');
1067            a.push(c);
1068            i += 2;
1069            continue;
1070        }
1071        a.push(c);
1072        i += 1;
1073    }
1074    // Pass 2: insert a space at Arabic↔Latin boundaries (bidi script switch) that
1075    // pdfium runs together — docling separates the embedded Latin run (`وPython`
1076    // → `و Python`).
1077    let mut out: Vec<char> = Vec::with_capacity(a.len());
1078    for (j, &c) in a.iter().enumerate() {
1079        if j > 0 {
1080            let p = a[j - 1];
1081            if (is_arabic_letter(p) && c.is_ascii_alphabetic())
1082                || (p.is_ascii_alphabetic() && is_arabic_letter(c))
1083            {
1084                out.push(' ');
1085            }
1086        }
1087        out.push(c);
1088    }
1089    out.into_iter().collect()
1090}
1091
1092/// docling's `PageAssembleModel._match_hyperlink`: the URI whose link
1093/// annotations cover at least half of the region's box, or `None`. Coverage is
1094/// intersection-over-region-area, **accumulated per URI** — a URL that wraps
1095/// across lines carries several annotation rects that sum toward the same
1096/// target. Ties resolve to the first-seen URI (Python's `max` over dict
1097/// insertion order); the winner still needs `>= 0.5`
1098/// (`_HYPERLINK_COVERAGE_THRESHOLD`).
1099pub(crate) fn region_hyperlink(
1100    region: &Region,
1101    links: &[crate::pdfium_backend::LinkAnnot],
1102) -> Option<String> {
1103    if links.is_empty() {
1104        return None;
1105    }
1106    let area = (region.r - region.l).max(0.0) * (region.b - region.t).max(0.0);
1107    if area <= 0.0 {
1108        return None;
1109    }
1110    let mut coverage: Vec<(&str, f32)> = Vec::new();
1111    for link in links {
1112        let ix = (region.r.min(link.r) - region.l.max(link.l)).max(0.0);
1113        let iy = (region.b.min(link.b) - region.t.max(link.t)).max(0.0);
1114        let c = ix * iy / area;
1115        match coverage.iter_mut().find(|(uri, _)| *uri == link.uri) {
1116            Some((_, acc)) => *acc += c,
1117            None => coverage.push((&link.uri, c)),
1118        }
1119    }
1120    let mut best: Option<(&str, f32)> = None;
1121    for (uri, c) in coverage {
1122        // Strictly greater keeps the first-seen URI on ties, like Python's max.
1123        if best.is_none_or(|(_, bc)| c > bc) {
1124            best = Some((uri, c));
1125        }
1126    }
1127    let (uri, c) = best?;
1128    (c >= 0.5).then(|| normalize_uri(uri))
1129}
1130
1131/// The pydantic-`AnyUrl` normalization docling's hyperlink value passes
1132/// through on its way to the serializer: a URL with an authority but no path
1133/// gains a trailing `/` (`https://arxiv.org` → `https://arxiv.org/`). Other
1134/// AnyUrl canonicalizations (scheme/host lowercasing, percent-encoding) don't
1135/// occur in PDF link annotations in practice, so they are not reproduced.
1136fn normalize_uri(uri: &str) -> String {
1137    if let Some((_, rest)) = uri.split_once("://") {
1138        if !rest.is_empty() && !rest.contains(['/', '?', '#']) {
1139            return format!("{uri}/");
1140        }
1141    }
1142    uri.to_string()
1143}
1144
1145/// Resolve each page hyperlink to the visible text it covers, as `(anchor, uri)`
1146/// in reading order. The anchor is the cells whose centre falls in the link rect,
1147/// joined left-to-right and cleaned the same way prose is (so it matches the
1148/// serialized text), deduped against the immediately-preceding link so pdfium's
1149/// occasional duplicate annotation doesn't double-list. Empty anchors are dropped.
1150pub(crate) fn resolve_link_anchors(page: &PdfPage) -> Vec<(String, String)> {
1151    let mut out: Vec<(String, String)> = Vec::new();
1152    // Use per-word cells, not the line-merged `cells`: a link rect covers a few
1153    // words on a line, and a whole merged line cell would over-capture (its centre
1154    // lands in one link's rect, grabbing the entire line as that link's anchor).
1155    let words = if page.word_cells.is_empty() {
1156        &page.cells
1157    } else {
1158        &page.word_cells
1159    };
1160    for link in &page.links {
1161        // A cell participates when its centre row is inside the rect and it
1162        // overlaps the rect horizontally. A cell can be *wider* than the rect:
1163        // PDFs often draw a whole header line as one text run ("LinkedIn |
1164        // GitHub | Credly"), which docling-parse's word grouping keeps as one
1165        // cell even though each label carries its own link annotation —
1166        // centre-in-rect alone would hand the entire line to every link.
1167        // [`cell_text_in_rect`] clips such a cell to the tokens under the rect.
1168        let mut inside: Vec<(&TextCell, String)> = words
1169            .iter()
1170            .filter(|c| {
1171                let cy = (c.t + c.b) / 2.0;
1172                cy >= link.t && cy <= link.b && c.r.min(link.r) > c.l.max(link.l)
1173            })
1174            .filter_map(|c| {
1175                let text = cell_text_in_rect(c, link.l, link.r);
1176                (!text.is_empty()).then_some((c, text))
1177            })
1178            .collect();
1179        // Reading order: top band then left-to-right (link anchors are LTR).
1180        let band = inside
1181            .iter()
1182            .map(|(c, _)| (c.b - c.t).abs())
1183            .fold(0.0f32, f32::max)
1184            .max(1.0);
1185        inside.sort_by_key(|(c, _)| ((c.t / band).round() as i64, (c.l * 10.0) as i64));
1186        let anchor = clean_text(
1187            &inside
1188                .iter()
1189                .map(|(_, t)| t.trim())
1190                .filter(|t| !t.is_empty())
1191                .collect::<Vec<_>>()
1192                .join(" "),
1193        );
1194        if anchor.is_empty() {
1195            continue;
1196        }
1197        if out
1198            .last()
1199            .is_some_and(|(a, u)| a == &anchor && u == &link.uri)
1200        {
1201            continue;
1202        }
1203        out.push((anchor, link.uri.clone()));
1204    }
1205    out
1206}
1207
1208/// The part of a cell's text that lies under a link rect's x-range. A cell
1209/// fully inside the rect (by centre) returns its whole text. A wider cell is
1210/// split into whitespace tokens whose x-spans are estimated proportionally to
1211/// their character positions (kerning makes this approximate, so selection
1212/// snaps to whole tokens, never characters); tokens whose estimated centre
1213/// falls inside the rect are kept. Returns "" when nothing falls inside.
1214fn cell_text_in_rect(c: &TextCell, l: f32, r: f32) -> String {
1215    let cx = (c.l + c.r) / 2.0;
1216    if cx >= l && cx <= r && c.l >= l - (c.r - c.l) * 0.25 && c.r <= r + (c.r - c.l) * 0.25 {
1217        return c.text.trim().to_string();
1218    }
1219    let chars: Vec<char> = c.text.chars().collect();
1220    let n = chars.len();
1221    if n == 0 || c.r <= c.l {
1222        return String::new();
1223    }
1224    let per = (c.r - c.l) / n as f32;
1225    let mut out: Vec<String> = Vec::new();
1226    let mut token = String::new();
1227    let mut start = 0usize;
1228    // A trailing sentinel space flushes the last token.
1229    for (i, &ch) in chars.iter().enumerate().chain(std::iter::once((n, &' '))) {
1230        if ch.is_whitespace() {
1231            if !token.is_empty() {
1232                let mid = c.l + (start as f32 + (i - start) as f32 / 2.0) * per;
1233                if mid >= l && mid <= r {
1234                    out.push(std::mem::take(&mut token));
1235                } else {
1236                    token.clear();
1237                }
1238            }
1239        } else {
1240            if token.is_empty() {
1241                start = i;
1242            }
1243            token.push(ch);
1244        }
1245    }
1246    out.join(" ")
1247}
1248
1249/// Cells assigned to a region (best container), in reading order, joined.
1250fn region_text(region: &Region, cells: &[TextCell]) -> String {
1251    let inside: Vec<&TextCell> = cells
1252        .iter()
1253        .filter(|c| {
1254            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1255            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1256        })
1257        .collect();
1258    cells_text(inside)
1259}
1260
1261/// docling's exclusive cell assignment (`_assign_cells_to_clusters`): every
1262/// non-empty cell goes to the single best-overlapping *regular* region at
1263/// intersection-over-self > 0.2, and each region serializes exactly its
1264/// assigned cells. A cell under two overlapping boxes is emitted once (by the
1265/// better-covering one), and a cell only partially under its region — e.g.
1266/// normal_4pages' big section numeral, ~30 % inside the heading box — still
1267/// joins it (`## 들어가며 1`) instead of leaking as an orphan. Pictures and
1268/// wrappers never claim (docling walks regular clusters only); ties go to the
1269/// first region, like docling's strict `>` best-overlap scan.
1270pub fn region_texts_exclusive(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
1271    let owned = assign_cells(regions, cells);
1272    // Non-claimers (tables/wrappers/pictures) keep the inclusive > 0.5 text:
1273    // docling fills a special cluster's cells from its contained children, and
1274    // downstream table assembly gates on that text being non-empty.
1275    regions
1276        .iter()
1277        .zip(owned)
1278        .map(|(r, cs)| {
1279            if claims_cells(r) {
1280                cells_text(cs.iter().map(|&i| &cells[i]).collect())
1281            } else {
1282                region_text(r, cells)
1283            }
1284        })
1285        .collect()
1286}
1287
1288/// A *regular* region in docling's sense — one that claims cells. Pictures and
1289/// the wrappers (`table`, `document_index`, `form`, `key_value_region`) fill
1290/// their cells from contained children instead.
1291fn claims_cells(r: &Region) -> bool {
1292    r.label != "picture" && !is_wrapper(r.label)
1293}
1294
1295/// docling's `_assign_cells_to_clusters`: each non-empty cell's index goes to
1296/// the single best-overlapping regular region at intersection-over-self > 0.2
1297/// (ties to the first region, like docling's strict `>` scan). One entry per
1298/// region, in region order.
1299fn assign_cells(regions: &[Region], cells: &[TextCell]) -> Vec<Vec<usize>> {
1300    let mut owned: Vec<Vec<usize>> = vec![Vec::new(); regions.len()];
1301    for (ci, c) in cells.iter().enumerate() {
1302        if c.text.trim().is_empty() {
1303            continue;
1304        }
1305        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1306        let mut best: Option<(usize, f32)> = None;
1307        for (i, r) in regions.iter().enumerate() {
1308            if !claims_cells(r) {
1309                continue;
1310            }
1311            let ov = inter(r, c.l, c.t, c.r, c.b) / ca;
1312            if ov > 0.2 && best.is_none_or(|(_, b)| ov > b) {
1313                best = Some((i, ov));
1314            }
1315        }
1316        if let Some((i, _)) = best {
1317            owned[i].push(ci);
1318        }
1319    }
1320    owned
1321}
1322
1323/// docling's regular-cluster refinement after cell assignment
1324/// (`LayoutPostprocessor._process_regular_clusters`, #419), run once the page's
1325/// cells are final and before reading order:
1326///
1327/// 1. every regular region's box becomes the union of the cells it claimed
1328///    (`_adjust_cluster_bboxes` — a regular cluster's bbox *is* its cells'
1329///    bbox; a table's is the union with the model box, and pictures keep
1330///    theirs, so neither is touched here);
1331/// 2. a regular region that claimed no cell is dropped (`keep_empty_clusters`
1332///    is off; a `formula` is kept, as upstream keeps it);
1333/// 3. an orphan text region (`score == 0.0`, from [`add_orphan_regions`]) that
1334///    now sits > 0.8 inside another regular region's fitted box is folded into
1335///    it (`_remove_overlapping_clusters` at containment 0.8, the larger box
1336///    winning the group) — up to three rounds, like upstream's loop.
1337///
1338/// Why it matters: the layout model's box can end partway through a line. That
1339/// line fails the 0.2 claim and becomes an orphan — recoverable — but the
1340/// *model* box still overlaps the orphan's line by a few points, so the
1341/// reading-order graph, which links only strictly-above pairs, gets no edge
1342/// between them and may emit the next paragraph first, stranding the line
1343/// after the paragraph it belongs in (1540 of 6050 text blocks on the #419
1344/// book began mid-sentence). Fitted to its cells, the box ends on a line
1345/// boundary and the orphan slots in between; an orphan the fitted box
1346/// swallows joins the paragraph outright. Cell assignment is untouched: a
1347/// region's fitted box contains every cell it claimed, so
1348/// [`region_texts_exclusive`] hands it the same cells afterwards.
1349///
1350/// A page with no cells yet (a scan before OCR) is left alone: dropping every
1351/// text region for want of cells would be wrong, and the OCR paths call this
1352/// again once the cells exist.
1353pub fn fit_regions_to_cells(regions: &mut Vec<Region>, cells: &[TextCell]) {
1354    if !cells.iter().any(|c| !c.text.trim().is_empty()) {
1355        return;
1356    }
1357    for _ in 0..3 {
1358        let owned = assign_cells(regions, cells);
1359        let mut fitted: Vec<Region> = Vec::with_capacity(regions.len());
1360        for (r, own) in regions.iter().zip(&owned) {
1361            if !claims_cells(r) {
1362                fitted.push(r.clone());
1363                continue;
1364            }
1365            if own.is_empty() {
1366                if r.label == "formula" {
1367                    fitted.push(r.clone());
1368                }
1369                continue;
1370            }
1371            let mut f = r.clone();
1372            f.l = own
1373                .iter()
1374                .map(|&i| cells[i].l)
1375                .fold(f32::INFINITY, f32::min);
1376            f.t = own
1377                .iter()
1378                .map(|&i| cells[i].t)
1379                .fold(f32::INFINITY, f32::min);
1380            f.r = own
1381                .iter()
1382                .map(|&i| cells[i].r)
1383                .fold(f32::NEG_INFINITY, f32::max);
1384            f.b = own
1385                .iter()
1386                .map(|&i| cells[i].b)
1387                .fold(f32::NEG_INFINITY, f32::max);
1388            fitted.push(f);
1389        }
1390        let mut changed = fitted.len() != regions.len();
1391        // Fold orphans into the regular region whose fitted box holds them.
1392        let mut drop = vec![false; fitted.len()];
1393        for i in 0..fitted.len() {
1394            let o = &fitted[i];
1395            if !(o.score == 0.0 && o.label == "text") {
1396                continue;
1397            }
1398            let oa = area(o.l, o.t, o.r, o.b).max(1.0);
1399            let mut best: Option<(usize, f32)> = None;
1400            for (j, r) in fitted.iter().enumerate() {
1401                if j == i || drop[j] || r.score == 0.0 || !claims_cells(r) {
1402                    continue;
1403                }
1404                let ov = inter(r, o.l, o.t, o.r, o.b) / oa;
1405                if ov > 0.8 && best.is_none_or(|(_, b)| ov > b) {
1406                    best = Some((j, ov));
1407                }
1408            }
1409            if let Some((j, _)) = best {
1410                let (l, t, r, b) = (o.l, o.t, o.r, o.b);
1411                let host = &mut fitted[j];
1412                host.l = host.l.min(l);
1413                host.t = host.t.min(t);
1414                host.r = host.r.max(r);
1415                host.b = host.b.max(b);
1416                drop[i] = true;
1417                changed = true;
1418            }
1419        }
1420        let mut drop = drop.into_iter();
1421        fitted.retain(|_| !drop.next().expect("aligned"));
1422        *regions = fitted;
1423        if !changed {
1424            break;
1425        }
1426    }
1427}
1428
1429/// Join a prefiltered cell list into the region's text (docling's
1430/// `sanitize_text` on the docling-parse path, gap-aware band join on legacy).
1431fn cells_text(mut inside: Vec<&TextCell>) -> String {
1432    // Quantize the top coordinate into ~line bands so cells on the same line
1433    // sort in reading order; this is a strict total order (a raw fuzzy comparator
1434    // is not transitive and makes Rust's sort panic). For a right-to-left
1435    // (Arabic-majority) region, cells on a line read right→left, so sort the band
1436    // by descending left edge.
1437    let band = inside
1438        .iter()
1439        .map(|c| (c.b - c.t).abs())
1440        .fold(0.0f32, f32::max)
1441        .max(1.0);
1442    let arabic = inside
1443        .iter()
1444        .flat_map(|c| c.text.chars())
1445        .filter(|&c| ('\u{0600}'..='\u{06FF}').contains(&c))
1446        .count();
1447    let latin = inside
1448        .iter()
1449        .flat_map(|c| c.text.chars())
1450        .filter(|c| c.is_ascii_alphabetic())
1451        .count();
1452    let rtl = arabic > latin;
1453    let dp = crate::pdfium_backend::use_dp_lines();
1454    if dp {
1455        // docling orders a cluster's cells by their docling-parse cell index
1456        // alone (`LayoutPostprocessor._sort_cells`: `sorted(cells, key=c.index)`)
1457        // — the sanitizer's output order, which our `cells` slice already is.
1458        // No geometric re-sort: normal_4pages' big section numerals paint
1459        // *after* their heading text, and docling's `## 들어가며 1` (numeral
1460        // last) only falls out of pure index order — a band sort dragged the
1461        // numeral to the front. The overlap-grouped line restore this replaced
1462        // measured strictly worse on the corpus (it fixed nothing the index
1463        // order broke, and broke the numerals).
1464    } else {
1465        inside.sort_by_key(|c| {
1466            let x = (c.l * 10.0) as i64;
1467            ((c.t / band).round() as i64, if rtl { -x } else { x })
1468        });
1469    }
1470    let joined = if dp {
1471        // docling's `PageAssembleModel.sanitize_text`, ported verbatim over the
1472        // parse-index-ordered lines: append a separating space to a line —
1473        // unless it ends with `-`. A dash-ending line whose last word and the
1474        // next line's first word are both alphanumeric is a wrapped word: the
1475        // dash is dropped and the lines fuse (`platforms-` + `reflects` →
1476        // `platformsreflects`, `pp. 545-` + `561` → `545561`). Any other
1477        // dash-ending line — e.g. the *bare* `-` cell a superscript ORCID or an
1478        // inline `–` bullet splits off (its word list is empty, so the fuse
1479        // test fails) — keeps its dash and still takes no trailing space:
1480        // `[0000` `-` `0002` joins as docling's `[0000 -0002`, and the OTSL
1481        // list's `-` + `"C" cell -` + `a new table cell` collapses to
1482        // `-"C" cell a new table cell`. Our cells still carry the raw dash
1483        // family (docling-parse normalizes to `-` before this; clean_text does
1484        // it after), so the endswith test matches them all.
1485        let texts: Vec<&str> = inside
1486            .iter()
1487            .map(|c| c.text.trim())
1488            // Skip whitespace-only cells (a justified line's trailing space
1489            // glyph): an empty line would double the separator.
1490            .filter(|t| !t.is_empty())
1491            .collect();
1492        let last_word_alnum = |s: &str| {
1493            s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
1494                .rfind(|w| !w.is_empty())
1495                .is_some_and(|w| w.chars().all(char::is_alphanumeric))
1496        };
1497        let first_word_alnum = |s: &str| {
1498            s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
1499                .find(|w| !w.is_empty())
1500                .is_some_and(|w| w.chars().all(char::is_alphanumeric))
1501        };
1502        let mut out = String::new();
1503        for (i, t) in texts.iter().enumerate() {
1504            if i > 0 {
1505                let prev = texts[i - 1];
1506                let dashish = matches!(
1507                    prev.chars().last(),
1508                    Some(
1509                        '-' | '\u{2010}'
1510                            | '\u{2011}'
1511                            | '\u{2012}'
1512                            | '\u{2013}'
1513                            | '\u{2014}'
1514                            | '\u{2015}'
1515                            | '\u{2212}'
1516                    )
1517                );
1518                // docling#4052 (2.122): a dash only splits a word when it is
1519                // *attached* to one — the character before it is alphanumeric.
1520                // A dash that follows whitespace (a separator dash, a bullet
1521                // marker, a wrapped `-prefixed` token, the bare `-` cell an
1522                // ORCID splits off) is a literal character: it is kept and the
1523                // lines join with the ordinary space.
1524                let attached = prev.chars().rev().nth(1).is_some_and(char::is_alphanumeric);
1525                if dashish && attached {
1526                    if last_word_alnum(prev) && first_word_alnum(t) {
1527                        out.pop(); // wrapped word: fuse without the dash
1528                    }
1529                    // an attached dash never takes a separating space
1530                } else {
1531                    out.push(' ');
1532                }
1533            }
1534            out.push_str(t);
1535        }
1536        out
1537    } else {
1538        // Legacy reconstruction: join same-band cells with a space only across a
1539        // real gap, because it can split a word into abutting segments
1540        // (`الت`|`ي` → `التي`).
1541        let mut out = String::new();
1542        let mut prev: Option<&&TextCell> = None;
1543        for c in &inside {
1544            let t = c.text.trim();
1545            if t.is_empty() {
1546                continue;
1547            }
1548            if let Some(p) = prev {
1549                let same_band = ((p.t / band).round() as i64) == ((c.t / band).round() as i64);
1550                let h = (c.b - c.t).abs().max((p.b - p.t).abs()).max(1.0);
1551                let gap = if rtl { p.l - c.r } else { c.l - p.r };
1552                if !same_band || gap > h * 0.25 {
1553                    out.push(' ');
1554                }
1555            }
1556            out.push_str(t);
1557            prev = Some(c);
1558        }
1559        out
1560    };
1561    clean_text(&joined)
1562}
1563
1564/// Tighten the spaces pdfium leaves around tight punctuation in a code line
1565/// (`console .log` → `console.log`, `add (3 , 5)` → `add(3, 5)`), matching
1566/// docling-parse's source spacing.
1567fn tighten_code_punct(s: &str) -> String {
1568    s.replace(" .", ".")
1569        .replace(" ,", ",")
1570        .replace(" ;", ";")
1571        .replace(" )", ")")
1572        .replace(" (", "(")
1573}
1574
1575/// Assemble a **code** region's text with its line structure preserved.
1576///
1577/// Unlike [`region_text`] — which joins every cell with a single space, the right
1578/// thing for prose reflow — a code block's line breaks and indentation are
1579/// significant. The `code_cells` are already one physical source line each
1580/// (grouped space-glyph-only, so monospace runs keep their spacing), so this:
1581///
1582/// 1. groups the cells into vertical line bands and orders them top→bottom,
1583///    left→right;
1584/// 2. joins the lines with `\n` (rather than spaces), keeping the carriage
1585///    returns; and
1586/// 3. reconstructs each line's leading indentation from its left offset, in units
1587///    of the block's estimated monospace character width, so nesting survives.
1588///
1589/// Typography is normalized per line via [`clean_text`] (smart quotes, dashes,
1590/// ellipsis), which never merges lines. Returns an empty string if the region has
1591/// no code cells (the caller falls back to the prose text).
1592fn code_region_text(region: &Region, cells: &[TextCell]) -> String {
1593    let mut inside: Vec<&TextCell> = cells
1594        .iter()
1595        .filter(|c| {
1596            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1597            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1598        })
1599        .filter(|c| !c.text.trim().is_empty())
1600        .collect();
1601    if inside.is_empty() {
1602        return String::new();
1603    }
1604
1605    // Quantize the top edge into ~line bands (like `region_text`), then order the
1606    // cells by band (top→bottom) and, within a band, by left edge.
1607    let band = inside
1608        .iter()
1609        .map(|c| (c.b - c.t).abs())
1610        .fold(0.0f32, f32::max)
1611        .max(1.0);
1612    let line_of = |c: &TextCell| (c.t / band).round() as i64;
1613    inside.sort_by_key(|c| (line_of(c), (c.l * 10.0) as i64));
1614
1615    // Estimate one monospace character's width (total ink width / total glyphs) to
1616    // convert a line's left offset into a count of leading spaces. Measured over
1617    // all lines so a single short line can't skew it.
1618    let (mut total_w, mut total_chars) = (0.0f32, 0usize);
1619    for c in &inside {
1620        let n = c.text.trim().chars().count();
1621        if n > 0 {
1622            total_w += (c.r - c.l).max(0.0);
1623            total_chars += n;
1624        }
1625    }
1626    let char_w = if total_chars > 0 {
1627        (total_w / total_chars as f32).max(1.0)
1628    } else {
1629        1.0
1630    };
1631    // The block's own left margin is the zero-indent baseline.
1632    let base_l = inside.iter().map(|c| c.l).fold(f32::INFINITY, f32::min);
1633
1634    let mut lines: Vec<String> = Vec::new();
1635    let mut cur: Option<i64> = None;
1636    for c in &inside {
1637        // Tighten pdfium's spaced punctuation per line (on the trimmed content, so
1638        // the reconstructed leading indentation is never nibbled).
1639        let text = tighten_code_punct(&clean_text(c.text.trim()));
1640        if Some(line_of(c)) == cur {
1641            // A second cell sharing this band (rare — e.g. split columns): keep it
1642            // on the same source line, separated by a space.
1643            if let Some(last) = lines.last_mut() {
1644                last.push(' ');
1645                last.push_str(&text);
1646            }
1647            continue;
1648        }
1649        let indent = ((c.l - base_l) / char_w).round().max(0.0) as usize;
1650        lines.push(format!("{}{}", " ".repeat(indent), text));
1651        cur = Some(line_of(c));
1652    }
1653    lines.join("\n")
1654}
1655
1656/// Reconstruct a table's grid geometrically from the text cells inside its
1657/// region: cluster cells into rows (by vertical centre) and columns (by clustered
1658/// left edges), then place each cell. A model-free stand-in for TableFormer that
1659/// recovers grid-aligned tables from the precise PDF text layer (it does not
1660/// resolve row/column spans).
1661pub fn reconstruct_table(region: &Region, cells: &[TextCell]) -> Vec<Vec<String>> {
1662    let mut inside: Vec<&TextCell> = cells
1663        .iter()
1664        .filter(|c| {
1665            let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1666            inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1667        })
1668        .collect();
1669    if inside.is_empty() {
1670        return Vec::new();
1671    }
1672    inside.sort_by(|a, b| a.t.total_cmp(&b.t));
1673
1674    // Rows: consecutive cells whose vertical centre is within ~0.7 line height.
1675    let mut rows: Vec<(f32, Vec<&TextCell>)> = Vec::new();
1676    for c in &inside {
1677        let cyc = (c.t + c.b) / 2.0;
1678        let lh = (c.b - c.t).abs().max(1.0);
1679        if let Some((ryc, row)) = rows.last_mut() {
1680            if (cyc - *ryc).abs() < lh * 0.7 {
1681                row.push(c);
1682                continue;
1683            }
1684        }
1685        rows.push((cyc, vec![c]));
1686    }
1687
1688    // Columns: cluster left edges (merge those within a tolerance).
1689    let tol = {
1690        let mut hs: Vec<f32> = inside.iter().map(|c| (c.b - c.t).abs()).collect();
1691        hs.sort_by(f32::total_cmp);
1692        hs[hs.len() / 2].max(4.0) * 1.5
1693    };
1694    let mut lefts: Vec<f32> = inside.iter().map(|c| c.l).collect();
1695    lefts.sort_by(f32::total_cmp);
1696    let mut col_starts: Vec<f32> = Vec::new();
1697    for l in lefts {
1698        if col_starts.last().is_none_or(|&last| l - last > tol) {
1699            col_starts.push(l);
1700        }
1701    }
1702    let ncols = col_starts.len().max(1);
1703    let col_of = |l: f32| -> usize {
1704        col_starts
1705            .iter()
1706            .rposition(|&s| l + tol * 0.5 >= s)
1707            .unwrap_or(0)
1708            .min(ncols - 1)
1709    };
1710
1711    let mut grid = Vec::with_capacity(rows.len());
1712    for (_, mut row) in rows {
1713        row.sort_by(|a, b| a.l.total_cmp(&b.l));
1714        let mut cols = vec![String::new(); ncols];
1715        for c in row {
1716            let ci = col_of(c.l);
1717            // Strip the wrap-hyphen control char so it never lands in a cell.
1718            let t = c.text.trim().replace(['\u{2}', '\u{ad}'], "");
1719            if cols[ci].is_empty() {
1720                cols[ci] = t;
1721            } else {
1722                cols[ci].push(' ');
1723                cols[ci].push_str(&t);
1724            }
1725        }
1726        grid.push(cols);
1727    }
1728    grid
1729}
1730
1731/// Does the geometric reconstruction of a table look trustworthy enough to use
1732/// as-is, instead of paying for TableFormer?
1733///
1734/// [`reconstruct_table`] derives columns by clustering cell **left edges**. On a
1735/// clean grid that is exact, but when a column's entries are not left-aligned
1736/// (or the OCR boxes wobble) the clustering splits one real column into several,
1737/// and the result is a wide, mostly-empty grid — the "spurious empty columns"
1738/// failure TableFormer exists to fix.
1739///
1740/// Two symptoms separate the two cases, and both are properties of the grid
1741/// alone (no model needed):
1742/// * **density** — a real table is mostly full; a split-up one is mostly holes;
1743/// * **thin columns** — a column carrying at most one entry across several rows
1744///   is almost always a split artefact rather than a real column.
1745///
1746/// Deliberately conservative: it answers `true` only for grids that are plainly
1747/// well-formed, so the expensive path stays the default whenever there is doubt.
1748/// A caller that skips TableFormer on `true` trades no quality for the time.
1749pub fn geometric_table_is_reliable(rows: &[Vec<String>]) -> bool {
1750    let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
1751    // Fewer than two columns is not a grid this heuristic can vouch for: it is
1752    // exactly the shape a collapsed table takes, and TableFormer may recover
1753    // real structure from it.
1754    if rows.len() < 2 || ncols < 2 {
1755        return false;
1756    }
1757    let filled = |c: &String| !c.trim().is_empty();
1758    let total = rows.len() * ncols;
1759    let full = rows.iter().flatten().filter(|c| filled(c)).count();
1760    if (full as f32) < MIN_TABLE_FILL * total as f32 {
1761        return false;
1762    }
1763    // A column used by at most one row, when there are rows enough to tell.
1764    if rows.len() >= 3 {
1765        for ci in 0..ncols {
1766            let used = rows
1767                .iter()
1768                .filter(|r| r.get(ci).is_some_and(filled))
1769                .count();
1770            if used <= 1 {
1771                return false;
1772            }
1773        }
1774    }
1775    true
1776}
1777
1778/// Share of a geometric grid's cells that must carry text for it to be trusted
1779/// without TableFormer. Chosen well above the density a left-edge split
1780/// produces (those land nearer a third) and below what a genuine table with a
1781/// few blank cells reaches.
1782const MIN_TABLE_FILL: f32 = 0.6;
1783
1784/// The union bbox of the text cells assigned to a region (same >50%-overlap
1785/// rule as [`region_text`]), or `None` when no cell lands in it. docling's
1786/// LayoutPostprocessor shrinks a regular cluster's bbox to its cells, and the
1787/// enrichment crops are taken from that cell-tight box — cropping the raw
1788/// detector box instead hands the VLM surrounding chrome (e.g. the `Listing N:`
1789/// caption under a code block) that changes its output.
1790pub fn region_cell_bbox(region: &Region, cells: &[TextCell]) -> Option<[f32; 4]> {
1791    let mut bbox: Option<[f32; 4]> = None;
1792    for c in cells {
1793        let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1794        if inter(region, c.l, c.t, c.r, c.b) / ca <= 0.5 {
1795            continue;
1796        }
1797        bbox = Some(match bbox {
1798            None => [c.l, c.t, c.r, c.b],
1799            Some([l, t, r, b]) => [l.min(c.l), t.min(c.t), r.max(c.r), b.max(c.b)],
1800        });
1801    }
1802    bbox
1803}
1804
1805/// One region's enrichment-model result, produced by the pipeline's opt-in
1806/// passes (issue #76) and applied during assembly.
1807#[derive(Debug, Clone)]
1808pub enum Enrichment {
1809    /// DocumentPictureClassifier predictions, descending confidence.
1810    PictureClasses(Vec<PictureClass>),
1811    /// CodeFormulaV2 output for a `code` region: the rewritten source text and
1812    /// the `<_language_>` prefix (when the model emitted one).
1813    Code {
1814        language: Option<String>,
1815        text: String,
1816    },
1817    /// CodeFormulaV2 output for a `formula` region: the decoded LaTeX.
1818    Formula { latex: String },
1819}
1820
1821/// Crop a region (page points, already expanded by the caller if needed) from
1822/// the rendered page image and resize it to `target_scale` pixels per point —
1823/// the enrichment-model equivalent of docling's
1824/// `page.get_image(scale=…, cropbox=…)`, sourced from the existing
1825/// [`crate::pdfium_backend::RENDER_SCALE`] render instead of a fresh pdfium
1826/// pass (the page bitmap is already the exact docling render at scale 2).
1827#[cfg(feature = "ml")]
1828pub fn crop_region_scaled(page: &PdfPage, bbox: [f32; 4], target_scale: f32) -> Option<RgbImage> {
1829    let s = page.scale;
1830    let [l, t, r, b] = bbox;
1831    let (iw, ih) = (page.image.width(), page.image.height());
1832    let x = (l * s).max(0.0) as u32;
1833    let y = (t * s).max(0.0) as u32;
1834    if x >= iw || y >= ih {
1835        return None;
1836    }
1837    let w = (((r - l.max(0.0)) * s) as u32).min(iw - x);
1838    let h = (((b - t.max(0.0)) * s) as u32).min(ih - y);
1839    if w == 0 || h == 0 {
1840        return None;
1841    }
1842    let crop = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
1843    // docling renders the crop at `target_scale` directly; from the scale-2
1844    // page render that is a resize to the same pixel geometry
1845    // (`round(width_points * scale)`, PIL's BICUBIC ≙ CatmullRom).
1846    let tw = ((w as f32 / s) * target_scale).round().max(1.0) as u32;
1847    let th = ((h as f32 / s) * target_scale).round().max(1.0) as u32;
1848    if (tw, th) == (w, h) {
1849        return Some(crop);
1850    }
1851    Some(image::imageops::resize(
1852        &crop,
1853        tw,
1854        th,
1855        image::imageops::FilterType::CatmullRom,
1856    ))
1857}
1858
1859/// Crop a layout region from the rendered page image and encode it as PNG (the
1860/// figure bytes docling stores on a `PictureItem`). Region coordinates are page
1861/// points; the image is rendered at `page.scale`.
1862#[cfg(feature = "ocr-prep")]
1863fn crop_region(page: &PdfPage, region: &Region) -> Option<PictureImage> {
1864    let s = page.scale;
1865    let (iw, ih) = (page.image.width(), page.image.height());
1866    let x = (region.l * s).max(0.0) as u32;
1867    let y = (region.t * s).max(0.0) as u32;
1868    if x >= iw || y >= ih {
1869        return None;
1870    }
1871    let w = (((region.r - region.l) * s) as u32).min(iw - x);
1872    let h = (((region.b - region.t) * s) as u32).min(ih - y);
1873    if w == 0 || h == 0 {
1874        return None;
1875    }
1876    let sub = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
1877    let mut buf = std::io::Cursor::new(Vec::new());
1878    sub.write_to(&mut buf, image::ImageFormat::Png).ok()?;
1879    Some(PictureImage {
1880        mimetype: "image/png".into(),
1881        width: w,
1882        height: h,
1883        data: buf.into_inner(),
1884    })
1885}
1886
1887/// For each `picture` region, find the `caption` region closest below it (and
1888/// horizontally overlapping); docling pairs them and emits the caption first.
1889/// Each caption is claimed by at most one picture.
1890fn pair_captions(regions: &[Region]) -> Vec<Option<usize>> {
1891    let mut pairs = vec![None; regions.len()];
1892    let mut taken = vec![false; regions.len()];
1893    for (pi, p) in regions.iter().enumerate() {
1894        if p.label != "picture" {
1895            continue;
1896        }
1897        let mut best: Option<(usize, f32)> = None;
1898        for (ci, c) in regions.iter().enumerate() {
1899            if c.label != "caption" || taken[ci] {
1900                continue;
1901            }
1902            let line_h = (c.b - c.t).abs().max(1.0);
1903            let gap = c.t - p.b; // caption sits below the picture
1904            let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
1905            if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
1906                let dist = gap.abs();
1907                if best.is_none_or(|(_, bd)| dist < bd) {
1908                    best = Some((ci, dist));
1909                }
1910            }
1911        }
1912        if let Some((ci, _)) = best {
1913            pairs[pi] = Some(ci);
1914            taken[ci] = true;
1915        }
1916    }
1917    pairs
1918}
1919
1920/// Pair each `code` region with the `caption` region just **above** it (a
1921/// `Listing N:` label). docling renders the code block first, then its caption,
1922/// so the caption is consumed from its own (earlier) reading-order slot and
1923/// re-emitted after the code.
1924fn pair_code_captions(regions: &[Region]) -> Vec<Option<usize>> {
1925    let mut pairs = vec![None; regions.len()];
1926    let mut taken = vec![false; regions.len()];
1927    for (pi, p) in regions.iter().enumerate() {
1928        if p.label != "code" {
1929            continue;
1930        }
1931        let mut best: Option<(usize, f32)> = None;
1932        for (ci, c) in regions.iter().enumerate() {
1933            if c.label != "caption" || taken[ci] {
1934                continue;
1935            }
1936            let line_h = (c.b - c.t).abs().max(1.0);
1937            let gap = p.t - c.b; // caption sits above the code
1938            let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
1939            if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
1940                let dist = gap.abs();
1941                if best.is_none_or(|(_, bd)| dist < bd) {
1942                    best = Some((ci, dist));
1943                }
1944            }
1945        }
1946        if let Some((ci, _)) = best {
1947            pairs[pi] = Some(ci);
1948            taken[ci] = true;
1949        }
1950    }
1951    pairs
1952}
1953
1954/// Pair each `table`/`document_index` region with its `caption` (#265) the way
1955/// docling's `ReadingOrderPredictor._find_to_captions` does: by **reading-order
1956/// adjacency**, not geometry. A caption claims the media element
1957/// (table/picture/code) immediately next to it in the ordered region sequence,
1958/// and only when exactly one side holds one — a caption sandwiched between two
1959/// media elements stays unattached, and a text paragraph between caption and
1960/// table breaks the bond. This is what lets a flush-left "Table 3: …" label
1961/// bind a centered grid it doesn't horizontally overlap, while a caption in
1962/// the neighbouring column of a two-column page — geometrically close — never
1963/// pairs across the gutter. Runs after the picture and code pairings (the
1964/// picture/code arms of the same upstream matcher), so a caption they claimed
1965/// stays claimed. docling attaches these as `TableItem.captions` refs; the
1966/// paired caption is consumed from its own reading-order slot and rides on the
1967/// table node instead.
1968fn pair_table_captions(regions: &[Region], taken: &mut [bool]) -> Vec<Option<usize>> {
1969    let is_media = |label: &str| is_table_like(label) || matches!(label, "picture" | "code");
1970    let mut pairs: Vec<Option<usize>> = vec![None; regions.len()];
1971    for ci in 0..regions.len() {
1972        if regions[ci].label != "caption" || taken[ci] {
1973            continue;
1974        }
1975        // Furniture (headers/footers, form chrome) is not part of docling's
1976        // body-element sequence, so it neither bonds nor blocks.
1977        let prev = regions[..ci].iter().rposition(|r| !is_skipped(r.label));
1978        let next = regions[ci + 1..]
1979            .iter()
1980            .position(|r| !is_skipped(r.label))
1981            .map(|off| ci + 1 + off);
1982        let prev_media = prev.is_some_and(|j| is_media(regions[j].label));
1983        let next_media = next.is_some_and(|j| is_media(regions[j].label));
1984        let target = match (prev_media, next_media) {
1985            (true, false) => prev,
1986            (false, true) => next,
1987            // Ambiguous (media on both sides) or no media at all: leave the
1988            // caption in its own reading-order slot, as docling does.
1989            _ => None,
1990        };
1991        if let Some(ti) = target {
1992            // A first claim wins (a table with captions above *and* below
1993            // keeps the earlier one — docling's nearest-first tiebreak).
1994            if is_table_like(regions[ti].label) && pairs[ti].is_none() {
1995                pairs[ti] = Some(ci);
1996                taken[ci] = true;
1997            }
1998        }
1999    }
2000    pairs
2001}
2002
2003/// Assemble one page from its (already overlap-resolved) layout regions and
2004/// text cells.
2005/// Normalize a layout region (page points, top-left origin) to DocLang's 0–511
2006/// location grid: `clamp(round(512 · coord / page_dim), 0, 511)`, per axis,
2007/// order `[x0, y0, x1, y1]`. Mirrors docling_core's
2008/// `_doclang_utils._create_location_tokens_for_bbox` (resolution 512) so the
2009/// emitted `<location>` tokens line up with the Python groundtruth. Our heron
2010/// cluster boxes match docling's to within ~1 grid unit; the residual (mainly
2011/// the aspect-ratio-stretch vs letterbox preprocessing difference) is absorbed
2012/// by the conformance harness's geometry tolerance.
2013fn norm_loc(region: &Region, page_w: f32, page_h: f32) -> [u16; 4] {
2014    let q = |v: f32, dim: f32| -> u16 {
2015        if dim <= 0.0 {
2016            return 0;
2017        }
2018        let g = (512.0 * (v as f64) / (dim as f64)).round() as i64;
2019        g.clamp(0, 511) as u16
2020    };
2021    [
2022        q(region.l, page_w),
2023        q(region.t, page_h),
2024        q(region.r, page_w),
2025        q(region.b, page_h),
2026    ]
2027}
2028
2029/// Wrap a node in its layout provenance so the DocLang serializer emits the four
2030/// `<location>` tokens as the element's head (Markdown/JSON render `inner`
2031/// unchanged).
2032fn located(loc: [u16; 4], inner: Node) -> Node {
2033    Node::Located {
2034        location: loc,
2035        inner: Box::new(inner),
2036    }
2037}
2038
2039/// Stamp the real 1-based page number onto a page's leading marker (see
2040/// [`assemble_page`], which emits it with `page_no: 0` because only the
2041/// document-level collector knows the true index — `--pages` windows shift it).
2042pub fn stamp_page_no(nodes: &mut [Node], page_no: usize) {
2043    if let Some(Node::PageInfo { page_no: p, .. }) = nodes.first_mut() {
2044        *p = page_no;
2045    }
2046}
2047
2048/// A dense table grid plus its first-class cells (#240): `rows` is the text
2049/// grid every serializer renders (spans replicate their anchor's text);
2050/// `cells` are the docling-parity per-cell records (text, page-point bbox,
2051/// span rectangle, OTSL header roles). Produced by the TableFormer paths
2052/// (`tf_core`); lives in this always-compiled module so the pure-text (wasm
2053/// `pdf-text`) build sees the type.
2054#[derive(Clone, Debug)]
2055pub struct TableGrid {
2056    pub rows: Vec<Vec<String>>,
2057    pub cells: Vec<docling_core::TableCell>,
2058}
2059
2060/// docling's `_RICH_CELL_PICTURE_COVERAGE_THRESHOLD`.
2061const RICH_CELL_PICTURE_COVERAGE: f32 = 0.8;
2062
2063/// docling `ReadingOrderModel._match_table_pictures` (#3906, 2.118.1): every
2064/// picture ≥ 80 % inside a TableFormer-structured table on the page is matched
2065/// to the cell covering it, and returned per table as `cell index → pictures`.
2066/// A picture that pairs with a caption stays a standalone figure (upstream
2067/// would nest it and lose the caption; keeping the caption is the better
2068/// failure). Tables without first-class cells (geometric fallback) have no cell
2069/// boxes to match against and nest nothing.
2070fn match_table_pictures(
2071    regions: &[Region],
2072    table_rows: &[Option<TableGrid>],
2073    caption_for: &[Option<usize>],
2074) -> std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> {
2075    let mut out: std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> =
2076        std::collections::HashMap::new();
2077    for (p, pic) in regions.iter().enumerate() {
2078        if pic.label != "picture" || caption_for.get(p).is_some_and(Option::is_some) {
2079            continue;
2080        }
2081        let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2082        let mut best: Option<(f32, usize, usize)> = None; // (coverage, table, cell)
2083        for (t, tbl) in regions.iter().enumerate() {
2084            if !is_table_like(tbl.label) {
2085                continue;
2086            }
2087            let Some(grid) = table_rows.get(t).and_then(Option::as_ref) else {
2088                continue;
2089            };
2090            if inter(pic, tbl.l, tbl.t, tbl.r, tbl.b) / pa < RICH_CELL_PICTURE_COVERAGE {
2091                continue;
2092            }
2093            if let Some((cov, cell)) = match_picture_to_cell(pic, &grid.cells) {
2094                if best.is_none_or(|(b, _, _)| cov > b) {
2095                    best = Some((cov, t, cell));
2096                }
2097            }
2098        }
2099        if let Some((_, t, cell)) = best {
2100            let entry = out.entry(t).or_default();
2101            match entry.iter_mut().find(|(c, _)| *c == cell) {
2102                Some((_, pics)) => pics.push(p),
2103                None => entry.push((cell, vec![p])),
2104            }
2105        }
2106    }
2107    out
2108}
2109
2110/// docling `_match_picture_to_table_cell`: among the cells covering ≥ 80 % of
2111/// the picture, prefer the one at the picture's inferred grid position (the
2112/// row / column whose median cell center is nearest the picture's center —
2113/// cell boxes can overlap across logical rows and columns), else the best
2114/// coverage. Returns `(coverage, cell index)`.
2115fn match_picture_to_cell(pic: &Region, cells: &[docling_core::TableCell]) -> Option<(f32, usize)> {
2116    let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2117    let cover = |b: &[f32; 4]| inter(pic, b[0], b[1], b[2], b[3]) / pa;
2118    let eligible: Vec<(f32, usize)> = cells
2119        .iter()
2120        .enumerate()
2121        .filter_map(|(i, c)| {
2122            let b = c.bbox.as_ref()?;
2123            let cov = cover(b);
2124            (cov >= RICH_CELL_PICTURE_COVERAGE).then_some((cov, i))
2125        })
2126        .collect();
2127    if eligible.is_empty() {
2128        return None;
2129    }
2130    let mut row_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2131    let mut col_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2132    for c in cells {
2133        let Some(b) = c.bbox.as_ref() else { continue };
2134        for r in c.start_row..c.start_row + c.row_span {
2135            row_centers.entry(r).or_default().push((b[1] + b[3]) / 2.0);
2136        }
2137        for k in c.start_col..c.start_col + c.col_span {
2138            col_centers.entry(k).or_default().push((b[0] + b[2]) / 2.0);
2139        }
2140    }
2141    let median = |v: &mut Vec<f32>| -> f32 {
2142        v.sort_by(f32::total_cmp);
2143        let n = v.len();
2144        if n % 2 == 1 {
2145            v[n / 2]
2146        } else {
2147            (v[n / 2 - 1] + v[n / 2]) / 2.0
2148        }
2149    };
2150    let (px, py) = ((pic.l + pic.r) / 2.0, (pic.t + pic.b) / 2.0);
2151    let nearest = |centers: &mut std::collections::BTreeMap<usize, Vec<f32>>, target: f32| {
2152        centers
2153            .iter_mut()
2154            .map(|(&i, v)| (i, (median(v) - target).abs()))
2155            .min_by(|a, b| a.1.total_cmp(&b.1))
2156            .map(|(i, _)| i)
2157    };
2158    let row = nearest(&mut row_centers, py);
2159    let col = nearest(&mut col_centers, px);
2160    let logical: Vec<(f32, usize)> = eligible
2161        .iter()
2162        .copied()
2163        .filter(|&(_, i)| {
2164            let c = &cells[i];
2165            row.is_some_and(|r| c.start_row <= r && r < c.start_row + c.row_span)
2166                && col.is_some_and(|k| c.start_col <= k && k < c.start_col + c.col_span)
2167        })
2168        .collect();
2169    let pool = if logical.is_empty() {
2170        &eligible
2171    } else {
2172        &logical
2173    };
2174    // Python's `max` over `(coverage, cell_index, cell)` tuples: highest
2175    // coverage, ties to the higher index.
2176    pool.iter()
2177        .copied()
2178        .max_by(|a, b| a.0.total_cmp(&b.0).then(a.1.cmp(&b.1)))
2179}
2180
2181/// The DocLang structure overlay derived from first-class cells: span
2182/// continuations (`lcel`/`ucel`/`xcel`) and per-cell header roles, so the
2183/// PDF path's DCLX carries real spans instead of a flat grid.
2184fn structure_from_cells(
2185    cells: &[docling_core::TableCell],
2186    nrows: usize,
2187    ncols: usize,
2188) -> docling_core::TableStructure {
2189    let grid = || vec![vec![false; ncols]; nrows];
2190    let mut col_cont = grid();
2191    let mut row_cont = grid();
2192    let mut row_header = grid();
2193    let mut col_header = grid();
2194    for c in cells {
2195        for r in c.start_row..(c.start_row + c.row_span).min(nrows) {
2196            for k in c.start_col..(c.start_col + c.col_span).min(ncols) {
2197                col_cont[r][k] = k > c.start_col;
2198                row_cont[r][k] = r > c.start_row;
2199                row_header[r][k] = c.row_header;
2200                col_header[r][k] = c.column_header;
2201            }
2202        }
2203    }
2204    docling_core::TableStructure {
2205        header_row: Vec::new(),
2206        col_continuation: col_cont,
2207        row_continuation: row_cont,
2208        row_header,
2209        col_header,
2210    }
2211}
2212
2213pub fn assemble_page(
2214    page: &PdfPage,
2215    regions: Vec<Region>,
2216    table_rows: &[Option<TableGrid>],
2217    enrichments: &[Option<Enrichment>],
2218) -> (Vec<Node>, Vec<(String, String)>) {
2219    let mut nodes: Vec<Node> = Vec::new();
2220    // Every page opens with an invisible page marker carrying its size in
2221    // points — what the JSON export needs to build docling's `pages` map and
2222    // denormalize the 0–511 `<location>` grid into point bboxes (#171). The
2223    // page *number* is stamped by the document-level collector (which knows
2224    // the real 1-based index, `--pages` windows included); every serializer
2225    // except JSON skips the marker, so Markdown/DocLang stay byte-identical.
2226    nodes.push(Node::PageInfo {
2227        page_no: 0,
2228        width: page.width,
2229        height: page.height,
2230    });
2231    // Recover this page's hyperlinks (anchor-precise pairs for strict
2232    // Markdown; whole-item docling-parity links are baked below and their
2233    // pairs dropped from this list so strict output doesn't double-wrap).
2234    let mut links = resolve_link_anchors(page);
2235    // Pair each region with its precomputed TableFormer grid and enrichment
2236    // (indexed by original order) and order by reading order together, so they
2237    // stay aligned.
2238    type RegionItem = (Region, Option<TableGrid>, Option<Enrichment>);
2239    let mut items: Vec<RegionItem> = regions
2240        .into_iter()
2241        .enumerate()
2242        .map(|(i, r)| {
2243            (
2244                r,
2245                table_rows.get(i).cloned().flatten(),
2246                enrichments.get(i).cloned().flatten(),
2247            )
2248        })
2249        .collect();
2250    order_with_containers(&mut items, page.width, page.height, |it| &it.0);
2251    // Float a margin page number to the front of reading order (docling parity:
2252    // right_to_left_02's bottom `11` is its first item). Stable, so everything
2253    // else keeps its order; no-op on pages without such a region.
2254    let page_h = page.height;
2255    items.sort_by_key(|(r, _, _)| !is_page_number(r, &page.cells, page_h));
2256    let table_rows: Vec<Option<TableGrid>> = items.iter().map(|(_, t, _)| t.clone()).collect();
2257    let enrichments: Vec<Option<Enrichment>> = items.iter().map(|(_, _, e)| e.clone()).collect();
2258    let regions: Vec<Region> = items.into_iter().map(|(r, _, _)| r).collect();
2259    // docling emits a figure's caption *before* the image marker. Pair each
2260    // picture with the caption region nearest below it and consume that caption,
2261    // so it isn't also emitted in its own (lower) reading-order position.
2262    let caption_for = pair_captions(&regions);
2263    let code_caption_for = pair_code_captions(&regions);
2264    let mut consumed = vec![false; regions.len()];
2265    for ci in caption_for.iter().flatten() {
2266        consumed[*ci] = true;
2267    }
2268    for ci in code_caption_for.iter().flatten() {
2269        consumed[*ci] = true;
2270    }
2271    // Table captions (#265) claim from what the picture/code pairings left.
2272    let mut caption_taken = consumed.clone();
2273    let table_caption_for = pair_table_captions(&regions, &mut caption_taken);
2274    for ci in table_caption_for.iter().flatten() {
2275        consumed[*ci] = true;
2276    }
2277    // Pictures inside a table become rich-cell content (docling#3906, 2.118.1):
2278    // the picture is nested in the cell it covers and not emitted standalone.
2279    let rich_cell_pictures = match_table_pictures(&regions, &table_rows, &caption_for);
2280    for (_, pics) in rich_cell_pictures.values().flatten() {
2281        for &p in pics {
2282            consumed[p] = true;
2283        }
2284    }
2285    // A code block's language label (`XML`, `C#`, …) is chrome, not content — the
2286    // detector emits it as its own region above the code; consume it.
2287    for (i, is_label) in code_language_labels(&regions, &page.cells)
2288        .into_iter()
2289        .enumerate()
2290    {
2291        if is_label {
2292            consumed[i] = true;
2293        }
2294    }
2295
2296    // docling `ReadingOrderPredictor.predict_merges`: join a text fragment with a
2297    // following text fragment strictly to its right (an author column that wraps
2298    // into the next, a paragraph continuing in the next column) into one block —
2299    // the intra-page half of docling's reading-order merges (cross-page/vertical
2300    // continuations stay with [`merge_continuations`]). Already-consumed regions
2301    // (paired captions, code labels) are excluded.
2302    // Exclusive docling cell assignment: computed once for the ordered region
2303    // list and reused for every serialization below, so a cell can never render
2304    // in two regions.
2305    let region_texts: Vec<String> = region_texts_exclusive(&regions, &page.cells);
2306    let is_text: Vec<bool> = regions
2307        .iter()
2308        .enumerate()
2309        .map(|(i, r)| r.label == "text" && !consumed[i])
2310        .collect();
2311    let is_skip: Vec<bool> = regions
2312        .iter()
2313        .enumerate()
2314        .map(|(i, r)| {
2315            consumed[i]
2316                || matches!(
2317                    r.label,
2318                    "page_header" | "page_footer" | "table" | "picture" | "caption" | "footnote"
2319                )
2320        })
2321        .collect();
2322    let boxes: Vec<(f32, f32, f32, f32)> = regions.iter().map(|r| (r.l, r.t, r.r, r.b)).collect();
2323    if docling_core::env::flag("DOCLING_RS_DEBUG_MERGES") {
2324        for (i, r) in regions.iter().enumerate() {
2325            eprintln!(
2326                "MRG {i:2} {} text={} skip={} [{:.0},{:.0},{:.0},{:.0}] {:?}",
2327                r.label,
2328                is_text[i],
2329                is_skip[i],
2330                r.l,
2331                r.t,
2332                r.r,
2333                r.b,
2334                region_texts[i].chars().take(40).collect::<String>()
2335            );
2336        }
2337    }
2338    let mut merge_suffix: Vec<String> = vec![String::new(); regions.len()];
2339    for (head, children) in
2340        crate::reading_order::predict_merges(&boxes, &region_texts, &is_text, &is_skip)
2341            .into_iter()
2342            .enumerate()
2343    {
2344        for c in children {
2345            let t = region_texts[c].trim();
2346            if !t.is_empty() {
2347                merge_suffix[head].push(' ');
2348                merge_suffix[head].push_str(t);
2349            }
2350            consumed[c] = true;
2351        }
2352    }
2353
2354    for (i, region) in regions.iter().enumerate() {
2355        if consumed[i] {
2356            continue;
2357        }
2358        // Page headers/footers: docling emits them as furniture blocks
2359        // (`<page_header>`/`<page_footer>` with a layer + location + text) at
2360        // their reading-order position, not as body — emit them, don't skip.
2361        if matches!(region.label, "page_header" | "page_footer") {
2362            let text = region_texts[i].clone();
2363            if !text.is_empty() {
2364                nodes.push(Node::PageFurniture {
2365                    footer: region.label == "page_footer",
2366                    location: norm_loc(region, page.width, page_h),
2367                    text: md_escape(&text),
2368                });
2369            }
2370            continue;
2371        }
2372        if is_skipped(region.label) {
2373            continue;
2374        }
2375        // Layout provenance for this region, normalized to docling's 0–511 grid.
2376        let loc = norm_loc(region, page.width, page_h);
2377        if region.label == "picture" {
2378            // The figure pixels are cropped from the page render for image export.
2379            // Captions are prose: markdown-escaped like a paragraph (the JSON
2380            // export unescapes back to the raw text, matching docling).
2381            let caption = caption_for[i]
2382                .map(|ci| md_escape(&region_texts[ci]))
2383                .filter(|t| !t.is_empty());
2384            let classification = match &enrichments[i] {
2385                Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
2386                _ => None,
2387            };
2388            // Without the page render (text-layer-only build) a picture keeps
2389            // its caption/classification but carries no cropped pixels.
2390            #[cfg(feature = "ocr-prep")]
2391            let image = crate::timing::timed("crop_region", || crop_region(page, region));
2392            #[cfg(not(feature = "ocr-prep"))]
2393            let image: Option<PictureImage> = None;
2394            nodes.push(located(
2395                loc,
2396                Node::Picture {
2397                    caption,
2398                    caption_href: None,
2399                    image,
2400                    classification,
2401                    // docling's layout pipeline parents a figure's caption to
2402                    // the picture itself (#390) — the one backend that does.
2403                    caption_parent: CaptionParent::Item,
2404                },
2405            ));
2406            continue;
2407        }
2408        let mut text = region_texts[i].clone();
2409        text.push_str(&merge_suffix[i]);
2410        if text.is_empty() {
2411            continue;
2412        }
2413        match region.label {
2414            // docling assembles checkboxes as TEXT_ELEM items (the region's
2415            // cells are the option label, e.g. right_to_left_03's بلی/خير)
2416            // and its Markdown serializer renders them as task-list lines
2417            // (`- [x] …`) — mirrored by [`Node::CheckboxItem`].
2418            "checkbox_selected" | "checkbox_unselected" => nodes.push(Node::CheckboxItem {
2419                checked: region.label == "checkbox_selected",
2420                text: md_escape(&text),
2421            }),
2422            // docling renders both the document title and section headers as
2423            // `##` (it never emits a top-level `#` for PDFs), so match that.
2424            "title" | "section_header" => nodes.push(located(
2425                loc,
2426                Node::Heading {
2427                    level: 2,
2428                    text: md_escape(&text),
2429                },
2430            )),
2431            // docling drops the rendered bullet glyph; the Markdown serializer
2432            // adds its own `- ` marker. An item whose text opens with an `N.`
2433            // enumeration marker is an ordered item (rendered `N. text`).
2434            // A leading dash stays: it is an ordinary text glyph that
2435            // docling-parse keeps, and docling's items carry it into the
2436            // Markdown (2305's OTSL list renders `- -"C" cell …`) — only the
2437            // symbol-font bullets docling-parse filters out are stripped.
2438            "list_item" => {
2439                let stripped = text
2440                    .trim_start_matches(['•', '◦', '▪', '·', '*'])
2441                    .trim_start()
2442                    .to_string();
2443                if let Some((number, rest)) = parse_ordered_marker(&stripped) {
2444                    nodes.push(Node::ListItem {
2445                        ordered: true,
2446                        number,
2447                        first_in_list: false,
2448                        text: md_escape(&rest),
2449                        level: 0,
2450                        marker: None,
2451                        location: Some(loc),
2452                        dclx: None,
2453                        href: None,
2454                        layer: None,
2455                    });
2456                } else {
2457                    nodes.push(Node::ListItem {
2458                        ordered: false,
2459                        number: 0,
2460                        first_in_list: false,
2461                        text: md_escape(&stripped),
2462                        level: 0,
2463                        // docling keeps the bullet as the DocLang list marker
2464                        // (`<ldiv><marker>·</marker></ldiv>`); Markdown ignores it.
2465                        marker: Some("·".into()),
2466                        location: Some(loc),
2467                        dclx: None,
2468                        href: None,
2469                        layer: None,
2470                    });
2471                }
2472            }
2473            // TableFormer structure (cells + spans, text matched from word cells)
2474            // when available; otherwise geometric grid reconstruction; finally a
2475            // single cell.
2476            "table" | "document_index" => {
2477                // TableFormer grids carry first-class cells (#240: text +
2478                // page-point bbox + span rectangle + OTSL header roles) into
2479                // the public model, and the DocLang structure overlay derives
2480                // from them so DCLX emits real span/header tokens. The
2481                // geometric fallback has no per-cell records.
2482                let (mut rows, cells, structure) = match table_rows[i].clone() {
2483                    Some(grid) => {
2484                        let nrows = grid.rows.len();
2485                        let ncols = grid.rows.first().map_or(0, Vec::len);
2486                        let structure = structure_from_cells(&grid.cells, nrows, ncols);
2487                        (grid.rows, Some(grid.cells), Some(structure))
2488                    }
2489                    None => {
2490                        let rows = reconstruct_table(region, &page.cells);
2491                        let rows = if rows.iter().any(|r| r.len() > 1) {
2492                            rows
2493                        } else {
2494                            vec![vec![text.clone()]]
2495                        };
2496                        (rows, None, None)
2497                    }
2498                };
2499                // The paired caption (#265) rides on the table — docling's
2500                // TableItem.captions ref; Markdown prints it above the grid,
2501                // the JSON export emits the $ref, DocLang the <caption>.
2502                let caption = table_caption_for[i]
2503                    .map(|ci| md_escape(&region_texts[ci]))
2504                    .filter(|t| !t.is_empty());
2505                // Rich cells (docling#3906): the covering cell's blocks are its
2506                // text followed by the nested picture(s). docling's Markdown
2507                // renders a `RichTableCell` through the serializer — the
2508                // group's children joined by blank lines, newlines flattened
2509                // to spaces — so the flat `rows` text becomes
2510                // `text  <!-- image -->`; the first-class `cells` (the JSON
2511                // `table_cells` / `grid`) keep the plain text, as upstream.
2512                let mut cell_blocks: Option<Vec<Vec<Vec<Node>>>> = None;
2513                if let (Some(by_cell), Some(fc)) = (rich_cell_pictures.get(&i), cells.as_ref()) {
2514                    let nrows = rows.len();
2515                    let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
2516                    let mut blocks = vec![vec![Vec::<Node>::new(); ncols]; nrows];
2517                    for (cell_idx, pics) in by_cell {
2518                        let cell = &fc[*cell_idx];
2519                        let (r, c) = (cell.start_row, cell.start_col);
2520                        if r >= nrows || c >= ncols {
2521                            continue;
2522                        }
2523                        let mut parts: Vec<String> = Vec::new();
2524                        let mut cell_nodes: Vec<Node> = Vec::new();
2525                        if !cell.text.trim().is_empty() {
2526                            parts.push(cell.text.clone());
2527                            cell_nodes.push(Node::Paragraph {
2528                                text: cell.text.clone(),
2529                            });
2530                        }
2531                        for &p in pics {
2532                            parts.push("<!-- image -->".to_string());
2533                            let classification = match &enrichments[p] {
2534                                Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
2535                                _ => None,
2536                            };
2537                            #[cfg(feature = "ocr-prep")]
2538                            let image = crop_region(page, &regions[p]);
2539                            #[cfg(not(feature = "ocr-prep"))]
2540                            let image: Option<PictureImage> = None;
2541                            cell_nodes.push(located(
2542                                norm_loc(&regions[p], page.width, page_h),
2543                                Node::Picture {
2544                                    caption: None,
2545                                    caption_href: None,
2546                                    image,
2547                                    classification,
2548                                    caption_parent: Default::default(),
2549                                },
2550                            ));
2551                        }
2552                        let rendered = parts.join("  ");
2553                        for row in rows.iter_mut().skip(r).take(cell.row_span) {
2554                            for slot in row.iter_mut().skip(c).take(cell.col_span) {
2555                                *slot = rendered.clone();
2556                            }
2557                        }
2558                        blocks[r][c] = cell_nodes;
2559                    }
2560                    cell_blocks = Some(blocks);
2561                }
2562                nodes.push(located(
2563                    loc,
2564                    Node::Table(Table {
2565                        rows,
2566                        location: None,
2567                        structure,
2568                        cell_blocks,
2569                        cells,
2570                        caption,
2571                        // As for pictures: the caption is the table's child.
2572                        caption_parent: CaptionParent::Item,
2573                    }),
2574                ));
2575            }
2576            // With formula enrichment the CodeFormula model decodes the region
2577            // to LaTeX; otherwise docling emits a placeholder comment rather
2578            // than the (garbled) raw glyph text.
2579            "formula" => match &enrichments[i] {
2580                Some(Enrichment::Formula { latex }) => nodes.push(Node::Formula {
2581                    latex: latex.clone(),
2582                    orig: text.clone(),
2583                    location: Some(loc),
2584                }),
2585                _ => nodes.push(Node::Paragraph {
2586                    text: "<!-- formula-not-decoded -->".into(),
2587                }),
2588            },
2589            // Code blocks: use the space-glyph-only grouping (monospace keeps its
2590            // source spacing) and emit a fenced block, preserving the line breaks
2591            // and indentation of the source (unlike prose, which reflows). pdfium
2592            // still inserts spaces around tight punctuation (`console .log`,
2593            // `add (3 , 5)`); tighten them to match docling-parse's source spacing.
2594            "code" => {
2595                // `code_region_text` preserves line breaks/indentation and tightens
2596                // each line itself; the fallback prose `text` is tightened here.
2597                let code = code_region_text(region, &page.code_cells);
2598                let code = if code.is_empty() {
2599                    tighten_code_punct(&text)
2600                } else {
2601                    code
2602                };
2603                // With code enrichment the CodeFormula model rewrites the block
2604                // (and names its language); `orig` keeps the raw extraction in
2605                // docling's shape — its parser has no line-preserving code
2606                // path, so its `orig` is the same code with the lines joined
2607                // by single spaces (indentation collapsed).
2608                // docling's parser has no line-preserving code path — its code
2609                // items carry the lines joined by single spaces. That flat
2610                // form is what every byte-conformance surface serializes
2611                // (legacy Markdown, JSON, DocLang); the line-preserving
2612                // extraction rides in `pretty` for strict Markdown only.
2613                let flat = code
2614                    .lines()
2615                    .map(str::trim)
2616                    .filter(|l| !l.is_empty())
2617                    .collect::<Vec<_>>()
2618                    .join(" ");
2619                let node = match &enrichments[i] {
2620                    Some(Enrichment::Code {
2621                        language,
2622                        text: enriched,
2623                    }) => Node::Code {
2624                        language: language.clone(),
2625                        text: enriched.clone(),
2626                        orig: Some(flat),
2627                        pretty: None,
2628                    },
2629                    _ => Node::Code {
2630                        language: None,
2631                        text: flat,
2632                        orig: None,
2633                        pretty: Some(code),
2634                    },
2635                };
2636                nodes.push(located(loc, node));
2637                // docling emits the `Listing N:` caption after the code block.
2638                if let Some(ci) = code_caption_for[i] {
2639                    let cap = md_escape(&region_texts[ci]);
2640                    if !cap.is_empty() {
2641                        nodes.push(Node::Paragraph { text: cap });
2642                    }
2643                }
2644            }
2645            // text, caption, footnote → paragraph
2646            _ => {
2647                // docling parity (`PageAssembleModel._match_hyperlink`): when
2648                // link annotations cover ≥ half of the region's box, the
2649                // hyperlink attaches to the item and the legacy Markdown
2650                // serializer wraps its full text — 2206.01062's footnote URLs
2651                // render as `[1 https://…](https://…)`. Sparse in-paragraph
2652                // citation links stay below the 0.5 coverage threshold and
2653                // remain plain text, exactly like docling.
2654                //
2655                // Scope: **footnote regions only.** Upstream's page_assemble
2656                // matches every TEXT_ELEM label, but published docling
2657                // observably carries the hyperlink into the document only for
2658                // footnote items — in both committed groundtruth generations
2659                // (docling-JSON and Markdown, independent runs) the fully
2660                // covered plain-text DOI line of 2206.01062 page 1 has
2661                // `hyperlink: None` while the equally covered footnotes carry
2662                // theirs. The corpus is the conformance reference, so match
2663                // the observed behavior; widen the label set if a future
2664                // groundtruth refresh starts linking plain text too.
2665                let escaped = md_escape(&text);
2666                let hyperlink = (region.label == "footnote")
2667                    .then(|| region_hyperlink(region, &page.links))
2668                    .flatten();
2669                let text = match hyperlink {
2670                    Some(uri) => {
2671                        // The strict-mode anchor pairs this item covers are
2672                        // superseded by the baked whole-item link.
2673                        links.retain(|(anchor, href)| {
2674                            !(href == &uri && region_texts[i].contains(anchor.as_str()))
2675                        });
2676                        format!("[{escaped}]({uri})")
2677                    }
2678                    None => escaped,
2679                };
2680                nodes.push(located(loc, Node::Paragraph { text }))
2681            }
2682        }
2683    }
2684    // A `/Rotate`-normalized scanned page (see `pdfium_backend`) was assembled
2685    // in upright space; rotate the finished geometry back so locations and the
2686    // page size are display-space, like docling and every viewer report them.
2687    if page.rotation != 0 {
2688        rotate_nodes_to_display(&mut nodes, page.rotation);
2689    }
2690    (nodes, links)
2691}
2692
2693/// Rotate one 0–511 location bbox 90° clockwise on the grid (top-left origin):
2694/// `(x, y) → (511 - y, x)`.
2695fn rot_loc_cw(l: [u16; 4]) -> [u16; 4] {
2696    [511 - l[3], l[0], 511 - l[1], l[2]]
2697}
2698
2699/// Map upright-space geometry back to display space for a page whose `/Rotate`
2700/// was normalized away before inference: every `<location>` rotates `rot`°
2701/// clockwise on the 0–511 grid (the grid is per-axis normalized, so no page
2702/// dims are needed), and the `PageInfo` size returns to the display box. Node
2703/// text and order are untouched — reading order was decided upright, which is
2704/// the whole point.
2705fn rotate_nodes_to_display(nodes: &mut [Node], rot: u16) {
2706    let quarter_turns = (rot / 90) as usize;
2707    let rot_loc = |l: &mut [u16; 4]| {
2708        for _ in 0..quarter_turns {
2709            *l = rot_loc_cw(*l);
2710        }
2711    };
2712    fn walk(node: &mut Node, rot_loc: &impl Fn(&mut [u16; 4]), swap_dims: bool) {
2713        match node {
2714            Node::PageInfo { width, height, .. } => {
2715                if swap_dims {
2716                    std::mem::swap(width, height);
2717                }
2718            }
2719            Node::Located { location, inner } => {
2720                rot_loc(location);
2721                walk(inner, rot_loc, swap_dims);
2722            }
2723            Node::Furniture { inner, .. } => walk(inner, rot_loc, swap_dims),
2724            Node::Group { children, .. } => {
2725                for c in children {
2726                    walk(c, rot_loc, swap_dims);
2727                }
2728            }
2729            Node::ListItem { location, .. }
2730            | Node::Formula { location, .. }
2731            | Node::Chart { location, .. } => {
2732                if let Some(l) = location {
2733                    rot_loc(l);
2734                }
2735            }
2736            Node::PageFurniture { location, .. } => rot_loc(location),
2737            Node::Table(t) => {
2738                if let Some(l) = &mut t.location {
2739                    rot_loc(l);
2740                }
2741            }
2742            _ => {}
2743        }
2744    }
2745    let swap_dims = quarter_turns % 2 == 1;
2746    for node in nodes {
2747        walk(node, &rot_loc, swap_dims);
2748    }
2749}
2750
2751/// Merge paragraph fragments split across a column or page break. docling joins a
2752/// paragraph whose previous fragment ends mid-sentence (a letter, not sentence
2753/// punctuation) with a lowercase continuation: `…definition of` + `lists in…` →
2754/// `…definition of lists in…`. The fragments are consecutive paragraphs, or
2755/// separated only by figure(s) the text wraps around: a column whose body flows
2756/// past a figure resumes below it (`…The wing type that is` ⟶[figure]⟶ `the most
2757/// common…`), and docling emits the whole paragraph before the figure. A heading,
2758/// table, or list between them ends the paragraph (no merge).
2759/// A paragraph that is really a figure/table caption (`Fig. 1. …`, `Table 2 …`).
2760/// Used to skip an unpaired caption when stitching a paragraph that wraps around
2761/// a figure.
2762fn looks_like_caption(text: &str) -> bool {
2763    let head: String = text.trim_start().chars().take(14).collect();
2764    (head.starts_with("Fig") || head.starts_with("Table"))
2765        && head.contains(|c: char| c.is_ascii_digit())
2766}
2767
2768/// A paragraph fragment is "open" — i.e. it might continue into the next
2769/// paragraph — when it ends mid-word (a letter) or with a wrap hyphen/dash.
2770/// docling joins `vocab-` + `ulary` → `vocab- ulary`.
2771fn paragraph_is_open(text: &str) -> bool {
2772    // docling's merge head test (`.+([a-z,\-\u00AD])\s*`): at least two chars,
2773    // ending in an ASCII lowercase letter, a comma, a hyphen, or a soft
2774    // hyphen. The comma matters: 2206's "…In phase four," resumes across the
2775    // page break. Uppercase/non-Latin endings do not merge, exactly as
2776    // upstream (the dash family is already `-` here — clean_text normalized).
2777    let t = text.trim_end();
2778    t.chars().count() >= 2
2779        && t.chars()
2780            .next_back()
2781            .is_some_and(|c| matches!(c, 'a'..='z' | ',' | '-' | '\u{ad}'))
2782}
2783
2784/// The paragraph text inside a node, looking through a [`Node::Located`]
2785/// provenance wrapper (PDF body paragraphs are wrapped since they carry a
2786/// `<location>`). Returns `None` for non-paragraph nodes.
2787fn as_paragraph(n: &Node) -> Option<&str> {
2788    match n {
2789        Node::Paragraph { text } => Some(text),
2790        Node::Located { inner, .. } => match inner.as_ref() {
2791            Node::Paragraph { text } => Some(text),
2792            _ => None,
2793        },
2794        _ => None,
2795    }
2796}
2797
2798/// Whether a node is a picture, looking through a [`Node::Located`] wrapper.
2799fn is_picture_node(n: &Node) -> bool {
2800    match n {
2801        Node::Picture { .. } => true,
2802        Node::Located { inner, .. } => matches!(inner.as_ref(), Node::Picture { .. }),
2803        _ => false,
2804    }
2805}
2806
2807/// A node a forward paragraph merge looks straight past: a figure or *table*
2808/// the text wraps around, or a page header/footer that falls between the two
2809/// fragments of a paragraph continuing across a page break (docling's merge
2810/// skip-labels: page_header, page_footer, table, picture, caption, footnote —
2811/// 2206's "…In phase four," resumes after a full caption+table+figure block).
2812fn is_merge_trailer(n: &Node) -> bool {
2813    is_picture_node(n)
2814        || matches!(
2815            n,
2816            Node::PageFurniture { .. } | Node::PageInfo { .. } | Node::Table(_)
2817        )
2818        || matches!(n, Node::Located { inner, .. } if matches!(inner.as_ref(), Node::Table(_)))
2819        || as_paragraph(n).is_some_and(looks_like_caption)
2820}
2821
2822/// Rebuild node `i` as a paragraph with `text`, preserving its `<location>`
2823/// wrapper (and thus provenance) if it had one.
2824fn reparagraph(node: &Node, text: String) -> Node {
2825    match node {
2826        Node::Located { location, .. } => located(*location, Node::Paragraph { text }),
2827        _ => Node::Paragraph { text },
2828    }
2829}
2830
2831pub(crate) fn merge_continuations(nodes: &mut Vec<Node>) {
2832    let mut i = 0;
2833    while i + 1 < nodes.len() {
2834        let Some(a) = as_paragraph(&nodes[i]) else {
2835            i += 1;
2836            continue;
2837        };
2838        // A figure/table caption is a self-contained unit; body text resuming
2839        // after a figure is the continuation case, not the caption itself. Never
2840        // stitch *from* a caption — otherwise a caption that ends in a lone glyph
2841        // (`Fig. 5. … PubTabNet. μ`) would swallow a following stray figure label
2842        // (a standalone `μ`) into `… μ μ`.
2843        if looks_like_caption(a) {
2844            i += 1;
2845            continue;
2846        }
2847        if !paragraph_is_open(a) {
2848            i += 1;
2849            continue;
2850        }
2851        // The continuation is the next paragraph, looking past any figures the
2852        // text wraps around — and a figure/table caption that was emitted as its
2853        // own paragraph (an above-the-figure caption that didn't pair), since the
2854        // body text resumes after the whole figure+caption block.
2855        let mut j = i + 1;
2856        while nodes.get(j).is_some_and(is_merge_trailer) {
2857            j += 1;
2858        }
2859        // docling's continuation regex allows either case, but its merge runs
2860        // over the pre-assembly element stream; at node level an uppercase
2861        // start is overwhelmingly a new sentence/heading fragment (allowing it
2862        // swallowed 2305's formula blocks and redp's chapter openers), so the
2863        // continuation stays lowercase-start here.
2864        let cont = nodes.get(j).and_then(as_paragraph).is_some_and(|b| {
2865            b.trim_start()
2866                .chars()
2867                .next()
2868                .is_some_and(char::is_lowercase)
2869        });
2870        if cont {
2871            let a = as_paragraph(&nodes[i]).unwrap().trim_end().to_string();
2872            let b = as_paragraph(&nodes[j]).unwrap().trim_start().to_string();
2873            // A soft hyphen -- or a hard hyphen followed by a lowercase
2874            // continuation (guaranteed lowercase by the `cont` gate above) --
2875            // is a word split across the break: strip it and join without a
2876            // space, docling#3888 ("vocab-" + "ulary" -> "vocabulary");
2877            // docling's older serializer kept the artifact ("vocab- ulary").
2878            // Everything else joins with the space, as before.
2879            let merged = match a.strip_suffix('\u{ad}').or_else(|| a.strip_suffix('-')) {
2880                Some(stem) => format!("{stem}{b}"),
2881                None => format!("{a} {b}"),
2882            };
2883            // Keep node i's provenance wrapper; docling's merged paragraph keeps
2884            // the first fragment's geometry as its primary location.
2885            nodes[i] = reparagraph(&nodes[i], merged);
2886            nodes.remove(j);
2887            // Re-check i: the merged paragraph may continue further.
2888        } else {
2889            i += 1;
2890        }
2891    }
2892}
2893
2894/// How many leading nodes of `nodes` are safe to flush now — i.e. cannot be
2895/// rewritten by a future [`merge_continuations`] once more pages are appended.
2896///
2897/// A forward merge can only start from an "open" paragraph (ends mid-word) and
2898/// only reaches across trailing pictures and figure/table captions. So we scan
2899/// from the end past those skippable trailers: if the first non-skippable node is
2900/// an open paragraph, it (and the trailers after it) must be held; anything else —
2901/// a closed paragraph, a heading, a table, a list — blocks any forward merge, so
2902/// the whole buffer is safe to flush.
2903fn hold_start(nodes: &[Node]) -> usize {
2904    for k in (0..nodes.len()).rev() {
2905        // Skippable trailers (figures, page furniture, captions): a forward merge
2906        // looks straight past them.
2907        if is_merge_trailer(&nodes[k]) {
2908            continue;
2909        }
2910        match as_paragraph(&nodes[k]) {
2911            // An open body paragraph might still pull a continuation off the next
2912            // page — hold from here to the end.
2913            Some(text) if paragraph_is_open(text) => return k,
2914            // A closed paragraph, heading, table, list, etc. ends the paragraph:
2915            // nothing after it can merge backwards across it. Flush everything.
2916            _ => return nodes.len(),
2917        }
2918    }
2919    // Only skippable trailers (or empty) and no open paragraph to anchor a merge.
2920    nodes.len()
2921}
2922
2923/// Streaming counterpart of [`merge_continuations`]: feed per-page node batches in
2924/// document order and get back the prefix that is final (its cross-page merges are
2925/// resolved and no future page can change it), holding back only the small tail
2926/// that might still merge into the next page. Concatenating every flushed batch
2927/// (then [`finish`](Self::finish)) yields exactly the same nodes as running
2928/// [`merge_continuations`] once over the whole document.
2929pub(crate) struct StreamAssembler {
2930    pending: Vec<Node>,
2931}
2932
2933impl StreamAssembler {
2934    pub(crate) fn new() -> Self {
2935        Self {
2936            pending: Vec::new(),
2937        }
2938    }
2939
2940    /// Append one page's nodes, resolve merges within the buffer, and return the
2941    /// now-final prefix to emit (possibly empty).
2942    pub(crate) fn push(&mut self, mut nodes: Vec<Node>) -> Vec<Node> {
2943        self.pending.append(&mut nodes);
2944        merge_continuations(&mut self.pending);
2945        let cut = hold_start(&self.pending);
2946        let tail = self.pending.split_off(cut);
2947        std::mem::replace(&mut self.pending, tail)
2948    }
2949
2950    /// Flush whatever is left after the last page (the held tail is final once no
2951    /// more pages can follow).
2952    pub(crate) fn finish(self) -> Vec<Node> {
2953        self.pending
2954    }
2955}
2956
2957#[cfg(test)]
2958mod tests {
2959    use super::{cells_text, clean_text};
2960    use super::{code_region_text, merge_continuations, resolve_link_anchors, StreamAssembler};
2961    use crate::layout::Region;
2962    use crate::pdfium_backend::{LinkAnnot, PdfPage, TextCell};
2963    use docling_core::Node;
2964
2965    /// The int8-layout guard's coverage metric: cells under detections count,
2966    /// cells outside don't, whitespace cells are ignored, and a cell-less page
2967    /// reads as fully covered (nothing to rescue).
2968    #[test]
2969    fn layout_cell_coverage_counts_claimed_text_cells() {
2970        let cell = |text: &str, l: f32, t: f32| TextCell {
2971            text: text.into(),
2972            l,
2973            t,
2974            r: l + 40.0,
2975            b: t + 10.0,
2976        };
2977        let region = Region {
2978            label: "text",
2979            score: 0.9,
2980            l: 0.0,
2981            t: 0.0,
2982            r: 100.0,
2983            b: 50.0,
2984        };
2985        let cells = vec![
2986            cell("inside", 10.0, 10.0),
2987            cell("also inside", 10.0, 30.0),
2988            cell("outside", 10.0, 200.0),
2989            cell("   ", 10.0, 210.0), // whitespace: not counted at all
2990        ];
2991        let cov = super::layout_cell_coverage(std::slice::from_ref(&region), &cells);
2992        assert!((cov - 2.0 / 3.0).abs() < 1e-6, "got {cov}");
2993        assert_eq!(super::layout_cell_coverage(&[], &[]), 1.0);
2994        assert_eq!(super::layout_cell_coverage(&[], &cells), 0.0);
2995    }
2996
2997    /// #165: a picture no longer claims cells at 0.2 intersection-over-self.
2998    /// A line straddling the figure border (≤80 % contained) becomes an orphan
2999    /// region and survives the contained-regulars drop — before the fix its
3000    /// cells were silently erased. A line fully inside the picture is still
3001    /// re-dropped, matching docling's Markdown (a picture's children never
3002    /// reach its serializer's output).
3003    #[test]
3004    fn border_straddling_lines_survive_picture_interior_is_still_dropped() {
3005        let pic = Region {
3006            label: "picture",
3007            score: 0.9,
3008            l: 0.0,
3009            t: 0.0,
3010            r: 100.0,
3011            b: 100.0,
3012        };
3013        // ~35 % of this cell overlaps the picture (l=90..120 of 0..100): above
3014        // the old 0.2 claim (was swallowed), below full containment (survives).
3015        let straddler = TextCell {
3016            text: "axis label".into(),
3017            l: 90.0,
3018            t: 40.0,
3019            r: 120.0,
3020            b: 48.0,
3021        };
3022        let interior = TextCell {
3023            text: "in-figure callout".into(),
3024            l: 10.0,
3025            t: 10.0,
3026            r: 60.0,
3027            b: 18.0,
3028        };
3029        let mut regions = vec![pic];
3030        super::add_orphan_regions(&mut regions, &[straddler, interior]);
3031        assert_eq!(
3032            regions.iter().filter(|r| r.label == "text").count(),
3033            2,
3034            "both unclaimed lines become orphans"
3035        );
3036        super::drop_contained_regulars(&mut regions);
3037        let texts: Vec<(f32, f32)> = regions
3038            .iter()
3039            .filter(|r| r.label == "text")
3040            .map(|r| (r.l, r.r))
3041            .collect();
3042        assert_eq!(
3043            texts,
3044            [(90.0, 120.0)],
3045            "the straddler is emitted, the fully-contained callout is not"
3046        );
3047    }
3048
3049    /// docling#3906's concern, pinned on our side: a picture detected fully
3050    /// inside a table region must survive the containment drop (upstream now
3051    /// attaches it to the table's cell; we keep it as a body sibling — either
3052    /// way it must not vanish). The text region inside the same table is the
3053    /// control: regulars are the ones the drop swallows.
3054    #[test]
3055    fn picture_inside_a_table_region_survives_the_containment_drop() {
3056        let mut regions = vec![
3057            region("table", 0.9, 0.0, 0.0, 200.0, 200.0),
3058            region("picture", 0.9, 20.0, 20.0, 120.0, 120.0),
3059            region("text", 0.9, 20.0, 140.0, 180.0, 180.0),
3060        ];
3061        super::drop_contained_regulars(&mut regions);
3062        let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
3063        assert_eq!(
3064            labels,
3065            ["table", "picture"],
3066            "the in-table picture stays; the in-table regular is the special's child"
3067        );
3068    }
3069
3070    /// Table–caption pairing (#265) is reading-order adjacency, docling's
3071    /// `_find_to_captions`: a caption binds the table directly next to it in
3072    /// the region sequence — above-caption and below-caption both work, and
3073    /// geometry is irrelevant (a same-page caption in the other column of a
3074    /// two-column layout is *not* adjacent, however close its box is). A
3075    /// caption with media on both sides, or separated from the table by a
3076    /// text paragraph, stays unattached.
3077    #[test]
3078    fn table_captions_pair_by_reading_order_adjacency() {
3079        // caption → table (above-caption), then table → caption (below-caption),
3080        // then a caption fenced off by a paragraph, then one between two tables.
3081        let regions = vec![
3082            region("text", 0.9, 0.0, 0.0, 100.0, 10.0), // 0 body text
3083            region("caption", 0.9, 0.0, 12.0, 60.0, 20.0), // 1 above-caption
3084            region("table", 0.9, 20.0, 22.0, 90.0, 60.0), // 2 ← pairs with 1
3085            region("table", 0.9, 0.0, 70.0, 100.0, 110.0), // 3 ← pairs with 4
3086            region("caption", 0.9, 0.0, 112.0, 60.0, 120.0), // 4 below-caption
3087            region("text", 0.9, 0.0, 130.0, 100.0, 140.0), // 5 body text
3088            region("caption", 0.9, 0.0, 142.0, 60.0, 150.0), // 6 fenced by 5/7
3089            region("text", 0.9, 0.0, 152.0, 100.0, 162.0), // 7 body text
3090            region("table", 0.9, 0.0, 170.0, 100.0, 200.0), // 8 unpaired
3091            region("caption", 0.9, 0.0, 202.0, 60.0, 210.0), // 9 ambiguous
3092            region("table", 0.9, 0.0, 212.0, 100.0, 240.0), // 10 unpaired
3093        ];
3094        let mut taken = vec![false; regions.len()];
3095        let pairs = super::pair_table_captions(&regions, &mut taken);
3096        assert_eq!(pairs[2], Some(1), "caption directly above its table pairs");
3097        assert_eq!(pairs[3], Some(4), "caption directly below its table pairs");
3098        assert_eq!(
3099            pairs[8], None,
3100            "a text paragraph between caption and table breaks the bond"
3101        );
3102        assert_eq!(
3103            pairs[10], None,
3104            "a caption between two tables is ambiguous and stays loose"
3105        );
3106        assert!(taken[1] && taken[4] && !taken[6] && !taken[9]);
3107    }
3108
3109    /// A colored terms-and-conditions panel detected as `picture` demotes into
3110    /// per-paragraph `text` regions (the blank line between C.7 and C.8 splits
3111    /// them); a chart whose only text is a few narrow axis labels keeps its
3112    /// crop untouched.
3113    #[test]
3114    fn text_panels_demote_to_paragraphs_but_charts_keep_their_crop() {
3115        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3116            text: text.to_string(),
3117            l,
3118            t,
3119            r,
3120            b,
3121        };
3122        let panel = Region {
3123            label: "picture",
3124            score: 0.9,
3125            l: 0.0,
3126            t: 0.0,
3127            r: 100.0,
3128            b: 100.0,
3129        };
3130        // Three tight lines, a blank-line gap, two more: two paragraphs.
3131        let cells = vec![
3132            cell(
3133                "C.7. Wenn Sie diesen Vertrag widerrufen,",
3134                5.0,
3135                10.0,
3136                95.0,
3137                18.0,
3138            ),
3139            cell(
3140                "haben wir Ihnen alle Zahlungen, die wir",
3141                5.0,
3142                20.0,
3143                95.0,
3144                28.0,
3145            ),
3146            cell(
3147                "von Ihnen erhalten haben, zurückzuzahlen.",
3148                5.0,
3149                30.0,
3150                90.0,
3151                38.0,
3152            ),
3153            cell(
3154                "C.8. Wir können die Rückzahlung verweigern,",
3155                5.0,
3156                52.0,
3157                95.0,
3158                60.0,
3159            ),
3160            cell(
3161                "bis wir die Waren wieder zurückerhalten haben.",
3162                5.0,
3163                62.0,
3164                92.0,
3165                70.0,
3166            ),
3167        ];
3168        let mut regions = vec![panel.clone()];
3169        super::recover_text_panels(&mut regions, &cells);
3170        assert_eq!(
3171            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3172            ["text", "text"],
3173            "dense panel must demote into one text region per paragraph"
3174        );
3175        assert!(regions[0].b < regions[1].t, "paragraphs split at the gap");
3176        // Sparse narrow labels (a chart): picture survives.
3177        let labels = vec![
3178            cell("0", 5.0, 90.0, 8.0, 95.0),
3179            cell("50", 5.0, 50.0, 10.0, 55.0),
3180            cell("100", 5.0, 10.0, 12.0, 15.0),
3181            cell("t, s", 45.0, 96.0, 55.0, 100.0),
3182        ];
3183        let mut regions = vec![panel];
3184        super::recover_text_panels(&mut regions, &labels);
3185        assert_eq!(
3186            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3187            ["picture"]
3188        );
3189    }
3190
3191    /// An uncaptioned chart on a scanned page whose title, axis labels, and
3192    /// OCR boxes over the plot area are dense and wide enough to pass the
3193    /// coverage/width gates still keeps its crop: its line heights are ragged
3194    /// (title face vs tick labels vs bar-area OCR), failing the uniform-leading
3195    /// gate — a real text panel is set with constant leading (#173).
3196    #[test]
3197    fn dense_titled_chart_keeps_its_crop() {
3198        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3199            text: text.to_string(),
3200            l,
3201            t,
3202            r,
3203            b,
3204        };
3205        let chart = Region {
3206            label: "picture",
3207            score: 0.9,
3208            l: 0.0,
3209            t: 0.0,
3210            r: 100.0,
3211            b: 100.0,
3212        };
3213        // Five wide lines at wildly different heights: a 12-pt title, 20-pt OCR
3214        // boxes over the bars, 4–5-pt tick/axis labels. Coverage and median
3215        // width both clear the panel thresholds.
3216        let cells = vec![
3217            cell("Underground Water Storage", 10.0, 5.0, 90.0, 17.0),
3218            cell("aquifer recharge zone", 15.0, 30.0, 75.0, 50.0),
3219            cell("confined | unconfined | perched", 12.0, 55.0, 80.0, 59.0),
3220            cell("saturated thickness", 8.0, 70.0, 60.0, 90.0),
3221            cell("distance from well, km", 20.0, 92.0, 85.0, 97.0),
3222        ];
3223        let mut regions = vec![chart];
3224        super::recover_text_panels(&mut regions, &cells);
3225        assert_eq!(
3226            regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3227            ["picture"],
3228            "ragged line heights mark a figure, not a text panel"
3229        );
3230    }
3231
3232    /// docling serializes a cluster's cells in docling-parse index order
3233    /// (`_sort_cells`) and joins them with `PageAssembleModel.sanitize_text`:
3234    /// a space after every line except one ending in `-`, which either fuses a
3235    /// wrapped word (alnum on both sides — dash dropped) or glues verbatim (a
3236    /// bare `-` cell: `[0000` `-` `0002` → `[0000 -0002`, the 2305 ORCID line;
3237    /// `-` + `"C" cell -` + `a new table cell` → `-"C" cell a new table cell`,
3238    /// its OTSL list). Verified against the corpus: pure index order beats any
3239    /// geometric re-sort (normal_4pages' heading numerals paint after their
3240    /// text and belong last: `## 들어가며 1`).
3241    #[test]
3242    fn cells_join_in_index_order_with_sanitize_text_rules() {
3243        let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3244            text: text.to_string(),
3245            l,
3246            t,
3247            r,
3248            b,
3249        };
3250        let region = Region {
3251            label: "text",
3252            score: 1.0,
3253            l: 0.0,
3254            t: 95.0,
3255            r: 200.0,
3256            b: 130.0,
3257        };
3258        // ORCID superscript: a bare dash cell is a *detached* dash — kept, and
3259        // since docling#4052 (2.122) it joins with the ordinary space on both
3260        // sides (`[0000 -0002 -6960]` before that fix).
3261        let orcid = vec![
3262            cell("[0000", 10.0, 100.0, 30.0, 110.0),
3263            cell("−", 30.0, 100.0, 34.0, 110.0),
3264            cell("0002", 34.0, 100.0, 50.0, 110.0),
3265            cell("−", 50.0, 100.0, 54.0, 110.0),
3266            cell("6960]", 54.0, 100.0, 70.0, 110.0),
3267        ];
3268        assert_eq!(super::region_text(&region, &orcid), "[0000 - 0002 - 6960]");
3269        // Wrapped word: dash dropped, lines fused (both boundary words alnum).
3270        let wrapped = vec![
3271            cell("platforms-", 10.0, 100.0, 60.0, 110.0),
3272            cell("reflects the design", 10.0, 112.0, 90.0, 122.0),
3273        ];
3274        assert_eq!(
3275            super::region_text(&region, &wrapped),
3276            "platformsreflects the design"
3277        );
3278        // Dash-ending lines that are *detached* dashes (a bare bullet cell, a
3279        // `cell -` separator): the dash stays and the lines join with a space
3280        // — docling#4052; before it they glued (`-"C" cell a new table cell`,
3281        // 2305's OTSL list bullets).
3282        let otsl = vec![
3283            cell("–", 10.0, 100.0, 14.0, 110.0),
3284            cell("\"C\" cell -", 16.0, 100.0, 60.0, 110.0),
3285            cell("a new table cell", 10.0, 112.0, 80.0, 122.0),
3286        ];
3287        assert_eq!(
3288            super::region_text(&region, &otsl),
3289            "- \"C\" cell - a new table cell"
3290        );
3291        // Index order is authoritative — no geometric re-sort.
3292        let numeral = vec![
3293            cell("들어가며", 30.0, 100.0, 80.0, 110.0),
3294            cell("1", 10.0, 98.0, 25.0, 112.0), // big numeral painted last
3295        ];
3296        assert_eq!(super::region_text(&region, &numeral), "들어가며 1");
3297    }
3298
3299    /// The geometric-reliability gate, on the two shapes it has to tell apart.
3300    #[test]
3301    fn geometric_reliability_rejects_split_column_grids() {
3302        let g = |rows: &[&[&str]]| -> Vec<Vec<String>> {
3303            rows.iter()
3304                .map(|r| r.iter().map(|c| c.to_string()).collect())
3305                .collect()
3306        };
3307        // A genuine grid: dense, every column carrying entries. Nothing for
3308        // TableFormer to improve, so geometry is used as-is.
3309        assert!(super::geometric_table_is_reliable(&g(&[
3310            &["Datum", "Leistung", "Anzahl", "Kosten"],
3311            &["04.07", "Internet", "1", "40.30"],
3312            &["04.07", "Telefon", "2", "8.06"],
3313        ])));
3314        // The left-edge split artefact (the shape a scanned invoice produced):
3315        // one real label column plus values scattered across three sparse ones.
3316        assert!(!super::geometric_table_is_reliable(&g(&[
3317            &["www.magenta.at/faq", "", "", ""],
3318            &["Serviceteam", "", "", ""],
3319            &["Telefon", "0676/2000", "", ""],
3320            &["Kundennummer", "", "", "1.21699482"],
3321            &["Rechnungsnummer", "", "922769430725", ""],
3322            &["Rechnungsdatum", "", "", "04.07.2025"],
3323        ])));
3324        // A column only one row ever uses is a split artefact even when the
3325        // grid is otherwise dense.
3326        assert!(!super::geometric_table_is_reliable(&g(&[
3327            &["a", "b", ""],
3328            &["c", "d", ""],
3329            &["e", "f", "g"],
3330        ])));
3331        // Degenerate shapes are never vouched for — TableFormer may recover
3332        // structure a collapsed reconstruction lost.
3333        assert!(!super::geometric_table_is_reliable(&g(&[&[
3334            "only one column"
3335        ]])));
3336        assert!(!super::geometric_table_is_reliable(&[]));
3337    }
3338
3339    /// A `picture` region is cropped out of the rendered page, whatever built
3340    /// that page. The browser pipeline (#157) has no pdfium but does hand over
3341    /// the rasterized bitmap through `from_cells_with_image`, so it must get
3342    /// the same figure bytes the native path does — that is what makes
3343    /// `images = "embedded"` inline real pixels instead of a placeholder.
3344    #[cfg(feature = "ocr-prep")]
3345    #[test]
3346    fn picture_regions_are_cropped_from_a_host_supplied_page_image() {
3347        let mut img = image::RgbImage::new(200, 200);
3348        // Paint the figure area so the crop is distinguishable from the page.
3349        for y in 100..160 {
3350            for x in 20..120 {
3351                img.put_pixel(x, y, image::Rgb([255, 0, 0]));
3352            }
3353        }
3354        // scale 2.0: the region is in page points, the bitmap in pixels.
3355        let page = PdfPage::from_cells_with_image(100.0, 100.0, 2.0, Vec::new(), img);
3356        let region = Region {
3357            label: "picture",
3358            score: 0.9,
3359            l: 10.0,
3360            t: 50.0,
3361            r: 60.0,
3362            b: 80.0,
3363        };
3364        let (nodes, _) = super::assemble_page(&page, vec![region], &[None], &[None]);
3365        // Layout-derived nodes carry provenance, so the picture arrives wrapped.
3366        let image = nodes
3367            .iter()
3368            .find_map(|n| match n {
3369                Node::Located { inner, .. } => match &**inner {
3370                    Node::Picture { image, .. } => image.as_ref(),
3371                    _ => None,
3372                },
3373                Node::Picture { image, .. } => image.as_ref(),
3374                _ => None,
3375            })
3376            .expect("a picture node with cropped pixels");
3377        assert_eq!(image.mimetype, "image/png");
3378        assert_eq!((image.width, image.height), (100, 60), "region × scale");
3379        assert!(!image.data.is_empty(), "PNG bytes were encoded");
3380    }
3381
3382    #[test]
3383    fn link_anchors_split_a_shared_word_cell_between_adjacent_links() {
3384        // A common header layout: one text run holds several pipe-separated
3385        // labels, each carrying its own link annotation. Every link must get
3386        // its own label as the anchor (and the "|" separators must belong to
3387        // none), not the whole run.
3388        let annot = |l: f32, r: f32, uri: &str| LinkAnnot {
3389            l,
3390            t: 100.0,
3391            r,
3392            b: 114.0,
3393            uri: uri.into(),
3394        };
3395        let page = PdfPage {
3396            width: 600.0,
3397            height: 800.0,
3398            scale: 2.0,
3399            cells: Vec::new(),
3400            code_cells: Vec::new(),
3401            // "LinkedIn | GitHub | Credly" = 26 chars over x 100..360.
3402            word_cells: vec![cell(
3403                "LinkedIn | GitHub | Credly",
3404                100.0,
3405                100.0,
3406                360.0,
3407                114.0,
3408            )],
3409            image: image::RgbImage::new(1, 1),
3410            image_layout: None,
3411            links: vec![
3412                annot(100.0, 180.0, "https://l"),
3413                annot(200.0, 260.0, "https://g"),
3414                annot(290.0, 360.0, "https://c"),
3415            ],
3416            rotation: 0,
3417        };
3418        assert_eq!(
3419            resolve_link_anchors(&page),
3420            vec![
3421                ("LinkedIn".to_string(), "https://l".to_string()),
3422                ("GitHub".to_string(), "https://g".to_string()),
3423                ("Credly".to_string(), "https://c".to_string()),
3424            ]
3425        );
3426    }
3427
3428    /// A one-line code cell at `[l, r] × [t, b]` (top-left coords).
3429    fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
3430        TextCell {
3431            text: text.into(),
3432            l,
3433            t,
3434            r,
3435            b,
3436        }
3437    }
3438
3439    fn region(label: &'static str, score: f32, l: f32, t: f32, r: f32, b: f32) -> Region {
3440        Region {
3441            label,
3442            score,
3443            l,
3444            t,
3445            r,
3446            b,
3447        }
3448    }
3449
3450    #[test]
3451    fn resolve_collapses_nested_code_keeping_the_larger_box() {
3452        // A tight high-score `code` box and a taller lower-score near-duplicate that
3453        // contains it must collapse to one — the *larger* box, so every cell stays
3454        // covered and nothing leaks out as orphan text.
3455        let tight = region("code", 0.95, 78.0, 292.0, 300.0, 330.0);
3456        let wide = region("code", 0.66, 63.0, 260.0, 320.0, 346.0);
3457        let kept = super::resolve(vec![tight, wide]);
3458        assert_eq!(kept.len(), 1, "nested code boxes must collapse to one");
3459        assert!(
3460            kept[0].l == 63.0 && kept[0].b == 346.0,
3461            "the larger containing box is kept"
3462        );
3463    }
3464
3465    #[test]
3466    fn resolve_keeps_distinct_and_differently_typed_regions() {
3467        // A text box fully inside a lower-score *table* must NOT be collapsed (the
3468        // code dedup is code-only), and two separate code blocks stay separate.
3469        let text = region("text", 0.95, 90.0, 210.0, 200.0, 230.0);
3470        let table = region("table", 0.60, 80.0, 200.0, 400.0, 500.0);
3471        assert_eq!(super::resolve(vec![text, table]).len(), 2);
3472
3473        let code_a = region("code", 0.9, 78.0, 100.0, 300.0, 140.0);
3474        let code_b = region("code", 0.9, 78.0, 300.0, 300.0, 360.0); // far below, no overlap
3475        assert_eq!(super::resolve(vec![code_a, code_b]).len(), 2);
3476    }
3477
3478    #[test]
3479    fn code_language_label_above_code_is_detected() {
3480        // A bare "XML" token directly above a code box is a language label; a real
3481        // heading above the same code is not; a language word with no code below is
3482        // left alone.
3483        let label = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
3484        let code = region("code", 0.7, 77.0, 552.0, 290.0, 640.0);
3485        let heading = region("section_header", 0.9, 76.0, 500.0, 260.0, 512.0);
3486        let cells = vec![
3487            cell("XML", 78.0, 541.0, 94.0, 548.0),       // inside `label`
3488            cell("Overview", 78.0, 501.0, 250.0, 511.0), // inside `heading`
3489        ];
3490        let drop = super::code_language_labels(&[label, code, heading], &cells);
3491        assert_eq!(drop, vec![true, false, false], "only the label is consumed");
3492
3493        // Same label with no code region present → not consumed.
3494        let label2 = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
3495        let only = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
3496        assert_eq!(super::code_language_labels(&[label2], &only), vec![false]);
3497
3498        // A label swallowed into the top of a wider code box (negative gap) is still
3499        // recognized.
3500        let inside_lbl = region("text", 0.9, 76.0, 540.0, 96.0, 549.0);
3501        let wide_code = region("code", 0.7, 63.0, 531.0, 320.0, 654.0);
3502        let cells2 = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
3503        assert_eq!(
3504            super::code_language_labels(&[inside_lbl, wide_code], &cells2),
3505            vec![true, false]
3506        );
3507
3508        assert!(super::is_code_language("XML") && super::is_code_language("c#"));
3509        assert!(!super::is_code_language("Configure") && !super::is_code_language("XML schema"));
3510    }
3511
3512    #[test]
3513    fn code_region_text_keeps_lines_and_indentation() {
3514        // Three source lines; each glyph is 6 units wide (width / chars = 6), so the
3515        // `int X;` line indented to x=22 is (22-10)/6 = 2 spaces in.
3516        let region = Region {
3517            label: "code",
3518            score: 1.0,
3519            l: 0.0,
3520            t: -5.0,
3521            r: 100.0,
3522            b: 40.0,
3523        };
3524        let cells = vec![
3525            cell("struct P {", 10.0, 0.0, 70.0, 10.0),
3526            cell("int X;", 22.0, 12.0, 58.0, 22.0),
3527            cell("}", 10.0, 24.0, 16.0, 34.0),
3528        ];
3529        assert_eq!(code_region_text(&region, &cells), "struct P {\n  int X;\n}");
3530    }
3531
3532    #[test]
3533    fn code_region_text_tightens_punctuation_without_eating_indentation() {
3534        // A fluent `.Foo()` line at x=22 (2 chars in). Per-line tightening must not
3535        // consume the leading indent space by matching " ." across it.
3536        let region = Region {
3537            label: "code",
3538            score: 1.0,
3539            l: 0.0,
3540            t: -5.0,
3541            r: 100.0,
3542            b: 40.0,
3543        };
3544        let cells = vec![
3545            cell("builder", 10.0, 0.0, 52.0, 10.0),
3546            // pdfium spaced the call: ".Foo (x)" tightens to ".Foo(x)", still 2-indented.
3547            cell(".Foo (x)", 22.0, 12.0, 70.0, 22.0),
3548        ];
3549        assert_eq!(code_region_text(&region, &cells), "builder\n  .Foo(x)");
3550    }
3551
3552    #[test]
3553    fn code_region_text_orders_out_of_order_cells_and_ignores_blank_lines() {
3554        let region = Region {
3555            label: "code",
3556            score: 1.0,
3557            l: 0.0,
3558            t: -5.0,
3559            r: 100.0,
3560            b: 60.0,
3561        };
3562        // Fed bottom-up and with a whitespace-only cell; output is top-down, no blank.
3563        let cells = vec![
3564            cell("b();", 10.0, 24.0, 34.0, 34.0),
3565            cell("   ", 10.0, 12.0, 20.0, 22.0),
3566            cell("a();", 10.0, 0.0, 34.0, 10.0),
3567        ];
3568        assert_eq!(code_region_text(&region, &cells), "a();\nb();");
3569        // No code cells → empty, so the caller falls back to the prose text.
3570        assert_eq!(code_region_text(&region, &[]), "");
3571    }
3572
3573    fn para(text: &str) -> Node {
3574        Node::Paragraph { text: text.into() }
3575    }
3576
3577    /// Run a node sequence through [`StreamAssembler`] with the given page splits
3578    /// and assert the flushed result equals one-shot [`merge_continuations`].
3579    fn assert_stream_eq(nodes: &[Node], splits: &[usize]) {
3580        let mut want = nodes.to_vec();
3581        merge_continuations(&mut want);
3582
3583        let mut asm = StreamAssembler::new();
3584        let mut got = Vec::new();
3585        let mut start = 0;
3586        for &end in splits {
3587            got.extend(asm.push(nodes[start..end].to_vec()));
3588            start = end;
3589        }
3590        got.extend(asm.push(nodes[start..].to_vec()));
3591        got.extend(asm.finish());
3592        assert_eq!(got, want, "stream assembly diverged (splits={splits:?})");
3593    }
3594
3595    #[test]
3596    fn stream_assembler_matches_merge_continuations() {
3597        // Open fragment + lowercase continuation split across a page boundary.
3598        let cross = [para("the definition of"), para("lists in scope")];
3599        assert_stream_eq(&cross, &[1]);
3600        assert_stream_eq(&cross, &[]);
3601
3602        // Continuation that wraps around a figure (+ its caption) on the boundary.
3603        let wrap = [
3604            para("the wing type that is"),
3605            Node::Picture {
3606                caption: None,
3607                caption_href: None,
3608                image: None,
3609                classification: None,
3610                caption_parent: Default::default(),
3611            },
3612            para("Fig. 1. a diagram"),
3613            para("the most common kind"),
3614        ];
3615        for splits in [&[][..], &[1][..], &[2][..], &[3][..], &[1, 3][..]] {
3616            assert_stream_eq(&wrap, splits);
3617        }
3618
3619        // A heading between fragments blocks the merge (must still flush correctly).
3620        let blocked = [
3621            para("ends mid word and"),
3622            Node::Heading {
3623                level: 2,
3624                text: "New Section".into(),
3625            },
3626            para("more body here"),
3627        ];
3628        for splits in [&[][..], &[1][..], &[2][..]] {
3629            assert_stream_eq(&blocked, splits);
3630        }
3631
3632        // A chain across three pages: each page is one open lowercase fragment.
3633        let chain = [
3634            para("alpha beta"),
3635            para("gamma delta"),
3636            para("epsilon zeta"),
3637        ];
3638        assert_stream_eq(&chain, &[1, 2]);
3639    }
3640
3641    #[test]
3642    fn clean_text_dehyphenates_and_normalizes_typography() {
3643        // U+0002 line-wrap hyphen + the join space → merged word (like docling).
3644        assert_eq!(clean_text("com\u{2} pact"), "compact");
3645        assert_eq!(clean_text("end-to\u{2} end deep"), "end-toend deep");
3646        // A stray wrap hyphen (no following join) is dropped.
3647        assert_eq!(clean_text("word\u{2}"), "word");
3648        // Typographic punctuation → ASCII: every curly quote becomes `'`
3649        // (docling-parse's sanitizer table), a literal `"` stays.
3650        assert_eq!(
3651            clean_text("Graph\u{2019}s \u{201c}x\u{201d} \"y\""),
3652            "Graph's 'x' \"y\""
3653        );
3654        assert_eq!(clean_text("a\u{2026}"), "a...");
3655        // The dp default (the docling-parse sanitizer) preserves internal spacing
3656        // it placed deliberately; line breaks/tabs normalize to a space, ends trim.
3657        assert_eq!(clean_text("a   b\nc"), "a   b c");
3658    }
3659
3660    /// docling#4064: a form's children are emitted together where the form
3661    /// sits in the top-level order, not interleaved with surrounding text.
3662    #[test]
3663    fn form_children_stay_together_in_reading_order() {
3664        let reg = |label: &'static str, l: f32, t: f32, r: f32, b: f32| Region {
3665            label,
3666            score: 0.9,
3667            l,
3668            t,
3669            r,
3670            b,
3671        };
3672        // Page: intro text, then a form spanning the left column with two
3673        // fields and a table inside, while a right-column paragraph sits
3674        // level with the form's first field (it would otherwise be read
3675        // between the form's children).
3676        let mut items = vec![
3677            reg("text", 50.0, 50.0, 550.0, 70.0),    // 0 intro
3678            reg("form", 50.0, 100.0, 300.0, 400.0),  // 1 container
3679            reg("text", 60.0, 110.0, 290.0, 130.0),  // 2 field A (child)
3680            reg("text", 320.0, 110.0, 550.0, 130.0), // 3 right column paragraph
3681            reg("table", 60.0, 150.0, 290.0, 300.0), // 4 table (child)
3682            reg("text", 60.0, 320.0, 290.0, 340.0),  // 5 field B (child)
3683            reg("text", 50.0, 450.0, 550.0, 470.0),  // 6 outro
3684        ];
3685        super::order_with_containers(&mut items, 600.0, 800.0, |r| r);
3686        let order: Vec<(&str, f32)> = items.iter().map(|r| (r.label, r.t)).collect();
3687        // The form block (container, then its children top-down) is one unit.
3688        let form_pos = order.iter().position(|(l, _)| *l == "form").unwrap();
3689        assert_eq!(
3690            &order[form_pos..form_pos + 4],
3691            &[
3692                ("form", 100.0),
3693                ("text", 110.0),
3694                ("table", 150.0),
3695                ("text", 320.0)
3696            ]
3697        );
3698        assert_eq!(order[0], ("text", 50.0));
3699        assert_eq!(order[order.len() - 1], ("text", 450.0));
3700        // Without a container the plain order interleaves by geometry.
3701        let mut flat: Vec<Region> = items
3702            .iter()
3703            .filter(|r| r.label != "form")
3704            .cloned()
3705            .collect();
3706        super::order_regions(&mut flat, 600.0, 800.0, |r| r);
3707        assert_ne!(
3708            flat.iter().map(|r| r.t).collect::<Vec<_>>(),
3709            order
3710                .iter()
3711                .filter(|(l, _)| *l != "form")
3712                .map(|(_, t)| *t)
3713                .collect::<Vec<_>>()
3714        );
3715    }
3716
3717    /// docling#3906: a picture inside a table lands in the covering cell,
3718    /// chosen by the picture's inferred grid position when cell boxes overlap.
3719    #[test]
3720    fn picture_matches_the_cell_at_its_grid_position() {
3721        let cell = |r: usize, c: usize, bbox: [f32; 4]| docling_core::TableCell {
3722            text: format!("r{r}c{c}"),
3723            bbox: Some(bbox),
3724            start_row: r,
3725            start_col: c,
3726            row_span: 1,
3727            col_span: 1,
3728            column_header: false,
3729            row_header: false,
3730            row_section: false,
3731        };
3732        // 2×2 grid; the (1,0) cell box is generous and also covers the picture.
3733        let cells = vec![
3734            cell(0, 0, [0.0, 0.0, 100.0, 50.0]),
3735            cell(0, 1, [100.0, 0.0, 200.0, 50.0]),
3736            cell(1, 0, [0.0, 50.0, 100.0, 100.0]),
3737            cell(1, 1, [100.0, 50.0, 200.0, 100.0]),
3738        ];
3739        let pic = Region {
3740            label: "picture",
3741            score: 0.9,
3742            l: 110.0,
3743            t: 60.0,
3744            r: 190.0,
3745            b: 95.0,
3746        };
3747        assert_eq!(super::match_picture_to_cell(&pic, &cells), Some((1.0, 3)));
3748        // A picture only half inside any cell is not nested.
3749        let straddling = Region {
3750            label: "picture",
3751            score: 0.9,
3752            l: 60.0,
3753            t: 60.0,
3754            r: 160.0,
3755            b: 95.0,
3756        };
3757        assert_eq!(super::match_picture_to_cell(&straddling, &cells), None);
3758    }
3759
3760    /// docling#4052 (2.122): a line-final dash fuses the wrapped word only
3761    /// when attached to it; a detached dash is a literal and the lines join
3762    /// with a space.
3763    #[test]
3764    fn line_final_hyphen_fuses_only_when_attached_to_a_word() {
3765        let line = |text: &str, t: f32| TextCell {
3766            text: text.to_string(),
3767            l: 0.0,
3768            t,
3769            r: 100.0,
3770            b: t + 10.0,
3771        };
3772        // `algo-` / `rithms`: attached hyphen, alnum on both sides → fused.
3773        assert_eq!(
3774            cells_text(vec![&line("algo-", 0.0), &line("rithms", 12.0)]),
3775            "algorithms"
3776        );
3777        // `pp. 545-` / `561`: attached, digits count as alnum → `545561` (upstream).
3778        assert_eq!(
3779            cells_text(vec![&line("pp. 545-", 0.0), &line("561", 12.0)]),
3780            "pp. 545561"
3781        );
3782        // A dash after whitespace — a separator or a lone `-` cell — is kept and
3783        // the lines take the ordinary joining space.
3784        assert_eq!(
3785            cells_text(vec![&line("range -", 0.0), &line("wide", 12.0)]),
3786            "range - wide"
3787        );
3788        assert_eq!(
3789            cells_text(vec![&line("-", 0.0), &line("item", 12.0)]),
3790            "- item"
3791        );
3792        // Attached but the next line opens with no word (`x-` / `...`): dash
3793        // kept and, as before, no separating space.
3794        assert_eq!(
3795            cells_text(vec![&line("x-", 0.0), &line("...", 12.0)]),
3796            "x-..."
3797        );
3798    }
3799
3800    #[test]
3801    fn lam_alef_only_swaps_a_genuinely_reversed_ligature() {
3802        // A mid-word `alef-variant + lam` is pdfium's reversed lam-alef ligature and
3803        // is swapped back to logical `lam + alef-variant` (`ب أ ل` → `ب ل أ`).
3804        assert_eq!(
3805            clean_text("\u{0628}\u{0623}\u{0644}"),
3806            "\u{0628}\u{0644}\u{0623}"
3807        );
3808        // But when the alef-variant is *already* preceded by a lam it is the logical
3809        // ligature `لآ`; the following lam is the next syllable's letter and must not
3810        // move. `التعلم الآلي` must stay `الآلي`, not become `اللآي`.
3811        assert_eq!(
3812            clean_text("\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"),
3813            "\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"
3814        );
3815    }
3816
3817    /// The #419 page, in points: three layout boxes over one paragraph, two of
3818    /// them ending partway through a line. The sliced lines miss the 0.2 claim
3819    /// and become orphans; the third model box starts above the second orphan,
3820    /// so unfitted the reading order emits that box first and strands the line.
3821    fn sliced_paragraph() -> (Vec<Region>, Vec<TextCell>) {
3822        let line = |text: &str, t: f32, r: f32| cell(text, 60.0, t, r, t + 11.0);
3823        let cells = vec![
3824            line("The mission of this series is to improve", 135.0, 458.0),
3825            line("The books in this series are technical,", 147.0, 458.0),
3826            line("substantial. The authors are", 159.0, 458.0),
3827            line("highly experienced craftsmen and", 171.5, 458.0), // sliced: 1.5/11 under box A
3828            line("actually works in practice, as opposed", 185.0, 458.0),
3829            line("about what the author has done, not", 197.0, 458.0),
3830            line("about programming, there will be lots", 210.5, 458.0), // sliced: 1.5/11 under box B
3831            line("will be lots of case studies from real", 223.0, 206.0), // C's line
3832        ];
3833        let regions = vec![
3834            region("text", 0.9, 60.0, 132.0, 458.0, 173.0), // A: three lines + a sliver of the 4th
3835            region("text", 0.9, 60.0, 184.0, 458.0, 212.0), // B: two lines + a sliver of the 7th
3836            region("text", 0.9, 60.0, 216.0, 206.0, 227.0), // C: last line, box opening 5.5pt too early
3837        ];
3838        (regions, cells)
3839    }
3840
3841    fn ordered_texts(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
3842        let mut items: Vec<Region> = regions.to_vec();
3843        super::order_regions(&mut items, 500.0, 700.0, |r| r);
3844        super::region_texts_exclusive(&items, cells)
3845            .into_iter()
3846            .map(|t| t.chars().take(9).collect())
3847            .collect()
3848    }
3849
3850    /// #419: fitted to its cells, a model box that cut a line in half no longer
3851    /// overlaps the orphan that line became, so the orphan orders where it
3852    /// reads; unfitted, the same page strands the line after the paragraph.
3853    #[test]
3854    fn fitting_boxes_to_cells_puts_a_sliced_line_back_in_order() {
3855        let (mut regions, cells) = sliced_paragraph();
3856        super::add_orphan_regions(&mut regions, &cells);
3857        assert_eq!(regions.len(), 5, "two orphan lines");
3858        // The defect, for the record: C (top 216) is not strictly below the
3859        // orphan at 210.5–221.5, so the graph orders C first.
3860        assert_eq!(
3861            ordered_texts(&regions, &cells).last().map(String::as_str),
3862            Some("about pro")
3863        );
3864
3865        super::fit_regions_to_cells(&mut regions, &cells);
3866        assert_eq!(regions.len(), 5);
3867        // A ends on its last claimed line, C starts on its only one.
3868        assert_eq!((regions[0].t, regions[0].b), (135.0, 170.0));
3869        assert_eq!((regions[2].t, regions[2].b), (223.0, 234.0));
3870        assert_eq!(
3871            ordered_texts(&regions, &cells),
3872            [
3873                "The missi",
3874                "highly ex",
3875                "actually ",
3876                "about pro",
3877                "will be l"
3878            ]
3879        );
3880    }
3881
3882    /// An orphan the fitted paragraph box surrounds (a short middle line the
3883    /// narrow model box missed while claiming the lines around it) is folded
3884    /// into the paragraph; an empty regular box goes away, a formula stays, a
3885    /// picture is never refitted, and a page with no cells is left untouched.
3886    #[test]
3887    fn fitting_folds_surrounded_orphans_and_drops_empty_regulars() {
3888        let wide = |text: &str, t: f32| cell(text, 60.0, t, 400.0, t + 11.0);
3889        let cells = vec![
3890            wide("first line of the paragraph", 100.0),
3891            cell("stray", 250.0, 112.0, 400.0, 123.0), // clear of the narrow box
3892            wide("third line of the paragraph", 124.0),
3893        ];
3894        let mut regions = vec![
3895            // Narrow box: claims the wide lines at 0.41, misses the short one.
3896            region("text", 0.9, 60.0, 98.0, 200.0, 136.0),
3897            region("section_header", 0.8, 60.0, 300.0, 200.0, 320.0), // no cells
3898            region("formula", 0.8, 60.0, 340.0, 200.0, 360.0),        // no cells, kept
3899            region("picture", 0.8, 0.0, 400.0, 500.0, 600.0),
3900        ];
3901        super::add_orphan_regions(&mut regions, &cells);
3902        assert_eq!(regions.len(), 5, "the short line became an orphan");
3903        super::fit_regions_to_cells(&mut regions, &cells);
3904        let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
3905        assert_eq!(labels, ["text", "formula", "picture"]);
3906        let para = &regions[0];
3907        assert_eq!(
3908            (para.l, para.t, para.r, para.b),
3909            (60.0, 100.0, 400.0, 135.0)
3910        );
3911        assert_eq!(
3912            super::region_texts_exclusive(&regions, &cells)[0],
3913            "first line of the paragraph stray third line of the paragraph"
3914        );
3915        assert_eq!(
3916            (regions[2].t, regions[2].b),
3917            (400.0, 600.0),
3918            "picture untouched"
3919        );
3920
3921        let mut untouched = vec![region("text", 0.9, 0.0, 0.0, 10.0, 10.0)];
3922        super::fit_regions_to_cells(&mut untouched, &[]);
3923        assert_eq!(untouched.len(), 1, "no cells yet: nothing dropped");
3924    }
3925}