docling_pdf/assemble.rs
1//! Layout-driven assembly: map detected [`Region`]s + text cells to a
2//! [`DoclingDocument`], mirroring docling's page-assembly + reading-order.
3//!
4//! Overlapping detections are resolved greedily by score, each text cell is
5//! assigned to its best-containing region, regions are ordered in reading order
6//! (two-column aware), and each becomes a typed node by its layout label.
7
8use docling_core::{CaptionParent, Node, PictureClass, PictureImage, Table};
9#[cfg(feature = "ml")]
10use image::RgbImage;
11
12use crate::layout::Region;
13use crate::pdfium_backend::{PdfPage, TextCell};
14
15fn area(l: f32, t: f32, r: f32, b: f32) -> f32 {
16 ((r - l).max(0.0)) * ((b - t).max(0.0))
17}
18
19/// Intersection area of two boxes.
20fn inter(a: &Region, l: f32, t: f32, r: f32, b: f32) -> f32 {
21 let il = a.l.max(l);
22 let it = a.t.max(t);
23 let ir = a.r.min(r);
24 let ib = a.b.min(b);
25 area(il, it, ir, ib)
26}
27
28/// Wrapper (structured-region) labels, ported from docling
29/// `LayoutPostprocessor.WRAPPER_TYPES`: a region that *contains* other regions
30/// and renders as a structured block (a table / table-of-contents index), not as
31/// its own flat text.
32fn is_wrapper(label: &str) -> bool {
33 matches!(
34 label,
35 "table" | "document_index" | "form" | "key_value_region"
36 )
37}
38
39/// Labels docling's table-structure (TableFormer) model runs on and that render
40/// as a Markdown table: a plain `table` and a `document_index` (a table of
41/// contents), which docling assembles as a `TableItem` too.
42pub fn is_table_like(label: &str) -> bool {
43 matches!(label, "table" | "document_index")
44}
45
46/// Greedily keep regions by descending score, dropping a region that is mostly
47/// covered by an already-kept one (RT-DETR emits overlapping duplicates).
48fn greedy(mut regions: Vec<Region>) -> Vec<Region> {
49 regions.sort_by(|a, b| b.score.total_cmp(&a.score));
50 let mut kept: Vec<Region> = Vec::new();
51 for r in regions {
52 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
53 let covered = kept.iter().any(|k| {
54 let i = inter(&r, k.l, k.t, k.r, k.b);
55 let ka = area(k.l, k.t, k.r, k.b).max(1.0);
56 // drop if most of r is inside k, or they strongly mutually overlap
57 i / ra > 0.7 || i / (ra + ka - i) > 0.5
58 });
59 if !covered {
60 kept.push(r);
61 }
62 }
63 kept
64}
65
66/// Resolve overlapping RT-DETR detections, ported from the bucket structure of
67/// docling's `LayoutPostprocessor`: regular, picture and wrapper clusters live in
68/// **separate** spatial indexes and are de-overlapped independently, so a
69/// high-score picture never suppresses a lower-score table or table-of-contents
70/// index (the redp5110 TOC that was otherwise replaced by a picture box). A
71/// cross-type pass first drops a picture that nearly coincides with a table
72/// (`_handle_cross_type_overlaps`), keeping the structured table.
73/// docling's `_remove_overlapping_clusters("picture")`: same-label picture
74/// detections whose boxes heavily overlap (IoU > 0.8, or either box > 80 %
75/// contained in the other) form one group, and a single survivor is kept per
76/// group. Survivor selection ports `_should_prefer_cluster` /
77/// `_select_best_cluster_from_group` with the picture params
78/// (`area_threshold` 2.0, `conf_threshold` 0.3): a candidate is rejected only
79/// when a rival is both comparable in size (candidate ≤ 2× its area) and
80/// clearly more confident (> 0.3); among the survivors the *larger* box wins
81/// unless it is > 0.3 less confident. Net effect on the corpus: a figure the
82/// detector proposes both whole and as its sub-panels (2206's four-thumbnail
83/// Figure 1) collapses to the whole-figure box, exactly like docling.
84pub(crate) fn dedup_pictures(regions: &mut Vec<Region>) {
85 let idx: Vec<usize> = (0..regions.len())
86 .filter(|&i| regions[i].label == "picture")
87 .collect();
88 if idx.len() < 2 {
89 return;
90 }
91 // Union-find over the picture subset.
92 let mut parent: Vec<usize> = (0..idx.len()).collect();
93 fn find(parent: &mut [usize], i: usize) -> usize {
94 let mut root = i;
95 while parent[root] != root {
96 root = parent[root];
97 }
98 let mut cur = i;
99 while parent[cur] != root {
100 let next = parent[cur];
101 parent[cur] = root;
102 cur = next;
103 }
104 root
105 }
106 let boxed = |r: &Region| (r.l, r.t, r.r, r.b);
107 for a in 0..idx.len() {
108 for b in (a + 1)..idx.len() {
109 let (ra, rb) = (®ions[idx[a]], ®ions[idx[b]]);
110 let (al, at, ar, ab_) = boxed(ra);
111 let (bl, bt, br, bb) = boxed(rb);
112 let ix = (ar.min(br) - al.max(bl)).max(0.0);
113 let iy = (ab_.min(bb) - at.max(bt)).max(0.0);
114 let inter = ix * iy;
115 let aa = area(al, at, ar, ab_).max(f32::EPSILON);
116 let ba = area(bl, bt, br, bb).max(f32::EPSILON);
117 let iou = inter / (aa + ba - inter).max(f32::EPSILON);
118 if iou > 0.8 || inter / aa > 0.8 || inter / ba > 0.8 {
119 let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
120 if pa != pb {
121 parent[pa] = pb;
122 }
123 }
124 }
125 }
126 // Per group, run docling's pairwise preference + larger-wins selection.
127 let mut groups: std::collections::HashMap<usize, Vec<usize>> = std::collections::HashMap::new();
128 for i in 0..idx.len() {
129 let root = find(&mut parent, i);
130 groups.entry(root).or_default().push(i);
131 }
132 let mut drop = vec![false; regions.len()];
133 for group in groups.values() {
134 if group.len() < 2 {
135 continue;
136 }
137 const AREA_THRESHOLD: f32 = 2.0;
138 const CONF_THRESHOLD: f32 = 0.3;
139 let area_of = |i: usize| {
140 let r = ®ions[idx[i]];
141 area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
142 };
143 let mut best: Option<usize> = None;
144 for &cand in group {
145 let passes = group.iter().all(|&other| {
146 if other == cand {
147 return true;
148 }
149 let area_ratio = area_of(cand) / area_of(other);
150 let conf_diff = regions[idx[other]].score - regions[idx[cand]].score;
151 !(area_ratio <= AREA_THRESHOLD && conf_diff > CONF_THRESHOLD)
152 });
153 if passes {
154 best = Some(match best {
155 None => cand,
156 Some(cur) => {
157 if area_of(cand) > area_of(cur)
158 && regions[idx[cur]].score - regions[idx[cand]].score <= CONF_THRESHOLD
159 {
160 cand
161 } else {
162 cur
163 }
164 }
165 });
166 }
167 }
168 // Every candidate rejected can't happen with docling's rule (rejection
169 // needs a strictly better rival); guard with highest score anyway.
170 let keep = best.unwrap_or_else(|| {
171 *group
172 .iter()
173 .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(®ions[idx[b]].score))
174 .expect("non-empty group")
175 });
176 for &i in group {
177 if i != keep {
178 drop[idx[i]] = true;
179 }
180 }
181 }
182 let mut keep_iter = drop.into_iter();
183 regions.retain(|_| !keep_iter.next().expect("aligned"));
184}
185
186/// `intersection_over_union` of two regions.
187fn iou(a: &Region, b: &Region) -> f32 {
188 let i = inter(a, b.l, b.t, b.r, b.b);
189 let u = area(a.l, a.t, a.r, a.b) + area(b.l, b.t, b.r, b.b) - i;
190 if u > 0.0 {
191 i / u
192 } else {
193 0.0
194 }
195}
196
197/// docling's `_resolve_coincident_pairs` (#4059, 2.122): for every (loser,
198/// winner) pair at a near-identical box (IoU > 0.8) whose confidences are
199/// within 0.1 (`loser.score - winner.score < 0.1`), the loser label is dropped
200/// so the label with the richer downstream semantic survives. Nothing else —
201/// containment, area — is considered; a clearly more confident loser stays.
202fn coincident_losers(regions: &[Region], losers: &[usize], winners: &[usize]) -> Vec<usize> {
203 let mut out = Vec::new();
204 for &li in losers {
205 for &wi in winners {
206 if iou(®ions[li], ®ions[wi]) > 0.8 && regions[li].score - regions[wi].score < 0.1
207 {
208 out.push(li);
209 break;
210 }
211 }
212 }
213 out
214}
215
216/// docling's `_handle_cross_type_overlaps` (2.122/2.123 shape): the layout
217/// model can emit one grounded region under several labels, and the picture /
218/// table / container buckets are de-overlapped independently, so such a region
219/// survives twice. Elect a winner for the near-identical pairs:
220///
221/// | pair | loser | winner |
222/// |---------------------------------------|-----------|---------------------|
223/// | TABLE vs DOCUMENT_INDEX | table | document_index |
224/// | PICTURE vs TABLE / DOCUMENT_INDEX | picture | the table-like |
225/// | FORM / KEY_VALUE_REGION vs *surviving* TABLE / DOCUMENT_INDEX / PICTURE | container | structured element |
226///
227/// IoU (not containment) so a genuine small figure inside a large table region
228/// is not removed; the confidence tolerance keeps a clearly more confident
229/// loser (an earlier port dropped every coincident picture regardless).
230fn handle_cross_type_overlaps(regions: Vec<Region>) -> Vec<Region> {
231 let by = |pred: &dyn Fn(&str) -> bool| -> Vec<usize> {
232 (0..regions.len())
233 .filter(|&i| pred(regions[i].label))
234 .collect()
235 };
236 let tables = by(&|l| l == "table");
237 let doc_indices = by(&|l| l == "document_index");
238 let pictures = by(&|l| l == "picture");
239 let containers = by(&|l| matches!(l, "form" | "key_value_region"));
240 let mut drop = vec![false; regions.len()];
241 for i in coincident_losers(®ions, &tables, &doc_indices) {
242 drop[i] = true;
243 }
244 let table_like: Vec<usize> = tables.iter().chain(&doc_indices).copied().collect();
245 for i in coincident_losers(®ions, &pictures, &table_like) {
246 drop[i] = true;
247 }
248 let structured: Vec<usize> = table_like
249 .iter()
250 .chain(&pictures)
251 .copied()
252 .filter(|&i| !drop[i])
253 .collect();
254 for i in coincident_losers(®ions, &containers, &structured) {
255 drop[i] = true;
256 }
257 let mut drop = drop.into_iter();
258 let mut regions = regions;
259 regions.retain(|_| !drop.next().expect("aligned"));
260 regions
261}
262
263pub fn resolve(regions: Vec<Region>) -> Vec<Region> {
264 let regions = handle_cross_type_overlaps(regions);
265 // De-overlap each bucket on its own.
266 let pictures = greedy(
267 regions
268 .iter()
269 .filter(|r| r.label == "picture")
270 .cloned()
271 .collect(),
272 );
273 // Tables and containers are separate buckets since docling 2.123
274 // (`TABLE_TYPES` vs `CONTAINER_TYPES`, docling#4064): a form drawn around a
275 // table no longer competes with it for survival — the table nests inside
276 // the container instead (`order_with_containers`).
277 let tables = greedy(
278 regions
279 .iter()
280 .filter(|r| is_table_like(r.label))
281 .cloned()
282 .collect(),
283 );
284 let containers = greedy(
285 regions
286 .iter()
287 .filter(|r| matches!(r.label, "form" | "key_value_region"))
288 .cloned()
289 .collect(),
290 );
291 let mut kept = greedy(
292 regions
293 .iter()
294 .filter(|r| r.label != "picture" && !is_wrapper(r.label))
295 .cloned()
296 .collect(),
297 );
298 dedup_nested_code(&mut kept);
299 kept.extend(pictures);
300 kept.extend(tables);
301 kept.extend(containers);
302 kept
303}
304
305/// Drop a regular region that is >80% contained in a surviving special region we
306/// render **as a single unit** — a table/table-of-contents index — ported from
307/// docling's "Remove regular clusters that are included in wrappers" step: the
308/// special absorbs it as a child (a table cell), so it must not also be emitted
309/// as its own paragraph/list-item. This stops the survey list-items from
310/// appearing both inside the detected table and again as bullets
311/// (`table_mislabeled_as_picture`).
312///
313/// `picture` regions stay in the swallow set even after #165: docling keeps a
314/// picture's contained clusters as the `PictureItem`'s *children* in the
315/// document JSON (`ReadingOrderModel._add_child_elements`), but its
316/// `MarkdownPictureSerializer` prints only the caption and the image — the
317/// children never reach the Markdown (verified against the corpus groundtruth:
318/// `amt_handbook`'s in-figure callout labels are absent). Dropping the
319/// fully-contained regulars here reproduces exactly that. What #165 *does*
320/// change is upstream, in [`add_orphan_regions`]: pictures no longer claim
321/// cells, so a line only partially under a figure box (straddling its border,
322/// ≤80 % contained) now forms an orphan region that survives this drop — those
323/// words were silently erased before, and docling emits them.
324///
325/// `form` / `key_value_region` wrappers are deliberately **excluded**: this
326/// pipeline does not render them as a structured block (they are skipped), so
327/// their textual content comes precisely from the contained regular regions —
328/// dropping those would erase the page (e.g. `right_to_left_03`'s form-heavy
329/// pages). Runs *after* [`drop_false_pictures`] so a phantom picture can't
330/// swallow real text on its way out.
331pub fn drop_contained_regulars(regions: &mut Vec<Region>) {
332 let specials: Vec<(f32, f32, f32, f32)> = regions
333 .iter()
334 .filter(|r| r.label == "picture" || is_table_like(r.label))
335 .map(|r| (r.l, r.t, r.r, r.b))
336 .collect();
337 if specials.is_empty() {
338 return;
339 }
340 regions.retain(|r| {
341 if r.label == "picture" || is_wrapper(r.label) {
342 return true;
343 }
344 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
345 !specials
346 .iter()
347 .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.8)
348 });
349}
350
351/// True for a bare, single-token source-code language label (`XML`, `C#`, `JSON`,
352/// `bash`, …) — the little header the docs render above a code block. Matched
353/// case-insensitively; anything with whitespace or longer than a token is out.
354fn is_code_language(t: &str) -> bool {
355 let t = t.trim();
356 if t.is_empty() || t.chars().any(char::is_whitespace) || t.chars().count() > 12 {
357 return false;
358 }
359 const LANGS: &[&str] = &[
360 "xml",
361 "html",
362 "xhtml",
363 "json",
364 "jsonc",
365 "yaml",
366 "yml",
367 "toml",
368 "ini",
369 "c#",
370 "csharp",
371 "f#",
372 "fsharp",
373 "vb",
374 "c",
375 "c++",
376 "cpp",
377 "java",
378 "kotlin",
379 "scala",
380 "go",
381 "golang",
382 "rust",
383 "swift",
384 "javascript",
385 "js",
386 "typescript",
387 "ts",
388 "jsx",
389 "tsx",
390 "python",
391 "py",
392 "ruby",
393 "rb",
394 "php",
395 "perl",
396 "lua",
397 "r",
398 "dart",
399 "bash",
400 "sh",
401 "shell",
402 "powershell",
403 "zsh",
404 "batch",
405 "cmd",
406 "sql",
407 "tsql",
408 "plsql",
409 "graphql",
410 "dockerfile",
411 "makefile",
412 "css",
413 "scss",
414 "sass",
415 "less",
416 "markdown",
417 "md",
418 "tex",
419 "latex",
420 "diff",
421 "proto",
422 "razor",
423 "cshtml",
424 "xaml",
425 "aspx",
426 "http",
427 ];
428 let lower = t.to_ascii_lowercase();
429 LANGS.contains(&lower.as_str())
430}
431
432/// Mark the region indices that are a code block's **language label** — a bare
433/// `XML`/`C#`/… token sitting directly above a `code` region — so they are consumed
434/// rather than emitted as their own stray paragraph/heading. The label may also be
435/// captured inside a wider code box (rendered as the fence's first line); dropping
436/// the standalone copy just removes the duplicate.
437fn code_language_labels(regions: &[Region], cells: &[TextCell]) -> Vec<bool> {
438 let mut drop = vec![false; regions.len()];
439 for (i, r) in regions.iter().enumerate() {
440 if matches!(r.label, "code" | "picture" | "table") {
441 continue;
442 }
443 if !is_code_language(®ion_text(r, cells)) {
444 continue;
445 }
446 // The label sits just above the code (a blank line's gap) or is swallowed
447 // into the top of a wider code box; either way it is that block's label.
448 // The window is generous because the label's own font is small, so a
449 // one-line gap is several times its height.
450 let line_h = (r.b - r.t).abs().max(1.0);
451 let window = (line_h * 4.0).max(28.0);
452 let labels_code = regions.iter().enumerate().any(|(j, c)| {
453 if j == i || c.label != "code" {
454 return false;
455 }
456 let gap = c.t - r.b; // >0 when the code is below the label
457 let h_overlap = (r.r.min(c.r) - r.l.max(c.l)).max(0.0);
458 gap > -line_h * 3.0 && gap < window && h_overlap > 0.0
459 });
460 if labels_code {
461 drop[i] = true;
462 }
463 }
464 drop
465}
466
467/// Collapse `code` regions where one is nested inside another, keeping the larger.
468///
469/// RT-DETR sometimes emits a tight code box *and* a wider near-duplicate that also
470/// captures the block's language label (`XML`, `C#`, …). When the tight box scores
471/// higher it is kept first, and the wider container — not "mostly inside" the tight
472/// box — survives [`resolve`]'s greedy pass, so the block is emitted twice. Keeping
473/// the **larger** box (rather than dropping it) collapses the pair without leaking
474/// the container's extra cells back out as orphan text, since the larger box still
475/// covers every cell. Restricted to `code` so genuinely distinct nested regions of
476/// other kinds are untouched.
477fn dedup_nested_code(kept: &mut Vec<Region>) {
478 let mut drop = vec![false; kept.len()];
479 for i in 0..kept.len() {
480 if kept[i].label != "code" {
481 continue;
482 }
483 let ai = area(kept[i].l, kept[i].t, kept[i].r, kept[i].b).max(1.0);
484 for j in 0..kept.len() {
485 if i == j || drop[j] || kept[j].label != "code" {
486 continue;
487 }
488 let aj = area(kept[j].l, kept[j].t, kept[j].r, kept[j].b).max(1.0);
489 // Drop i when it is mostly inside a strictly larger code box j.
490 let overlap = inter(&kept[i], kept[j].l, kept[j].t, kept[j].r, kept[j].b);
491 if aj > ai && overlap / ai > 0.7 {
492 drop[i] = true;
493 break;
494 }
495 }
496 }
497 let mut keep = drop.iter();
498 kept.retain(|_| !*keep.next().unwrap());
499}
500
501/// Fraction of the page's non-empty text cells that some detected region
502/// claims (>0.2 intersection-over-self, docling's assignment rule). 1.0 for a
503/// page without text cells.
504///
505/// The int8-layout guard keys off this: a dense digital page whose detections
506/// cover almost none of its text is the signature of quantized confidences
507/// flipping under the 0.5 label thresholds on this CPU's kernels — not of a
508/// genuinely empty layout — and is worth re-running on the fp32 graph.
509pub fn layout_cell_coverage(regions: &[Region], cells: &[TextCell]) -> f32 {
510 let mut total = 0usize;
511 let mut covered = 0usize;
512 for c in cells {
513 if c.text.trim().is_empty() {
514 continue;
515 }
516 total += 1;
517 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
518 if regions
519 .iter()
520 .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
521 {
522 covered += 1;
523 }
524 }
525 if total == 0 {
526 1.0
527 } else {
528 covered as f32 / total as f32
529 }
530}
531
532/// Append `text` regions for cells the layout left uncovered ("orphan cells"),
533/// the way docling's `LayoutPostprocessor` does (`create_orphan_clusters`): any
534/// non-empty cell that no kept region covers (>50% of the cell's area) becomes a
535/// text region of its own, so text the detector missed (a stray `.`, a small
536/// label) is still emitted instead of silently dropped. Adjacent orphan cells on a
537/// line are merged so a missed paragraph doesn't shatter into one block per line.
538pub fn add_orphan_regions(regions: &mut Vec<Region>, cells: &[TextCell]) {
539 // docling assigns each cell to its single best-overlapping cluster at
540 // intersection-over-self > 0.2 and serializes exactly the assigned cells —
541 // and since [`region_texts_exclusive`] now emits under that very rule, the
542 // claim test here matches it: any cell over 0.2 will actually render in
543 // its best region, everything else becomes an orphan. Completeness by
544 // construction, with no (0.2, 0.5] hole (the old > 0.5 serializer needed
545 // the claim test raised to > 0.5 to keep right_to_left_03's `20300` from
546 // vanishing; the exclusive port closes that structurally).
547 //
548 // Only *regular* clusters claim cells: docling's `_find_unassigned_cells`
549 // walks `regular_clusters` alone, so a cell under a `picture` or a wrapper
550 // (`table`/`document_index`/`form`/`key_value_region`) that no regular
551 // cluster covers still becomes an orphan text cluster (#165). The orphans
552 // that end up *fully* inside the special are re-dropped by
553 // [`drop_contained_regulars`] (docling's Markdown drops them the same way
554 // — a picture's children never reach its `MarkdownPictureSerializer`
555 // output, a table's text renders through the reconstructed grid). The
556 // observable fix is the border-straddlers: a line only partially under a
557 // figure box used to lose its cells to the picture's 0.2 claim and vanish
558 // — now it forms an orphan region and is emitted, as docling does.
559 let assigned = |c: &TextCell| {
560 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
561 regions
562 .iter()
563 .filter(|r| r.label != "picture" && !is_wrapper(r.label))
564 .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
565 };
566 // Collect orphan cells (non-empty, unassigned), in page order.
567 let mut orphans: Vec<&TextCell> = cells
568 .iter()
569 .filter(|c| !c.text.trim().is_empty() && !assigned(c))
570 .collect();
571 if orphans.is_empty() {
572 return;
573 }
574 orphans.sort_by(|a, b| a.t.total_cmp(&b.t).then(a.l.total_cmp(&b.l)));
575 // Merge cells that sit on the same line and nearly touch into one region, so a
576 // dropped multi-word line stays one block (docling's refinement merges these).
577 let mut merged: Vec<Region> = Vec::new();
578 for c in orphans {
579 let h = (c.b - c.t).abs().max(1.0);
580 if let Some(last) = merged.last_mut() {
581 let same_line = (last.t - c.t).abs() < h * 0.5;
582 let touching = c.l <= last.r + h && c.l >= last.l - h;
583 if same_line && touching {
584 last.l = last.l.min(c.l);
585 last.r = last.r.max(c.r);
586 last.t = last.t.min(c.t);
587 last.b = last.b.max(c.b);
588 continue;
589 }
590 }
591 merged.push(Region {
592 label: "text",
593 score: 0.0,
594 l: c.l,
595 t: c.t,
596 r: c.r,
597 b: c.b,
598 });
599 }
600 regions.extend(merged);
601}
602
603/// Demote a `picture` region that is really a **text panel** — a paragraph block
604/// the layout model boxed as a figure because it is typeset on a colored
605/// background (terms-and-conditions callouts, quote boxes) — into ordinary
606/// `text` regions, one per paragraph, so its words are read instead of shipped
607/// as pixels. docling loses this text the same way (cells assigned to a picture
608/// cluster are never serialized); this is a deliberate improvement, not parity.
609///
610/// The gate is conservative so a genuine figure keeps its crop: the region must
611/// contain at least three text lines whose median width spans most of the panel
612/// (axis labels and chat bubbles are narrow and varied) and whose cells cover a
613/// substantial fraction of its area (a photo or chart with sparse labels does
614/// not). Paragraph boundaries are re-derived from the line pitch: a vertical gap
615/// clearly larger than the panel's own leading starts a new `text` region, so
616/// the panel doesn't collapse into one giant paragraph.
617///
618/// Works on any cell source — the digital text layer or OCR lines recognized
619/// from the picture crop — so the native and browser paths, with or without
620/// force-OCR, demote identically.
621pub fn recover_text_panels(regions: &mut Vec<Region>, cells: &[TextCell]) {
622 // A *captioned* picture is a genuine figure whatever it contains — the
623 // corpus is full of document screenshots ("Figure 3: …" above a page
624 // image) that are exactly as dense and wide as a text panel. Only an
625 // uncaptioned picture is a demotion candidate.
626 let captioned: Vec<bool> = regions
627 .iter()
628 .map(|r| {
629 r.label == "picture"
630 && regions.iter().any(|c| {
631 c.label == "caption" && c.r.min(r.r) - c.l.max(r.l) > 0.0 && {
632 let gap = if c.t >= r.b {
633 c.t - r.b
634 } else if r.t >= c.b {
635 r.t - c.b
636 } else {
637 f32::MAX // vertically overlapping: not a caption
638 };
639 gap <= 25.0
640 }
641 })
642 })
643 .collect();
644 let mut out: Vec<Region> = Vec::with_capacity(regions.len());
645 // Synthesized paragraphs and the demoted panels' boxes are kept separate
646 // from `out` until the end: the dedup filter below must not confuse a
647 // paragraph we just built with a pre-existing region inside the panel.
648 let mut demoted_paras: Vec<Region> = Vec::new();
649 let mut demoted_boxes: Vec<(f32, f32, f32, f32)> = Vec::new();
650 for (i, r) in regions.drain(..).enumerate() {
651 if r.label != "picture" || captioned[i] {
652 out.push(r);
653 continue;
654 }
655 let inside: Vec<&TextCell> = cells
656 .iter()
657 .filter(|c| {
658 !c.text.trim().is_empty() && {
659 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
660 inter(&r, c.l, c.t, c.r, c.b) / ca > 0.5
661 }
662 })
663 .collect();
664 // Group the contained cells into lines by vertical overlap (the same
665 // rule region_text orders by), tracking each line's union box.
666 let mut lines: Vec<(f32, f32, f32, f32)> = Vec::new(); // (t, b, l, r)
667 for c in &inside {
668 let (ct, cb) = (c.t.min(c.b), c.t.max(c.b));
669 match lines.iter_mut().find(|(lt, lb, _, _)| {
670 let ov = cb.min(*lb) - ct.max(*lt);
671 ov > 0.5 * (cb - ct).min(*lb - *lt).max(1.0)
672 }) {
673 Some((lt, lb, ll, lr)) => {
674 *lt = lt.min(ct);
675 *lb = lb.max(cb);
676 *ll = ll.min(c.l);
677 *lr = lr.max(c.r);
678 }
679 None => lines.push((ct, cb, c.l, c.r)),
680 }
681 }
682 if lines.len() < 3 {
683 out.push(r);
684 continue;
685 }
686 let panel_w = (r.r - r.l).max(1.0);
687 let coverage = inside.iter().map(|c| area(c.l, c.t, c.r, c.b)).sum::<f32>()
688 / area(r.l, r.t, r.r, r.b).max(1.0);
689 let mut widths: Vec<f32> = lines.iter().map(|(_, _, l, rr)| rr - l).collect();
690 widths.sort_by(f32::total_cmp);
691 // A figure's text is ragged: a title line, small axis/tick labels, and
692 // OCR boxes over the plot area come out at wildly different heights,
693 // whereas a real text panel is set in one face with constant leading.
694 // Require near-uniform line heights (median absolute deviation ≤ 35%
695 // of the median) so an uncaptioned chart keeps its crop even when its
696 // labels are dense enough to pass the coverage gate (#173) — garbled
697 // OCR of its bars is not content.
698 let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
699 heights.sort_by(f32::total_cmp);
700 let h_med = heights[heights.len() / 2].max(1.0);
701 let mut devs: Vec<f32> = heights.iter().map(|h| (h - h_med).abs()).collect();
702 devs.sort_by(f32::total_cmp);
703 let uniform = devs[devs.len() / 2] <= 0.35 * h_med;
704 let text_panel = coverage >= 0.2 && widths[widths.len() / 2] >= 0.45 * panel_w && uniform;
705 if !text_panel {
706 out.push(r);
707 continue;
708 }
709 lines.sort_by(|a, b| a.0.total_cmp(&b.0));
710 let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
711 heights.sort_by(f32::total_cmp);
712 let h = heights[heights.len() / 2].max(1.0);
713 let mut gaps: Vec<f32> = lines
714 .windows(2)
715 .map(|w| (w[1].0 - w[0].1).max(0.0))
716 .collect();
717 gaps.sort_by(f32::total_cmp);
718 let leading = if gaps.is_empty() {
719 0.0
720 } else {
721 gaps[gaps.len() / 2]
722 };
723 let brk = (1.8 * leading).max(0.75 * h);
724 let mut para: Option<(f32, f32, f32, f32)> = None; // (l, t, r, b) union
725 for (t, b, l, rr) in &lines {
726 match &mut para {
727 Some((pl, _, pr, pb)) if *t - *pb <= brk => {
728 *pl = pl.min(*l);
729 *pr = pr.max(*rr);
730 *pb = pb.max(*b);
731 }
732 _ => {
733 if let Some((pl, pt, pr, pb)) = para.take() {
734 demoted_paras.push(Region {
735 label: "text",
736 score: r.score,
737 l: pl,
738 t: pt,
739 r: pr,
740 b: pb,
741 });
742 }
743 para = Some((*l, *t, *rr, *b));
744 }
745 }
746 }
747 if let Some((pl, pt, pr, pb)) = para {
748 demoted_paras.push(Region {
749 label: "text",
750 score: r.score,
751 l: pl,
752 t: pt,
753 r: pr,
754 b: pb,
755 });
756 }
757 demoted_boxes.push((r.l, r.t, r.r, r.b));
758 }
759 // The paragraphs are rebuilt from *all* of the panel's cells, so any
760 // surviving text region inside a demoted panel (an orphan cluster or a
761 // layout-detected fragment — pictures no longer swallow them, #165) would
762 // say the same words twice. Consume those; wrappers and pictures stay.
763 if !demoted_boxes.is_empty() {
764 out.retain(|r| {
765 r.label == "picture" || is_wrapper(r.label) || {
766 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
767 !demoted_boxes
768 .iter()
769 .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.5)
770 }
771 });
772 }
773 out.extend(demoted_paras);
774 *regions = out;
775}
776
777/// Drop a `picture` detection that is a small, empty, low-confidence margin box on
778/// a **text page** — a false positive the RT-DETR layout sometimes emits (e.g.
779/// `right_to_left_02`'s phantom right-column picture, score 0.40); docling does not
780/// emit it. The gate is deliberately narrow so a genuine figure is never dropped:
781/// (1) only on pages with a digital text layer — image/scanned/figure pages have
782/// no `cells` yet at this point (OCR runs later), so their pictures, which *are*
783/// the content, are kept; (2) only a box covering < 25 % of the page (a margin
784/// artifact, not a dominant figure); (3) only when it contains no text and scores
785/// below 0.5 (real empty figures in the corpus all score ≥ 0.86).
786pub fn drop_false_pictures(
787 regions: &mut Vec<Region>,
788 cells: &[TextCell],
789 page_w: f32,
790 page_h: f32,
791) {
792 if cells.iter().all(|c| c.text.trim().is_empty()) {
793 return; // no digital text layer (image/scanned page) — keep all pictures
794 }
795 // A text-document page carries several text-bearing non-picture regions (so a
796 // spurious margin picture is clearly extra). A slide / figure page has at most
797 // one — there the picture is the content, so never drop it.
798 let content_regions = regions
799 .iter()
800 .filter(|r| r.label != "picture" && !region_text(r, cells).trim().is_empty())
801 .count();
802 if content_regions < 2 {
803 return;
804 }
805 let page_area = (page_w * page_h).max(1.0);
806 regions.retain(|r| {
807 if r.label != "picture" || r.score >= 0.5 {
808 return true;
809 }
810 if area(r.l, r.t, r.r, r.b) / page_area >= 0.25 {
811 return true; // a dominant figure, not a margin artifact
812 }
813 // Keep it if any text cell falls mostly inside (a real captioned/labelled
814 // figure); drop only the genuinely empty low-confidence boxes.
815 cells.iter().any(|c| {
816 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
817 !c.text.trim().is_empty() && inter(r, c.l, c.t, c.r, c.b) / ca > 0.5
818 })
819 });
820}
821
822/// A small digit-only region in the top/bottom margin: a page number. docling
823/// emits `right_to_left_02`'s bottom `11` as the page's *first* text item (its
824/// reading-order model floats the page number to the front), whereas our
825/// position-based ordering would place a bottom region last.
826fn is_page_number(region: &Region, cells: &[TextCell], page_h: f32) -> bool {
827 let t = region_text(region, cells);
828 let t = t.trim();
829 !t.is_empty()
830 && t.chars().all(|c| c.is_ascii_digit())
831 && (region.b - region.t).abs() < 30.0
832 && (region.t < page_h * 0.12 || region.b > page_h * 0.88)
833}
834
835/// docling's `form` / `key_value_region` *containers* (2.123, docling#4064):
836/// every region sitting > 0.8 inside one — text, list items, and since #4064
837/// tables and pictures too — is that container's child. Children are
838/// reading-ordered among themselves and emitted as one block where the
839/// container falls in the page's top-level order (a `form_area` /
840/// `key_value_area` group upstream), instead of interleaving with the text
841/// around the form. A child inside several containers belongs to the smallest
842/// (then most confident, then first); a container with children shrinks to
843/// their union for the top-level ordering, like upstream's bbox adjustment.
844///
845/// The containers themselves are still not emitted (`is_skipped`), so the
846/// Markdown is exactly upstream's — a group prints only its children.
847///
848/// `cids` are the items' positions in docling's assembly order
849/// ([`cluster_cids`]) — the reading-order predictor's same-row rule (#424)
850/// pairs consecutive ones, within the top level and within each container.
851fn order_with_containers<T: Clone>(
852 items: &mut Vec<T>,
853 cids: &[usize],
854 page_w: f32,
855 page_h: f32,
856 reg: impl Fn(&T) -> &Region,
857) {
858 let is_container = |r: &Region| matches!(r.label, "form" | "key_value_region");
859 let containers: Vec<usize> = (0..items.len())
860 .filter(|&i| is_container(reg(&items[i])))
861 .collect();
862 if containers.is_empty() {
863 order_regions(items, cids, page_w, page_h, reg);
864 return;
865 }
866 // Parent container per item (containers never nest in each other here —
867 // upstream assigns regulars and tables/pictures only).
868 let mut parent: Vec<Option<usize>> = vec![None; items.len()];
869 for i in 0..items.len() {
870 let r = reg(&items[i]);
871 if is_container(r) {
872 continue;
873 }
874 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
875 let mut best: Option<(usize, f32, f32)> = None; // (idx, area, -score)
876 for &c in &containers {
877 let cr = reg(&items[c]);
878 if inter(r, cr.l, cr.t, cr.r, cr.b) / ra > 0.8 {
879 let key = (area(cr.l, cr.t, cr.r, cr.b), -cr.score);
880 if best.is_none_or(|(_, a, s)| key.0 < a || (key.0 == a && key.1 < s)) {
881 best = Some((c, key.0, key.1));
882 }
883 }
884 }
885 parent[i] = best.map(|(c, _, _)| c);
886 }
887 // Top-level pass: non-children plus the containers, the latter shrunk to
888 // their children's union.
889 let mut top: Vec<(usize, Region)> = Vec::new();
890 for i in 0..items.len() {
891 if parent[i].is_some() {
892 continue;
893 }
894 let mut r = reg(&items[i]).clone();
895 if is_container(&r) {
896 let kids: Vec<&Region> = (0..items.len())
897 .filter(|&k| parent[k] == Some(i))
898 .map(|k| reg(&items[k]))
899 .collect();
900 if !kids.is_empty() {
901 r.l = kids.iter().map(|k| k.l).fold(f32::INFINITY, f32::min);
902 r.t = kids.iter().map(|k| k.t).fold(f32::INFINITY, f32::min);
903 r.r = kids.iter().map(|k| k.r).fold(f32::NEG_INFINITY, f32::max);
904 r.b = kids.iter().map(|k| k.b).fold(f32::NEG_INFINITY, f32::max);
905 }
906 }
907 top.push((i, r));
908 }
909 let top_cids: Vec<usize> = top.iter().map(|(i, _)| cids[*i]).collect();
910 order_regions(&mut top, &top_cids, page_w, page_h, |it| &it.1);
911 let mut out: Vec<T> = Vec::with_capacity(items.len());
912 for (i, _) in top {
913 if is_container(reg(&items[i])) {
914 let kid_idx: Vec<usize> = (0..items.len()).filter(|&k| parent[k] == Some(i)).collect();
915 let mut kids: Vec<T> = kid_idx.iter().map(|&k| items[k].clone()).collect();
916 let kid_cids: Vec<usize> = kid_idx.iter().map(|&k| cids[k]).collect();
917 order_regions(&mut kids, &kid_cids, page_w, page_h, ®);
918 out.push(items[i].clone());
919 out.extend(kids);
920 } else {
921 out.push(items[i].clone());
922 }
923 }
924 *items = out;
925}
926
927/// Furniture / not-yet-emitted labels.
928fn is_skipped(label: &str) -> bool {
929 matches!(
930 label,
931 "page_header" | "page_footer" | "form" | "key_value_region"
932 )
933}
934
935/// Reading-order sort of a page's regions, via the ported rule-based
936/// [`reading_order`](crate::reading_order) predictor (docling's
937/// `ReadingOrderPredictor`): an up/down geometry graph with same-row links
938/// between `cids`-consecutive elements (#424), horizontal dilation and a
939/// depth-first traversal, with `page_header`/`page_footer` ordered as their own
940/// groups (first/last) as docling does.
941fn order_regions<T: Clone>(
942 items: &mut Vec<T>,
943 cids: &[usize],
944 page_w: f32,
945 page_h: f32,
946 reg: impl Fn(&T) -> &Region,
947) {
948 let boxes: Vec<(f32, f32, f32, f32)> = items
949 .iter()
950 .map(|it| {
951 let r = reg(it);
952 (r.l, r.t, r.r, r.b)
953 })
954 .collect();
955 let is_header: Vec<bool> = items
956 .iter()
957 .map(|it| reg(it).label == "page_header")
958 .collect();
959 let is_footer: Vec<bool> = items
960 .iter()
961 .map(|it| reg(it).label == "page_footer")
962 .collect();
963 let order =
964 crate::reading_order::order_page(&boxes, cids, &is_header, &is_footer, page_w, page_h);
965 *items = order.iter().map(|&i| items[i].clone()).collect();
966}
967
968/// docling's assembly order of a page's clusters (`LayoutPostprocessor`'s
969/// final `_sort_clusters(mode="id")`, #424): each region's rank when sorted by
970/// its first source cell, then by top edge, then left edge; a region with no
971/// cells sorts after every one that has some. docling numbers its page
972/// elements (`cid`) in this order, and the reading-order predictor's same-row
973/// rule pairs elements with consecutive numbers, so the ranks are what
974/// [`order_with_containers`] hands the predictor.
975///
976/// A regular region's first cell is the smallest index among the cells it
977/// claims. A table, picture or container has no cells of its own upstream
978/// either — its cells are its *children's*: the regular clusters > 0.8 inside
979/// it, and upstream every cell no regular cluster claimed is an orphan cluster
980/// of its own, so a table's interior text (which no regular cluster claims)
981/// reaches the table through those orphans. Here that is the cells > 0.8
982/// inside the region plus the claimed cells of the regular regions > 0.8
983/// inside it. Without the interior cells every table would sort last, and two
984/// side-by-side tables would then be consecutive and row-linked — reading the
985/// right table's caption ahead of the left column's headings (2206 page 8).
986pub fn cluster_cids(regions: &[Region], cells: &[TextCell]) -> Vec<usize> {
987 let owned = assign_cells(regions, cells);
988 let first_cell: Vec<usize> = regions
989 .iter()
990 .enumerate()
991 .map(|(i, r)| {
992 if claims_cells(r) {
993 return owned[i].iter().copied().min().unwrap_or(usize::MAX);
994 }
995 let interior = cells
996 .iter()
997 .enumerate()
998 .filter(|(_, c)| {
999 !c.text.trim().is_empty()
1000 && inter(r, c.l, c.t, c.r, c.b) / area(c.l, c.t, c.r, c.b).max(1.0) > 0.8
1001 })
1002 .map(|(ci, _)| ci)
1003 .min();
1004 let children = regions
1005 .iter()
1006 .enumerate()
1007 .filter(|(j, child)| {
1008 *j != i && claims_cells(child) && {
1009 let ca = area(child.l, child.t, child.r, child.b).max(1.0);
1010 inter(r, child.l, child.t, child.r, child.b) / ca > 0.8
1011 }
1012 })
1013 .filter_map(|(j, _)| owned[j].iter().copied().min())
1014 .min();
1015 interior
1016 .into_iter()
1017 .chain(children)
1018 .min()
1019 .unwrap_or(usize::MAX)
1020 })
1021 .collect();
1022 let mut by_source: Vec<usize> = (0..regions.len()).collect();
1023 // Stable, like Python's `sorted`: full ties keep the layout order.
1024 by_source.sort_by(|&a, &b| {
1025 first_cell[a]
1026 .cmp(&first_cell[b])
1027 .then(regions[a].t.total_cmp(®ions[b].t))
1028 .then(regions[a].l.total_cmp(®ions[b].l))
1029 });
1030 let mut cids = vec![0; regions.len()];
1031 for (rank, &i) in by_source.iter().enumerate() {
1032 cids[i] = rank;
1033 }
1034 cids
1035}
1036
1037/// Clean a region's assembled text: undo soft-hyphen line wraps, map curly
1038/// quotes and the ellipsis to ASCII (matching docling), and collapse runs of
1039/// whitespace. pdfium emits the line-wrap hyphen as U+0002 in this corpus
1040/// (U+00AD elsewhere), so `word\u{2} continuation` is one hyphenated word —
1041/// drop the hyphen + the joining space and merge (`com\u{2} pact` → `compact`,
1042/// `end-to\u{2} end` → `end-toend`), exactly as docling does.
1043///
1044/// Token spacing is otherwise left as the geometric join produced it. We do not
1045/// tighten punctuation spacing: docling preserves the PDF's own spaces (it keeps
1046/// `{ ahn }`, `Name 1 .`, `[ 9 ]`), and a geometric gap heuristic diverges from
1047/// it more than a plain single-space join does.
1048/// An ordered-list enumeration marker at the start of a list item: leading ASCII
1049/// digits followed by `.`, e.g. `1. Undo/Redo` → `(1, "Undo/Redo")`. Returns
1050/// `None` when the text doesn't start with `digits.`.
1051fn parse_ordered_marker(s: &str) -> Option<(u64, String)> {
1052 let digits: String = s.chars().take_while(|c| c.is_ascii_digit()).collect();
1053 if digits.is_empty() {
1054 return None;
1055 }
1056 let rest = s[digits.len()..].strip_prefix('.')?;
1057 let number = digits.parse().ok()?;
1058 Some((number, rest.trim_start().to_string()))
1059}
1060
1061/// Escape markdown special characters the way docling-core's markdown serializer
1062/// does (`markdown.py` post_process): `_` → `\_`, then HTML-escape `&`, `<`, `>`
1063/// (quote=False, so quotes are left). Applied to prose (headings, list items,
1064/// paragraphs); code blocks, the formula placeholder, and table cells are left raw.
1065fn md_escape(text: &str) -> String {
1066 text.replace('_', "\\_")
1067 .replace('&', "&")
1068 .replace('<', "<")
1069 .replace('>', ">")
1070}
1071
1072fn clean_text(text: &str) -> String {
1073 // Typographic-quote normalization follows docling-parse's sanitizer table
1074 // (`pdf_sanitators/constants.h`): every curly quote — single *and double* —
1075 // becomes the ASCII apostrophe `'`, and `‚` a comma. A `"` in docling's
1076 // output only ever comes from a literal `quotedbl` glyph, never from `“ ”`
1077 // (2206's `'text in the wild"` pairs a curly open with a literal-quote
1078 // close). This replaces an earlier Hangul-only special case that patched
1079 // one symptom of mapping `“ ”` to `"`.
1080 let replaced = text
1081 .replace("\u{2} ", "")
1082 .replace("\u{ad} ", "")
1083 .replace(['\u{2}', '\u{ad}'], "") // any stray wrap hyphens not at a join
1084 .replace(
1085 [
1086 '\u{2018}', '\u{2019}', '\u{201b}', '\u{201c}', '\u{201d}', '\u{201e}', '\u{201f}',
1087 ],
1088 "'",
1089 ) // ‘ ’ ‛ “ ” „ ‟ → '
1090 .replace('\u{201a}', ",") // ‚ → ,
1091 .replace(
1092 [
1093 '\u{2010}', '\u{2011}', '\u{2012}', '\u{2013}', '\u{2014}', '\u{2015}', '\u{2212}',
1094 ],
1095 "-",
1096 ) // hyphen/dash family → -
1097 .replace('\u{2044}', "/") // ⁄ fraction slash → /
1098 .replace('\u{2022}', "\u{b7}") // • → · (docling never emits •; inline CCS-concept separators)
1099 .replace('\u{2026}', "..."); // … → ...
1100 let out = if crate::pdfium_backend::use_dp_lines() {
1101 // The docling-parse sanitizer already placed the correct spacing (e.g.
1102 // justified double spaces); preserve internal runs of spaces, only
1103 // normalizing line breaks/tabs and trimming the ends.
1104 replaced.replace(['\n', '\r', '\t'], " ").trim().to_string()
1105 } else {
1106 // Legacy: collapse all whitespace runs to single spaces.
1107 replaced.split_whitespace().collect::<Vec<_>>().join(" ")
1108 };
1109 fix_arabic_lam_alef(&out)
1110}
1111
1112/// pdfium decomposes the Arabic lam-alef ligature (لا / لإ / لأ / لآ) into its
1113/// glyph constituents in *visual* order — `alef-variant, lam` — but docling keeps
1114/// logical order, `lam, alef-variant`. Swap a mid-word `alef-variant + lam` back
1115/// to `lam + alef-variant`. "Mid-word" (the previous char is an Arabic letter)
1116/// distinguishes the ligature from the definite article `ال` (word-initial
1117/// `alef + lam`), which must stay. No-op for non-Arabic text.
1118fn fix_arabic_lam_alef(s: &str) -> String {
1119 let is_arabic_letter = |c: char| ('\u{0620}'..='\u{064A}').contains(&c);
1120 let chars: Vec<char> = s.chars().collect();
1121 if !chars.iter().any(|&c| is_arabic_letter(c)) {
1122 return s.to_string(); // no-op for non-Arabic text
1123 }
1124 // Pass 1: swap mid-word `alef-variant + lam` → `lam + alef-variant`. Only the
1125 // hamza/madda alef variants (إ أ آ) are safe: the definite article is always
1126 // plain `ا + ل`, so plain `alef + lam` is ambiguous (a legitimate `فعالة` vs a
1127 // reversed `لا` ligature look identical) — leaving plain alef alone avoids
1128 // corrupting legitimate words.
1129 let mut a: Vec<char> = Vec::with_capacity(chars.len());
1130 let mut i = 0;
1131 while i < chars.len() {
1132 let c = chars[i];
1133 if matches!(c, '\u{0622}' | '\u{0623}' | '\u{0625}')
1134 && chars.get(i + 1) == Some(&'\u{0644}')
1135 && i > 0
1136 && is_arabic_letter(chars[i - 1])
1137 // A preceding lam means this alef-variant is *already* the logical
1138 // `lam + alef` ligature; the following lam is the next syllable's
1139 // letter, not a reversed ligature — swapping it corrupts `لآل` → `للآ`
1140 // (e.g. التعلم الآلي → الآلي, not اللآي).
1141 && chars[i - 1] != '\u{0644}'
1142 {
1143 a.push('\u{0644}');
1144 a.push(c);
1145 i += 2;
1146 continue;
1147 }
1148 a.push(c);
1149 i += 1;
1150 }
1151 // Pass 2: insert a space at Arabic↔Latin boundaries (bidi script switch) that
1152 // pdfium runs together — docling separates the embedded Latin run (`وPython`
1153 // → `و Python`).
1154 let mut out: Vec<char> = Vec::with_capacity(a.len());
1155 for (j, &c) in a.iter().enumerate() {
1156 if j > 0 {
1157 let p = a[j - 1];
1158 if (is_arabic_letter(p) && c.is_ascii_alphabetic())
1159 || (p.is_ascii_alphabetic() && is_arabic_letter(c))
1160 {
1161 out.push(' ');
1162 }
1163 }
1164 out.push(c);
1165 }
1166 out.into_iter().collect()
1167}
1168
1169/// docling's `PageAssembleModel._match_hyperlink`: the URI whose link
1170/// annotations cover at least half of the region's box, or `None`. Coverage is
1171/// intersection-over-region-area, **accumulated per URI** — a URL that wraps
1172/// across lines carries several annotation rects that sum toward the same
1173/// target. Ties resolve to the first-seen URI (Python's `max` over dict
1174/// insertion order); the winner still needs `>= 0.5`
1175/// (`_HYPERLINK_COVERAGE_THRESHOLD`).
1176pub(crate) fn region_hyperlink(
1177 region: &Region,
1178 links: &[crate::pdfium_backend::LinkAnnot],
1179) -> Option<String> {
1180 if links.is_empty() {
1181 return None;
1182 }
1183 let area = (region.r - region.l).max(0.0) * (region.b - region.t).max(0.0);
1184 if area <= 0.0 {
1185 return None;
1186 }
1187 let mut coverage: Vec<(&str, f32)> = Vec::new();
1188 for link in links {
1189 let ix = (region.r.min(link.r) - region.l.max(link.l)).max(0.0);
1190 let iy = (region.b.min(link.b) - region.t.max(link.t)).max(0.0);
1191 let c = ix * iy / area;
1192 match coverage.iter_mut().find(|(uri, _)| *uri == link.uri) {
1193 Some((_, acc)) => *acc += c,
1194 None => coverage.push((&link.uri, c)),
1195 }
1196 }
1197 let mut best: Option<(&str, f32)> = None;
1198 for (uri, c) in coverage {
1199 // Strictly greater keeps the first-seen URI on ties, like Python's max.
1200 if best.is_none_or(|(_, bc)| c > bc) {
1201 best = Some((uri, c));
1202 }
1203 }
1204 let (uri, c) = best?;
1205 (c >= 0.5).then(|| normalize_uri(uri))
1206}
1207
1208/// The pydantic-`AnyUrl` normalization docling's hyperlink value passes
1209/// through on its way to the serializer: a URL with an authority but no path
1210/// gains a trailing `/` (`https://arxiv.org` → `https://arxiv.org/`). Other
1211/// AnyUrl canonicalizations (scheme/host lowercasing, percent-encoding) don't
1212/// occur in PDF link annotations in practice, so they are not reproduced.
1213fn normalize_uri(uri: &str) -> String {
1214 if let Some((_, rest)) = uri.split_once("://") {
1215 if !rest.is_empty() && !rest.contains(['/', '?', '#']) {
1216 return format!("{uri}/");
1217 }
1218 }
1219 uri.to_string()
1220}
1221
1222/// Resolve each page hyperlink to the visible text it covers, as `(anchor, uri)`
1223/// in reading order. The anchor is the cells whose centre falls in the link rect,
1224/// joined left-to-right and cleaned the same way prose is (so it matches the
1225/// serialized text), deduped against the immediately-preceding link so pdfium's
1226/// occasional duplicate annotation doesn't double-list. Empty anchors are dropped.
1227pub(crate) fn resolve_link_anchors(page: &PdfPage) -> Vec<(String, String)> {
1228 let mut out: Vec<(String, String)> = Vec::new();
1229 // Use per-word cells, not the line-merged `cells`: a link rect covers a few
1230 // words on a line, and a whole merged line cell would over-capture (its centre
1231 // lands in one link's rect, grabbing the entire line as that link's anchor).
1232 let words = if page.word_cells.is_empty() {
1233 &page.cells
1234 } else {
1235 &page.word_cells
1236 };
1237 for link in &page.links {
1238 // A cell participates when its centre row is inside the rect and it
1239 // overlaps the rect horizontally. A cell can be *wider* than the rect:
1240 // PDFs often draw a whole header line as one text run ("LinkedIn |
1241 // GitHub | Credly"), which docling-parse's word grouping keeps as one
1242 // cell even though each label carries its own link annotation —
1243 // centre-in-rect alone would hand the entire line to every link.
1244 // [`cell_text_in_rect`] clips such a cell to the tokens under the rect.
1245 let mut inside: Vec<(&TextCell, String)> = words
1246 .iter()
1247 .filter(|c| {
1248 let cy = (c.t + c.b) / 2.0;
1249 cy >= link.t && cy <= link.b && c.r.min(link.r) > c.l.max(link.l)
1250 })
1251 .filter_map(|c| {
1252 let text = cell_text_in_rect(c, link.l, link.r);
1253 (!text.is_empty()).then_some((c, text))
1254 })
1255 .collect();
1256 // Reading order: top band then left-to-right (link anchors are LTR).
1257 let band = inside
1258 .iter()
1259 .map(|(c, _)| (c.b - c.t).abs())
1260 .fold(0.0f32, f32::max)
1261 .max(1.0);
1262 inside.sort_by_key(|(c, _)| ((c.t / band).round() as i64, (c.l * 10.0) as i64));
1263 let anchor = clean_text(
1264 &inside
1265 .iter()
1266 .map(|(_, t)| t.trim())
1267 .filter(|t| !t.is_empty())
1268 .collect::<Vec<_>>()
1269 .join(" "),
1270 );
1271 if anchor.is_empty() {
1272 continue;
1273 }
1274 if out
1275 .last()
1276 .is_some_and(|(a, u)| a == &anchor && u == &link.uri)
1277 {
1278 continue;
1279 }
1280 out.push((anchor, link.uri.clone()));
1281 }
1282 out
1283}
1284
1285/// The part of a cell's text that lies under a link rect's x-range. A cell
1286/// fully inside the rect (by centre) returns its whole text. A wider cell is
1287/// split into whitespace tokens whose x-spans are estimated proportionally to
1288/// their character positions (kerning makes this approximate, so selection
1289/// snaps to whole tokens, never characters); tokens whose estimated centre
1290/// falls inside the rect are kept. Returns "" when nothing falls inside.
1291fn cell_text_in_rect(c: &TextCell, l: f32, r: f32) -> String {
1292 let cx = (c.l + c.r) / 2.0;
1293 if cx >= l && cx <= r && c.l >= l - (c.r - c.l) * 0.25 && c.r <= r + (c.r - c.l) * 0.25 {
1294 return c.text.trim().to_string();
1295 }
1296 let chars: Vec<char> = c.text.chars().collect();
1297 let n = chars.len();
1298 if n == 0 || c.r <= c.l {
1299 return String::new();
1300 }
1301 let per = (c.r - c.l) / n as f32;
1302 let mut out: Vec<String> = Vec::new();
1303 let mut token = String::new();
1304 let mut start = 0usize;
1305 // A trailing sentinel space flushes the last token.
1306 for (i, &ch) in chars.iter().enumerate().chain(std::iter::once((n, &' '))) {
1307 if ch.is_whitespace() {
1308 if !token.is_empty() {
1309 let mid = c.l + (start as f32 + (i - start) as f32 / 2.0) * per;
1310 if mid >= l && mid <= r {
1311 out.push(std::mem::take(&mut token));
1312 } else {
1313 token.clear();
1314 }
1315 }
1316 } else {
1317 if token.is_empty() {
1318 start = i;
1319 }
1320 token.push(ch);
1321 }
1322 }
1323 out.join(" ")
1324}
1325
1326/// Cells assigned to a region (best container), in reading order, joined.
1327fn region_text(region: &Region, cells: &[TextCell]) -> String {
1328 let inside: Vec<&TextCell> = cells
1329 .iter()
1330 .filter(|c| {
1331 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1332 inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1333 })
1334 .collect();
1335 cells_text(inside)
1336}
1337
1338/// docling's exclusive cell assignment (`_assign_cells_to_clusters`): every
1339/// non-empty cell goes to the single best-overlapping *regular* region at
1340/// intersection-over-self > 0.2, and each region serializes exactly its
1341/// assigned cells. A cell under two overlapping boxes is emitted once (by the
1342/// better-covering one), and a cell only partially under its region — e.g.
1343/// normal_4pages' big section numeral, ~30 % inside the heading box — still
1344/// joins it (`## 들어가며 1`) instead of leaking as an orphan. Pictures and
1345/// wrappers never claim (docling walks regular clusters only); ties go to the
1346/// first region, like docling's strict `>` best-overlap scan.
1347pub fn region_texts_exclusive(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
1348 let owned = assign_cells(regions, cells);
1349 // Non-claimers (tables/wrappers/pictures) keep the inclusive > 0.5 text:
1350 // docling fills a special cluster's cells from its contained children, and
1351 // downstream table assembly gates on that text being non-empty.
1352 regions
1353 .iter()
1354 .zip(owned)
1355 .map(|(r, cs)| {
1356 if claims_cells(r) {
1357 cells_text(cs.iter().map(|&i| &cells[i]).collect())
1358 } else {
1359 region_text(r, cells)
1360 }
1361 })
1362 .collect()
1363}
1364
1365/// A *regular* region in docling's sense — one that claims cells. Pictures and
1366/// the wrappers (`table`, `document_index`, `form`, `key_value_region`) fill
1367/// their cells from contained children instead.
1368fn claims_cells(r: &Region) -> bool {
1369 r.label != "picture" && !is_wrapper(r.label)
1370}
1371
1372/// docling's `_assign_cells_to_clusters`: each non-empty cell's index goes to
1373/// the single best-overlapping regular region at intersection-over-self > 0.2
1374/// (ties to the first region, like docling's strict `>` scan). One entry per
1375/// region, in region order.
1376fn assign_cells(regions: &[Region], cells: &[TextCell]) -> Vec<Vec<usize>> {
1377 let mut owned: Vec<Vec<usize>> = vec![Vec::new(); regions.len()];
1378 for (ci, c) in cells.iter().enumerate() {
1379 if c.text.trim().is_empty() {
1380 continue;
1381 }
1382 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1383 let mut best: Option<(usize, f32)> = None;
1384 for (i, r) in regions.iter().enumerate() {
1385 if !claims_cells(r) {
1386 continue;
1387 }
1388 let ov = inter(r, c.l, c.t, c.r, c.b) / ca;
1389 if ov > 0.2 && best.is_none_or(|(_, b)| ov > b) {
1390 best = Some((i, ov));
1391 }
1392 }
1393 if let Some((i, _)) = best {
1394 owned[i].push(ci);
1395 }
1396 }
1397 owned
1398}
1399
1400/// docling's regular-cluster refinement after cell assignment
1401/// (`LayoutPostprocessor._process_regular_clusters`, #419), run once the page's
1402/// cells are final and before reading order:
1403///
1404/// 1. every regular region's box becomes the union of the cells it claimed
1405/// (`_adjust_cluster_bboxes` — a regular cluster's bbox *is* its cells'
1406/// bbox; a table's is the union with the model box, and pictures keep
1407/// theirs, so neither is touched here);
1408/// 2. a regular region that claimed no cell is dropped (`keep_empty_clusters`
1409/// is off; a `formula` is kept, as upstream keeps it);
1410/// 3. an orphan text region (`score == 0.0`, from [`add_orphan_regions`]) that
1411/// now sits > 0.8 inside another regular region's fitted box is folded into
1412/// it (`_remove_overlapping_clusters` at containment 0.8, the larger box
1413/// winning the group) — up to three rounds, like upstream's loop.
1414///
1415/// Why it matters: the layout model's box can end partway through a line. That
1416/// line fails the 0.2 claim and becomes an orphan — recoverable — but the
1417/// *model* box still overlaps the orphan's line by a few points, so the
1418/// reading-order graph, which links only strictly-above pairs, gets no edge
1419/// between them and may emit the next paragraph first, stranding the line
1420/// after the paragraph it belongs in (1540 of 6050 text blocks on the #419
1421/// book began mid-sentence). Fitted to its cells, the box ends on a line
1422/// boundary and the orphan slots in between; an orphan the fitted box
1423/// swallows joins the paragraph outright. Cell assignment is untouched: a
1424/// region's fitted box contains every cell it claimed, so
1425/// [`region_texts_exclusive`] hands it the same cells afterwards.
1426///
1427/// A page with no cells yet (a scan before OCR) is left alone: dropping every
1428/// text region for want of cells would be wrong, and the OCR paths call this
1429/// again once the cells exist.
1430pub fn fit_regions_to_cells(regions: &mut Vec<Region>, cells: &[TextCell]) {
1431 if !cells.iter().any(|c| !c.text.trim().is_empty()) {
1432 return;
1433 }
1434 for _ in 0..3 {
1435 let owned = assign_cells(regions, cells);
1436 let mut fitted: Vec<Region> = Vec::with_capacity(regions.len());
1437 for (r, own) in regions.iter().zip(&owned) {
1438 if !claims_cells(r) {
1439 fitted.push(r.clone());
1440 continue;
1441 }
1442 if own.is_empty() {
1443 if r.label == "formula" {
1444 fitted.push(r.clone());
1445 }
1446 continue;
1447 }
1448 let mut f = r.clone();
1449 f.l = own
1450 .iter()
1451 .map(|&i| cells[i].l)
1452 .fold(f32::INFINITY, f32::min);
1453 f.t = own
1454 .iter()
1455 .map(|&i| cells[i].t)
1456 .fold(f32::INFINITY, f32::min);
1457 f.r = own
1458 .iter()
1459 .map(|&i| cells[i].r)
1460 .fold(f32::NEG_INFINITY, f32::max);
1461 f.b = own
1462 .iter()
1463 .map(|&i| cells[i].b)
1464 .fold(f32::NEG_INFINITY, f32::max);
1465 fitted.push(f);
1466 }
1467 let mut changed = fitted.len() != regions.len();
1468 // Fold orphans into the regular region whose fitted box holds them.
1469 let mut drop = vec![false; fitted.len()];
1470 for i in 0..fitted.len() {
1471 let o = &fitted[i];
1472 if !(o.score == 0.0 && o.label == "text") {
1473 continue;
1474 }
1475 let oa = area(o.l, o.t, o.r, o.b).max(1.0);
1476 let mut best: Option<(usize, f32)> = None;
1477 for (j, r) in fitted.iter().enumerate() {
1478 if j == i || drop[j] || r.score == 0.0 || !claims_cells(r) {
1479 continue;
1480 }
1481 let ov = inter(r, o.l, o.t, o.r, o.b) / oa;
1482 if ov > 0.8 && best.is_none_or(|(_, b)| ov > b) {
1483 best = Some((j, ov));
1484 }
1485 }
1486 if let Some((j, _)) = best {
1487 let (l, t, r, b) = (o.l, o.t, o.r, o.b);
1488 let host = &mut fitted[j];
1489 host.l = host.l.min(l);
1490 host.t = host.t.min(t);
1491 host.r = host.r.max(r);
1492 host.b = host.b.max(b);
1493 drop[i] = true;
1494 changed = true;
1495 }
1496 }
1497 let mut drop = drop.into_iter();
1498 fitted.retain(|_| !drop.next().expect("aligned"));
1499 *regions = fitted;
1500 if !changed {
1501 break;
1502 }
1503 }
1504}
1505
1506/// Join a prefiltered cell list into the region's text (docling's
1507/// `sanitize_text` on the docling-parse path, gap-aware band join on legacy).
1508fn cells_text(mut inside: Vec<&TextCell>) -> String {
1509 // Quantize the top coordinate into ~line bands so cells on the same line
1510 // sort in reading order; this is a strict total order (a raw fuzzy comparator
1511 // is not transitive and makes Rust's sort panic). For a right-to-left
1512 // (Arabic-majority) region, cells on a line read right→left, so sort the band
1513 // by descending left edge.
1514 let band = inside
1515 .iter()
1516 .map(|c| (c.b - c.t).abs())
1517 .fold(0.0f32, f32::max)
1518 .max(1.0);
1519 let arabic = inside
1520 .iter()
1521 .flat_map(|c| c.text.chars())
1522 .filter(|&c| ('\u{0600}'..='\u{06FF}').contains(&c))
1523 .count();
1524 let latin = inside
1525 .iter()
1526 .flat_map(|c| c.text.chars())
1527 .filter(|c| c.is_ascii_alphabetic())
1528 .count();
1529 let rtl = arabic > latin;
1530 let dp = crate::pdfium_backend::use_dp_lines();
1531 if dp {
1532 // docling orders a cluster's cells by their docling-parse cell index
1533 // alone (`LayoutPostprocessor._sort_cells`: `sorted(cells, key=c.index)`)
1534 // — the sanitizer's output order, which our `cells` slice already is.
1535 // No geometric re-sort: normal_4pages' big section numerals paint
1536 // *after* their heading text, and docling's `## 들어가며 1` (numeral
1537 // last) only falls out of pure index order — a band sort dragged the
1538 // numeral to the front. The overlap-grouped line restore this replaced
1539 // measured strictly worse on the corpus (it fixed nothing the index
1540 // order broke, and broke the numerals).
1541 } else {
1542 inside.sort_by_key(|c| {
1543 let x = (c.l * 10.0) as i64;
1544 ((c.t / band).round() as i64, if rtl { -x } else { x })
1545 });
1546 }
1547 let joined = if dp {
1548 // docling's `PageAssembleModel.sanitize_text`, ported verbatim over the
1549 // parse-index-ordered lines: append a separating space to a line —
1550 // unless it ends with `-`. A dash-ending line whose last word and the
1551 // next line's first word are both alphanumeric is a wrapped word: the
1552 // dash is dropped and the lines fuse (`platforms-` + `reflects` →
1553 // `platformsreflects`, `pp. 545-` + `561` → `545561`). Any other
1554 // dash-ending line — e.g. the *bare* `-` cell a superscript ORCID or an
1555 // inline `–` bullet splits off (its word list is empty, so the fuse
1556 // test fails) — keeps its dash and still takes no trailing space:
1557 // `[0000` `-` `0002` joins as docling's `[0000 -0002`, and the OTSL
1558 // list's `-` + `"C" cell -` + `a new table cell` collapses to
1559 // `-"C" cell a new table cell`. Our cells still carry the raw dash
1560 // family (docling-parse normalizes to `-` before this; clean_text does
1561 // it after), so the endswith test matches them all.
1562 let texts: Vec<&str> = inside
1563 .iter()
1564 .map(|c| c.text.trim())
1565 // Skip whitespace-only cells (a justified line's trailing space
1566 // glyph): an empty line would double the separator.
1567 .filter(|t| !t.is_empty())
1568 .collect();
1569 let last_word_alnum = |s: &str| {
1570 s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
1571 .rfind(|w| !w.is_empty())
1572 .is_some_and(|w| w.chars().all(char::is_alphanumeric))
1573 };
1574 let first_word_alnum = |s: &str| {
1575 s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
1576 .find(|w| !w.is_empty())
1577 .is_some_and(|w| w.chars().all(char::is_alphanumeric))
1578 };
1579 let mut out = String::new();
1580 for (i, t) in texts.iter().enumerate() {
1581 if i > 0 {
1582 let prev = texts[i - 1];
1583 let dashish = matches!(
1584 prev.chars().last(),
1585 Some(
1586 '-' | '\u{2010}'
1587 | '\u{2011}'
1588 | '\u{2012}'
1589 | '\u{2013}'
1590 | '\u{2014}'
1591 | '\u{2015}'
1592 | '\u{2212}'
1593 )
1594 );
1595 // docling#4052 (2.122): a dash only splits a word when it is
1596 // *attached* to one — the character before it is alphanumeric.
1597 // A dash that follows whitespace (a separator dash, a bullet
1598 // marker, a wrapped `-prefixed` token, the bare `-` cell an
1599 // ORCID splits off) is a literal character: it is kept and the
1600 // lines join with the ordinary space.
1601 let attached = prev.chars().rev().nth(1).is_some_and(char::is_alphanumeric);
1602 if dashish && attached {
1603 if last_word_alnum(prev) && first_word_alnum(t) {
1604 out.pop(); // wrapped word: fuse without the dash
1605 }
1606 // an attached dash never takes a separating space
1607 } else {
1608 out.push(' ');
1609 }
1610 }
1611 out.push_str(t);
1612 }
1613 out
1614 } else {
1615 // Legacy reconstruction: join same-band cells with a space only across a
1616 // real gap, because it can split a word into abutting segments
1617 // (`الت`|`ي` → `التي`).
1618 let mut out = String::new();
1619 let mut prev: Option<&&TextCell> = None;
1620 for c in &inside {
1621 let t = c.text.trim();
1622 if t.is_empty() {
1623 continue;
1624 }
1625 if let Some(p) = prev {
1626 let same_band = ((p.t / band).round() as i64) == ((c.t / band).round() as i64);
1627 let h = (c.b - c.t).abs().max((p.b - p.t).abs()).max(1.0);
1628 let gap = if rtl { p.l - c.r } else { c.l - p.r };
1629 if !same_band || gap > h * 0.25 {
1630 out.push(' ');
1631 }
1632 }
1633 out.push_str(t);
1634 prev = Some(c);
1635 }
1636 out
1637 };
1638 clean_text(&joined)
1639}
1640
1641/// Tighten the spaces pdfium leaves around tight punctuation in a code line
1642/// (`console .log` → `console.log`, `add (3 , 5)` → `add(3, 5)`), matching
1643/// docling-parse's source spacing.
1644fn tighten_code_punct(s: &str) -> String {
1645 s.replace(" .", ".")
1646 .replace(" ,", ",")
1647 .replace(" ;", ";")
1648 .replace(" )", ")")
1649 .replace(" (", "(")
1650}
1651
1652/// Assemble a **code** region's text with its line structure preserved.
1653///
1654/// Unlike [`region_text`] — which joins every cell with a single space, the right
1655/// thing for prose reflow — a code block's line breaks and indentation are
1656/// significant. The `code_cells` are already one physical source line each
1657/// (grouped space-glyph-only, so monospace runs keep their spacing), so this:
1658///
1659/// 1. groups the cells into vertical line bands and orders them top→bottom,
1660/// left→right;
1661/// 2. joins the lines with `\n` (rather than spaces), keeping the carriage
1662/// returns; and
1663/// 3. reconstructs each line's leading indentation from its left offset, in units
1664/// of the block's estimated monospace character width, so nesting survives.
1665///
1666/// Typography is normalized per line via [`clean_text`] (smart quotes, dashes,
1667/// ellipsis), which never merges lines. Returns an empty string if the region has
1668/// no code cells (the caller falls back to the prose text).
1669fn code_region_text(region: &Region, cells: &[TextCell]) -> String {
1670 let mut inside: Vec<&TextCell> = cells
1671 .iter()
1672 .filter(|c| {
1673 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1674 inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1675 })
1676 .filter(|c| !c.text.trim().is_empty())
1677 .collect();
1678 if inside.is_empty() {
1679 return String::new();
1680 }
1681
1682 // Quantize the top edge into ~line bands (like `region_text`), then order the
1683 // cells by band (top→bottom) and, within a band, by left edge.
1684 let band = inside
1685 .iter()
1686 .map(|c| (c.b - c.t).abs())
1687 .fold(0.0f32, f32::max)
1688 .max(1.0);
1689 let line_of = |c: &TextCell| (c.t / band).round() as i64;
1690 inside.sort_by_key(|c| (line_of(c), (c.l * 10.0) as i64));
1691
1692 // Estimate one monospace character's width (total ink width / total glyphs) to
1693 // convert a line's left offset into a count of leading spaces. Measured over
1694 // all lines so a single short line can't skew it.
1695 let (mut total_w, mut total_chars) = (0.0f32, 0usize);
1696 for c in &inside {
1697 let n = c.text.trim().chars().count();
1698 if n > 0 {
1699 total_w += (c.r - c.l).max(0.0);
1700 total_chars += n;
1701 }
1702 }
1703 let char_w = if total_chars > 0 {
1704 (total_w / total_chars as f32).max(1.0)
1705 } else {
1706 1.0
1707 };
1708 // The block's own left margin is the zero-indent baseline.
1709 let base_l = inside.iter().map(|c| c.l).fold(f32::INFINITY, f32::min);
1710
1711 let mut lines: Vec<String> = Vec::new();
1712 let mut cur: Option<i64> = None;
1713 for c in &inside {
1714 // Tighten pdfium's spaced punctuation per line (on the trimmed content, so
1715 // the reconstructed leading indentation is never nibbled).
1716 let text = tighten_code_punct(&clean_text(c.text.trim()));
1717 if Some(line_of(c)) == cur {
1718 // A second cell sharing this band (rare — e.g. split columns): keep it
1719 // on the same source line, separated by a space.
1720 if let Some(last) = lines.last_mut() {
1721 last.push(' ');
1722 last.push_str(&text);
1723 }
1724 continue;
1725 }
1726 let indent = ((c.l - base_l) / char_w).round().max(0.0) as usize;
1727 lines.push(format!("{}{}", " ".repeat(indent), text));
1728 cur = Some(line_of(c));
1729 }
1730 lines.join("\n")
1731}
1732
1733/// Reconstruct a table's grid geometrically from the text cells inside its
1734/// region: cluster cells into rows (by vertical centre) and columns (by clustered
1735/// left edges), then place each cell. A model-free stand-in for TableFormer that
1736/// recovers grid-aligned tables from the precise PDF text layer (it does not
1737/// resolve row/column spans).
1738pub fn reconstruct_table(region: &Region, cells: &[TextCell]) -> Vec<Vec<String>> {
1739 let mut inside: Vec<&TextCell> = cells
1740 .iter()
1741 .filter(|c| {
1742 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1743 inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1744 })
1745 .collect();
1746 if inside.is_empty() {
1747 return Vec::new();
1748 }
1749 inside.sort_by(|a, b| a.t.total_cmp(&b.t));
1750
1751 // Rows: consecutive cells whose vertical centre is within ~0.7 line height.
1752 let mut rows: Vec<(f32, Vec<&TextCell>)> = Vec::new();
1753 for c in &inside {
1754 let cyc = (c.t + c.b) / 2.0;
1755 let lh = (c.b - c.t).abs().max(1.0);
1756 if let Some((ryc, row)) = rows.last_mut() {
1757 if (cyc - *ryc).abs() < lh * 0.7 {
1758 row.push(c);
1759 continue;
1760 }
1761 }
1762 rows.push((cyc, vec![c]));
1763 }
1764
1765 // Columns: cluster left edges (merge those within a tolerance).
1766 let tol = {
1767 let mut hs: Vec<f32> = inside.iter().map(|c| (c.b - c.t).abs()).collect();
1768 hs.sort_by(f32::total_cmp);
1769 hs[hs.len() / 2].max(4.0) * 1.5
1770 };
1771 let mut lefts: Vec<f32> = inside.iter().map(|c| c.l).collect();
1772 lefts.sort_by(f32::total_cmp);
1773 let mut col_starts: Vec<f32> = Vec::new();
1774 for l in lefts {
1775 if col_starts.last().is_none_or(|&last| l - last > tol) {
1776 col_starts.push(l);
1777 }
1778 }
1779 let ncols = col_starts.len().max(1);
1780 let col_of = |l: f32| -> usize {
1781 col_starts
1782 .iter()
1783 .rposition(|&s| l + tol * 0.5 >= s)
1784 .unwrap_or(0)
1785 .min(ncols - 1)
1786 };
1787
1788 let mut grid = Vec::with_capacity(rows.len());
1789 for (_, mut row) in rows {
1790 row.sort_by(|a, b| a.l.total_cmp(&b.l));
1791 let mut cols = vec![String::new(); ncols];
1792 for c in row {
1793 let ci = col_of(c.l);
1794 // Strip the wrap-hyphen control char so it never lands in a cell.
1795 let t = c.text.trim().replace(['\u{2}', '\u{ad}'], "");
1796 if cols[ci].is_empty() {
1797 cols[ci] = t;
1798 } else {
1799 cols[ci].push(' ');
1800 cols[ci].push_str(&t);
1801 }
1802 }
1803 grid.push(cols);
1804 }
1805 grid
1806}
1807
1808/// Does the geometric reconstruction of a table look trustworthy enough to use
1809/// as-is, instead of paying for TableFormer?
1810///
1811/// [`reconstruct_table`] derives columns by clustering cell **left edges**. On a
1812/// clean grid that is exact, but when a column's entries are not left-aligned
1813/// (or the OCR boxes wobble) the clustering splits one real column into several,
1814/// and the result is a wide, mostly-empty grid — the "spurious empty columns"
1815/// failure TableFormer exists to fix.
1816///
1817/// Two symptoms separate the two cases, and both are properties of the grid
1818/// alone (no model needed):
1819/// * **density** — a real table is mostly full; a split-up one is mostly holes;
1820/// * **thin columns** — a column carrying at most one entry across several rows
1821/// is almost always a split artefact rather than a real column.
1822///
1823/// Deliberately conservative: it answers `true` only for grids that are plainly
1824/// well-formed, so the expensive path stays the default whenever there is doubt.
1825/// A caller that skips TableFormer on `true` trades no quality for the time.
1826pub fn geometric_table_is_reliable(rows: &[Vec<String>]) -> bool {
1827 let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
1828 // Fewer than two columns is not a grid this heuristic can vouch for: it is
1829 // exactly the shape a collapsed table takes, and TableFormer may recover
1830 // real structure from it.
1831 if rows.len() < 2 || ncols < 2 {
1832 return false;
1833 }
1834 let filled = |c: &String| !c.trim().is_empty();
1835 let total = rows.len() * ncols;
1836 let full = rows.iter().flatten().filter(|c| filled(c)).count();
1837 if (full as f32) < MIN_TABLE_FILL * total as f32 {
1838 return false;
1839 }
1840 // A column used by at most one row, when there are rows enough to tell.
1841 if rows.len() >= 3 {
1842 for ci in 0..ncols {
1843 let used = rows
1844 .iter()
1845 .filter(|r| r.get(ci).is_some_and(filled))
1846 .count();
1847 if used <= 1 {
1848 return false;
1849 }
1850 }
1851 }
1852 true
1853}
1854
1855/// Share of a geometric grid's cells that must carry text for it to be trusted
1856/// without TableFormer. Chosen well above the density a left-edge split
1857/// produces (those land nearer a third) and below what a genuine table with a
1858/// few blank cells reaches.
1859const MIN_TABLE_FILL: f32 = 0.6;
1860
1861/// The union bbox of the text cells assigned to a region (same >50%-overlap
1862/// rule as [`region_text`]), or `None` when no cell lands in it. docling's
1863/// LayoutPostprocessor shrinks a regular cluster's bbox to its cells, and the
1864/// enrichment crops are taken from that cell-tight box — cropping the raw
1865/// detector box instead hands the VLM surrounding chrome (e.g. the `Listing N:`
1866/// caption under a code block) that changes its output.
1867pub fn region_cell_bbox(region: &Region, cells: &[TextCell]) -> Option<[f32; 4]> {
1868 let mut bbox: Option<[f32; 4]> = None;
1869 for c in cells {
1870 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1871 if inter(region, c.l, c.t, c.r, c.b) / ca <= 0.5 {
1872 continue;
1873 }
1874 bbox = Some(match bbox {
1875 None => [c.l, c.t, c.r, c.b],
1876 Some([l, t, r, b]) => [l.min(c.l), t.min(c.t), r.max(c.r), b.max(c.b)],
1877 });
1878 }
1879 bbox
1880}
1881
1882/// One region's enrichment-model result, produced by the pipeline's opt-in
1883/// passes (issue #76) and applied during assembly.
1884#[derive(Debug, Clone)]
1885pub enum Enrichment {
1886 /// DocumentPictureClassifier predictions, descending confidence.
1887 PictureClasses(Vec<PictureClass>),
1888 /// CodeFormulaV2 output for a `code` region: the rewritten source text and
1889 /// the `<_language_>` prefix (when the model emitted one).
1890 Code {
1891 language: Option<String>,
1892 text: String,
1893 },
1894 /// CodeFormulaV2 output for a `formula` region: the decoded LaTeX.
1895 Formula { latex: String },
1896}
1897
1898/// Crop a region (page points, already expanded by the caller if needed) from
1899/// the rendered page image and resize it to `target_scale` pixels per point —
1900/// the enrichment-model equivalent of docling's
1901/// `page.get_image(scale=…, cropbox=…)`, sourced from the existing
1902/// [`crate::pdfium_backend::RENDER_SCALE`] render instead of a fresh pdfium
1903/// pass (the page bitmap is already the exact docling render at scale 2).
1904#[cfg(feature = "ml")]
1905pub fn crop_region_scaled(page: &PdfPage, bbox: [f32; 4], target_scale: f32) -> Option<RgbImage> {
1906 let s = page.scale;
1907 let [l, t, r, b] = bbox;
1908 let (iw, ih) = (page.image.width(), page.image.height());
1909 let x = (l * s).max(0.0) as u32;
1910 let y = (t * s).max(0.0) as u32;
1911 if x >= iw || y >= ih {
1912 return None;
1913 }
1914 let w = (((r - l.max(0.0)) * s) as u32).min(iw - x);
1915 let h = (((b - t.max(0.0)) * s) as u32).min(ih - y);
1916 if w == 0 || h == 0 {
1917 return None;
1918 }
1919 let crop = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
1920 // docling renders the crop at `target_scale` directly; from the scale-2
1921 // page render that is a resize to the same pixel geometry
1922 // (`round(width_points * scale)`, PIL's BICUBIC ≙ CatmullRom).
1923 let tw = ((w as f32 / s) * target_scale).round().max(1.0) as u32;
1924 let th = ((h as f32 / s) * target_scale).round().max(1.0) as u32;
1925 if (tw, th) == (w, h) {
1926 return Some(crop);
1927 }
1928 Some(image::imageops::resize(
1929 &crop,
1930 tw,
1931 th,
1932 image::imageops::FilterType::CatmullRom,
1933 ))
1934}
1935
1936/// Crop a layout region from the rendered page image and encode it as PNG (the
1937/// figure bytes docling stores on a `PictureItem`). Region coordinates are page
1938/// points; the image is rendered at `page.scale`.
1939#[cfg(feature = "ocr-prep")]
1940fn crop_region(page: &PdfPage, region: &Region) -> Option<PictureImage> {
1941 let s = page.scale;
1942 let (iw, ih) = (page.image.width(), page.image.height());
1943 let x = (region.l * s).max(0.0) as u32;
1944 let y = (region.t * s).max(0.0) as u32;
1945 if x >= iw || y >= ih {
1946 return None;
1947 }
1948 let w = (((region.r - region.l) * s) as u32).min(iw - x);
1949 let h = (((region.b - region.t) * s) as u32).min(ih - y);
1950 if w == 0 || h == 0 {
1951 return None;
1952 }
1953 let sub = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
1954 let mut buf = std::io::Cursor::new(Vec::new());
1955 sub.write_to(&mut buf, image::ImageFormat::Png).ok()?;
1956 Some(PictureImage {
1957 mimetype: "image/png".into(),
1958 width: w,
1959 height: h,
1960 data: buf.into_inner(),
1961 })
1962}
1963
1964/// For each `picture` region, find the `caption` region closest below it (and
1965/// horizontally overlapping); docling pairs them and emits the caption first.
1966/// Each caption is claimed by at most one picture.
1967fn pair_captions(regions: &[Region]) -> Vec<Option<usize>> {
1968 let mut pairs = vec![None; regions.len()];
1969 let mut taken = vec![false; regions.len()];
1970 for (pi, p) in regions.iter().enumerate() {
1971 if p.label != "picture" {
1972 continue;
1973 }
1974 let mut best: Option<(usize, f32)> = None;
1975 for (ci, c) in regions.iter().enumerate() {
1976 if c.label != "caption" || taken[ci] {
1977 continue;
1978 }
1979 let line_h = (c.b - c.t).abs().max(1.0);
1980 let gap = c.t - p.b; // caption sits below the picture
1981 let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
1982 if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
1983 let dist = gap.abs();
1984 if best.is_none_or(|(_, bd)| dist < bd) {
1985 best = Some((ci, dist));
1986 }
1987 }
1988 }
1989 if let Some((ci, _)) = best {
1990 pairs[pi] = Some(ci);
1991 taken[ci] = true;
1992 }
1993 }
1994 pairs
1995}
1996
1997/// Pair each `code` region with the `caption` region just **above** it (a
1998/// `Listing N:` label). docling renders the code block first, then its caption,
1999/// so the caption is consumed from its own (earlier) reading-order slot and
2000/// re-emitted after the code.
2001fn pair_code_captions(regions: &[Region]) -> Vec<Option<usize>> {
2002 let mut pairs = vec![None; regions.len()];
2003 let mut taken = vec![false; regions.len()];
2004 for (pi, p) in regions.iter().enumerate() {
2005 if p.label != "code" {
2006 continue;
2007 }
2008 let mut best: Option<(usize, f32)> = None;
2009 for (ci, c) in regions.iter().enumerate() {
2010 if c.label != "caption" || taken[ci] {
2011 continue;
2012 }
2013 let line_h = (c.b - c.t).abs().max(1.0);
2014 let gap = p.t - c.b; // caption sits above the code
2015 let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
2016 if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
2017 let dist = gap.abs();
2018 if best.is_none_or(|(_, bd)| dist < bd) {
2019 best = Some((ci, dist));
2020 }
2021 }
2022 }
2023 if let Some((ci, _)) = best {
2024 pairs[pi] = Some(ci);
2025 taken[ci] = true;
2026 }
2027 }
2028 pairs
2029}
2030
2031/// Pair each `table`/`document_index` region with its `caption` (#265) the way
2032/// docling's `ReadingOrderPredictor._find_to_captions` does: by **reading-order
2033/// adjacency**, not geometry. A caption claims the media element
2034/// (table/picture/code) immediately next to it in the ordered region sequence,
2035/// and only when exactly one side holds one — a caption sandwiched between two
2036/// media elements stays unattached, and a text paragraph between caption and
2037/// table breaks the bond. This is what lets a flush-left "Table 3: …" label
2038/// bind a centered grid it doesn't horizontally overlap, while a caption in
2039/// the neighbouring column of a two-column page — geometrically close — never
2040/// pairs across the gutter. Runs after the picture and code pairings (the
2041/// picture/code arms of the same upstream matcher), so a caption they claimed
2042/// stays claimed. docling attaches these as `TableItem.captions` refs; the
2043/// paired caption is consumed from its own reading-order slot and rides on the
2044/// table node instead.
2045fn pair_table_captions(regions: &[Region], taken: &mut [bool]) -> Vec<Option<usize>> {
2046 let is_media = |label: &str| is_table_like(label) || matches!(label, "picture" | "code");
2047 let mut pairs: Vec<Option<usize>> = vec![None; regions.len()];
2048 for ci in 0..regions.len() {
2049 if regions[ci].label != "caption" || taken[ci] {
2050 continue;
2051 }
2052 // Furniture (headers/footers, form chrome) is not part of docling's
2053 // body-element sequence, so it neither bonds nor blocks.
2054 let prev = regions[..ci].iter().rposition(|r| !is_skipped(r.label));
2055 let next = regions[ci + 1..]
2056 .iter()
2057 .position(|r| !is_skipped(r.label))
2058 .map(|off| ci + 1 + off);
2059 let prev_media = prev.is_some_and(|j| is_media(regions[j].label));
2060 let next_media = next.is_some_and(|j| is_media(regions[j].label));
2061 let target = match (prev_media, next_media) {
2062 (true, false) => prev,
2063 (false, true) => next,
2064 // Ambiguous (media on both sides) or no media at all: leave the
2065 // caption in its own reading-order slot, as docling does.
2066 _ => None,
2067 };
2068 if let Some(ti) = target {
2069 // A first claim wins (a table with captions above *and* below
2070 // keeps the earlier one — docling's nearest-first tiebreak).
2071 if is_table_like(regions[ti].label) && pairs[ti].is_none() {
2072 pairs[ti] = Some(ci);
2073 taken[ci] = true;
2074 }
2075 }
2076 }
2077 pairs
2078}
2079
2080/// Assemble one page from its (already overlap-resolved) layout regions and
2081/// text cells.
2082/// Normalize a layout region (page points, top-left origin) to DocLang's 0–511
2083/// location grid: `clamp(round(512 · coord / page_dim), 0, 511)`, per axis,
2084/// order `[x0, y0, x1, y1]`. Mirrors docling_core's
2085/// `_doclang_utils._create_location_tokens_for_bbox` (resolution 512) so the
2086/// emitted `<location>` tokens line up with the Python groundtruth. Our heron
2087/// cluster boxes match docling's to within ~1 grid unit; the residual (mainly
2088/// the aspect-ratio-stretch vs letterbox preprocessing difference) is absorbed
2089/// by the conformance harness's geometry tolerance.
2090fn norm_loc(region: &Region, page_w: f32, page_h: f32) -> [u16; 4] {
2091 let q = |v: f32, dim: f32| -> u16 {
2092 if dim <= 0.0 {
2093 return 0;
2094 }
2095 let g = (512.0 * (v as f64) / (dim as f64)).round() as i64;
2096 g.clamp(0, 511) as u16
2097 };
2098 [
2099 q(region.l, page_w),
2100 q(region.t, page_h),
2101 q(region.r, page_w),
2102 q(region.b, page_h),
2103 ]
2104}
2105
2106/// Wrap a node in its layout provenance so the DocLang serializer emits the four
2107/// `<location>` tokens as the element's head (Markdown/JSON render `inner`
2108/// unchanged).
2109fn located(loc: [u16; 4], inner: Node) -> Node {
2110 Node::Located {
2111 location: loc,
2112 inner: Box::new(inner),
2113 }
2114}
2115
2116/// Stamp the real 1-based page number onto a page's leading marker (see
2117/// [`assemble_page`], which emits it with `page_no: 0` because only the
2118/// document-level collector knows the true index — `--pages` windows shift it).
2119pub fn stamp_page_no(nodes: &mut [Node], page_no: usize) {
2120 if let Some(Node::PageInfo { page_no: p, .. }) = nodes.first_mut() {
2121 *p = page_no;
2122 }
2123}
2124
2125/// A dense table grid plus its first-class cells (#240): `rows` is the text
2126/// grid every serializer renders (spans replicate their anchor's text);
2127/// `cells` are the docling-parity per-cell records (text, page-point bbox,
2128/// span rectangle, OTSL header roles). Produced by the TableFormer paths
2129/// (`tf_core`); lives in this always-compiled module so the pure-text (wasm
2130/// `pdf-text`) build sees the type.
2131#[derive(Clone, Debug)]
2132pub struct TableGrid {
2133 pub rows: Vec<Vec<String>>,
2134 pub cells: Vec<docling_core::TableCell>,
2135}
2136
2137/// docling's `_RICH_CELL_PICTURE_COVERAGE_THRESHOLD`.
2138const RICH_CELL_PICTURE_COVERAGE: f32 = 0.8;
2139
2140/// docling `ReadingOrderModel._match_table_pictures` (#3906, 2.118.1): every
2141/// picture ≥ 80 % inside a TableFormer-structured table on the page is matched
2142/// to the cell covering it, and returned per table as `cell index → pictures`.
2143/// A picture that pairs with a caption stays a standalone figure (upstream
2144/// would nest it and lose the caption; keeping the caption is the better
2145/// failure). Tables without first-class cells (geometric fallback) have no cell
2146/// boxes to match against and nest nothing.
2147fn match_table_pictures(
2148 regions: &[Region],
2149 table_rows: &[Option<TableGrid>],
2150 caption_for: &[Option<usize>],
2151) -> std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> {
2152 let mut out: std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> =
2153 std::collections::HashMap::new();
2154 for (p, pic) in regions.iter().enumerate() {
2155 if pic.label != "picture" || caption_for.get(p).is_some_and(Option::is_some) {
2156 continue;
2157 }
2158 let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2159 let mut best: Option<(f32, usize, usize)> = None; // (coverage, table, cell)
2160 for (t, tbl) in regions.iter().enumerate() {
2161 if !is_table_like(tbl.label) {
2162 continue;
2163 }
2164 let Some(grid) = table_rows.get(t).and_then(Option::as_ref) else {
2165 continue;
2166 };
2167 if inter(pic, tbl.l, tbl.t, tbl.r, tbl.b) / pa < RICH_CELL_PICTURE_COVERAGE {
2168 continue;
2169 }
2170 if let Some((cov, cell)) = match_picture_to_cell(pic, &grid.cells) {
2171 if best.is_none_or(|(b, _, _)| cov > b) {
2172 best = Some((cov, t, cell));
2173 }
2174 }
2175 }
2176 if let Some((_, t, cell)) = best {
2177 let entry = out.entry(t).or_default();
2178 match entry.iter_mut().find(|(c, _)| *c == cell) {
2179 Some((_, pics)) => pics.push(p),
2180 None => entry.push((cell, vec![p])),
2181 }
2182 }
2183 }
2184 out
2185}
2186
2187/// docling `_match_picture_to_table_cell`: among the cells covering ≥ 80 % of
2188/// the picture, prefer the one at the picture's inferred grid position (the
2189/// row / column whose median cell center is nearest the picture's center —
2190/// cell boxes can overlap across logical rows and columns), else the best
2191/// coverage. Returns `(coverage, cell index)`.
2192fn match_picture_to_cell(pic: &Region, cells: &[docling_core::TableCell]) -> Option<(f32, usize)> {
2193 let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2194 let cover = |b: &[f32; 4]| inter(pic, b[0], b[1], b[2], b[3]) / pa;
2195 let eligible: Vec<(f32, usize)> = cells
2196 .iter()
2197 .enumerate()
2198 .filter_map(|(i, c)| {
2199 let b = c.bbox.as_ref()?;
2200 let cov = cover(b);
2201 (cov >= RICH_CELL_PICTURE_COVERAGE).then_some((cov, i))
2202 })
2203 .collect();
2204 if eligible.is_empty() {
2205 return None;
2206 }
2207 let mut row_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2208 let mut col_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2209 for c in cells {
2210 let Some(b) = c.bbox.as_ref() else { continue };
2211 for r in c.start_row..c.start_row + c.row_span {
2212 row_centers.entry(r).or_default().push((b[1] + b[3]) / 2.0);
2213 }
2214 for k in c.start_col..c.start_col + c.col_span {
2215 col_centers.entry(k).or_default().push((b[0] + b[2]) / 2.0);
2216 }
2217 }
2218 let median = |v: &mut Vec<f32>| -> f32 {
2219 v.sort_by(f32::total_cmp);
2220 let n = v.len();
2221 if n % 2 == 1 {
2222 v[n / 2]
2223 } else {
2224 (v[n / 2 - 1] + v[n / 2]) / 2.0
2225 }
2226 };
2227 let (px, py) = ((pic.l + pic.r) / 2.0, (pic.t + pic.b) / 2.0);
2228 let nearest = |centers: &mut std::collections::BTreeMap<usize, Vec<f32>>, target: f32| {
2229 centers
2230 .iter_mut()
2231 .map(|(&i, v)| (i, (median(v) - target).abs()))
2232 .min_by(|a, b| a.1.total_cmp(&b.1))
2233 .map(|(i, _)| i)
2234 };
2235 let row = nearest(&mut row_centers, py);
2236 let col = nearest(&mut col_centers, px);
2237 let logical: Vec<(f32, usize)> = eligible
2238 .iter()
2239 .copied()
2240 .filter(|&(_, i)| {
2241 let c = &cells[i];
2242 row.is_some_and(|r| c.start_row <= r && r < c.start_row + c.row_span)
2243 && col.is_some_and(|k| c.start_col <= k && k < c.start_col + c.col_span)
2244 })
2245 .collect();
2246 let pool = if logical.is_empty() {
2247 &eligible
2248 } else {
2249 &logical
2250 };
2251 // Python's `max` over `(coverage, cell_index, cell)` tuples: highest
2252 // coverage, ties to the higher index.
2253 pool.iter()
2254 .copied()
2255 .max_by(|a, b| a.0.total_cmp(&b.0).then(a.1.cmp(&b.1)))
2256}
2257
2258/// The DocLang structure overlay derived from first-class cells: span
2259/// continuations (`lcel`/`ucel`/`xcel`) and per-cell header roles, so the
2260/// PDF path's DCLX carries real spans instead of a flat grid.
2261fn structure_from_cells(
2262 cells: &[docling_core::TableCell],
2263 nrows: usize,
2264 ncols: usize,
2265) -> docling_core::TableStructure {
2266 let grid = || vec![vec![false; ncols]; nrows];
2267 let mut col_cont = grid();
2268 let mut row_cont = grid();
2269 let mut row_header = grid();
2270 let mut col_header = grid();
2271 for c in cells {
2272 for r in c.start_row..(c.start_row + c.row_span).min(nrows) {
2273 for k in c.start_col..(c.start_col + c.col_span).min(ncols) {
2274 col_cont[r][k] = k > c.start_col;
2275 row_cont[r][k] = r > c.start_row;
2276 row_header[r][k] = c.row_header;
2277 col_header[r][k] = c.column_header;
2278 }
2279 }
2280 }
2281 docling_core::TableStructure {
2282 header_row: Vec::new(),
2283 col_continuation: col_cont,
2284 row_continuation: row_cont,
2285 row_header,
2286 col_header,
2287 }
2288}
2289
2290pub fn assemble_page(
2291 page: &PdfPage,
2292 regions: Vec<Region>,
2293 table_rows: &[Option<TableGrid>],
2294 enrichments: &[Option<Enrichment>],
2295) -> (Vec<Node>, Vec<(String, String)>) {
2296 let mut nodes: Vec<Node> = Vec::new();
2297 // Every page opens with an invisible page marker carrying its size in
2298 // points — what the JSON export needs to build docling's `pages` map and
2299 // denormalize the 0–511 `<location>` grid into point bboxes (#171). The
2300 // page *number* is stamped by the document-level collector (which knows
2301 // the real 1-based index, `--pages` windows included); every serializer
2302 // except JSON skips the marker, so Markdown/DocLang stay byte-identical.
2303 nodes.push(Node::PageInfo {
2304 page_no: 0,
2305 width: page.width,
2306 height: page.height,
2307 });
2308 // Recover this page's hyperlinks (anchor-precise pairs for strict
2309 // Markdown; whole-item docling-parity links are baked below and their
2310 // pairs dropped from this list so strict output doesn't double-wrap).
2311 let mut links = resolve_link_anchors(page);
2312 // Pair each region with its precomputed TableFormer grid and enrichment
2313 // (indexed by original order) and order by reading order together, so they
2314 // stay aligned.
2315 // docling's assembly order of the regions — what its reading-order
2316 // predictor knows as `cid` (#424) — before they are shuffled.
2317 let cids = cluster_cids(®ions, &page.cells);
2318 type RegionItem = (Region, Option<TableGrid>, Option<Enrichment>);
2319 let mut items: Vec<RegionItem> = regions
2320 .into_iter()
2321 .enumerate()
2322 .map(|(i, r)| {
2323 (
2324 r,
2325 table_rows.get(i).cloned().flatten(),
2326 enrichments.get(i).cloned().flatten(),
2327 )
2328 })
2329 .collect();
2330 order_with_containers(&mut items, &cids, page.width, page.height, |it| &it.0);
2331 // Float a margin page number to the front of reading order (docling parity:
2332 // right_to_left_02's bottom `11` is its first item). Stable, so everything
2333 // else keeps its order; no-op on pages without such a region.
2334 let page_h = page.height;
2335 items.sort_by_key(|(r, _, _)| !is_page_number(r, &page.cells, page_h));
2336 let table_rows: Vec<Option<TableGrid>> = items.iter().map(|(_, t, _)| t.clone()).collect();
2337 let enrichments: Vec<Option<Enrichment>> = items.iter().map(|(_, _, e)| e.clone()).collect();
2338 let regions: Vec<Region> = items.into_iter().map(|(r, _, _)| r).collect();
2339 // docling emits a figure's caption *before* the image marker. Pair each
2340 // picture with the caption region nearest below it and consume that caption,
2341 // so it isn't also emitted in its own (lower) reading-order position.
2342 let caption_for = pair_captions(®ions);
2343 let code_caption_for = pair_code_captions(®ions);
2344 let mut consumed = vec![false; regions.len()];
2345 for ci in caption_for.iter().flatten() {
2346 consumed[*ci] = true;
2347 }
2348 for ci in code_caption_for.iter().flatten() {
2349 consumed[*ci] = true;
2350 }
2351 // Table captions (#265) claim from what the picture/code pairings left.
2352 let mut caption_taken = consumed.clone();
2353 let table_caption_for = pair_table_captions(®ions, &mut caption_taken);
2354 for ci in table_caption_for.iter().flatten() {
2355 consumed[*ci] = true;
2356 }
2357 // Pictures inside a table become rich-cell content (docling#3906, 2.118.1):
2358 // the picture is nested in the cell it covers and not emitted standalone.
2359 let rich_cell_pictures = match_table_pictures(®ions, &table_rows, &caption_for);
2360 for (_, pics) in rich_cell_pictures.values().flatten() {
2361 for &p in pics {
2362 consumed[p] = true;
2363 }
2364 }
2365 // A code block's language label (`XML`, `C#`, …) is chrome, not content — the
2366 // detector emits it as its own region above the code; consume it.
2367 for (i, is_label) in code_language_labels(®ions, &page.cells)
2368 .into_iter()
2369 .enumerate()
2370 {
2371 if is_label {
2372 consumed[i] = true;
2373 }
2374 }
2375
2376 // docling `ReadingOrderPredictor.predict_merges`: join a text fragment with a
2377 // following text fragment strictly to its right (an author column that wraps
2378 // into the next, a paragraph continuing in the next column) into one block —
2379 // the intra-page half of docling's reading-order merges (cross-page/vertical
2380 // continuations stay with [`merge_continuations`]). Already-consumed regions
2381 // (paired captions, code labels) are excluded.
2382 // Exclusive docling cell assignment: computed once for the ordered region
2383 // list and reused for every serialization below, so a cell can never render
2384 // in two regions.
2385 let region_texts: Vec<String> = region_texts_exclusive(®ions, &page.cells);
2386 let is_text: Vec<bool> = regions
2387 .iter()
2388 .enumerate()
2389 .map(|(i, r)| r.label == "text" && !consumed[i])
2390 .collect();
2391 let is_skip: Vec<bool> = regions
2392 .iter()
2393 .enumerate()
2394 .map(|(i, r)| {
2395 consumed[i]
2396 || matches!(
2397 r.label,
2398 "page_header" | "page_footer" | "table" | "picture" | "caption" | "footnote"
2399 )
2400 })
2401 .collect();
2402 let boxes: Vec<(f32, f32, f32, f32)> = regions.iter().map(|r| (r.l, r.t, r.r, r.b)).collect();
2403 if docling_core::env::flag("DOCLING_RS_DEBUG_MERGES") {
2404 for (i, r) in regions.iter().enumerate() {
2405 eprintln!(
2406 "MRG {i:2} {} text={} skip={} [{:.0},{:.0},{:.0},{:.0}] {:?}",
2407 r.label,
2408 is_text[i],
2409 is_skip[i],
2410 r.l,
2411 r.t,
2412 r.r,
2413 r.b,
2414 region_texts[i].chars().take(40).collect::<String>()
2415 );
2416 }
2417 }
2418 let mut merge_suffix: Vec<String> = vec![String::new(); regions.len()];
2419 for (head, children) in
2420 crate::reading_order::predict_merges(&boxes, ®ion_texts, &is_text, &is_skip)
2421 .into_iter()
2422 .enumerate()
2423 {
2424 for c in children {
2425 let t = region_texts[c].trim();
2426 if !t.is_empty() {
2427 merge_suffix[head].push(' ');
2428 merge_suffix[head].push_str(t);
2429 }
2430 consumed[c] = true;
2431 }
2432 }
2433
2434 for (i, region) in regions.iter().enumerate() {
2435 if consumed[i] {
2436 continue;
2437 }
2438 // Page headers/footers: docling emits them as furniture blocks
2439 // (`<page_header>`/`<page_footer>` with a layer + location + text) at
2440 // their reading-order position, not as body — emit them, don't skip.
2441 if matches!(region.label, "page_header" | "page_footer") {
2442 let text = region_texts[i].clone();
2443 if !text.is_empty() {
2444 nodes.push(Node::PageFurniture {
2445 footer: region.label == "page_footer",
2446 location: norm_loc(region, page.width, page_h),
2447 text: md_escape(&text),
2448 });
2449 }
2450 continue;
2451 }
2452 if is_skipped(region.label) {
2453 continue;
2454 }
2455 // Layout provenance for this region, normalized to docling's 0–511 grid.
2456 let loc = norm_loc(region, page.width, page_h);
2457 if region.label == "picture" {
2458 // The figure pixels are cropped from the page render for image export.
2459 // Captions are prose: markdown-escaped like a paragraph (the JSON
2460 // export unescapes back to the raw text, matching docling).
2461 let caption = caption_for[i]
2462 .map(|ci| md_escape(®ion_texts[ci]))
2463 .filter(|t| !t.is_empty());
2464 let classification = match &enrichments[i] {
2465 Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
2466 _ => None,
2467 };
2468 // Without the page render (text-layer-only build) a picture keeps
2469 // its caption/classification but carries no cropped pixels.
2470 #[cfg(feature = "ocr-prep")]
2471 let image = crate::timing::timed("crop_region", || crop_region(page, region));
2472 #[cfg(not(feature = "ocr-prep"))]
2473 let image: Option<PictureImage> = None;
2474 nodes.push(located(
2475 loc,
2476 Node::Picture {
2477 caption,
2478 caption_href: None,
2479 image,
2480 classification,
2481 // docling's layout pipeline parents a figure's caption to
2482 // the picture itself (#390) — the one backend that does.
2483 caption_parent: CaptionParent::Item,
2484 },
2485 ));
2486 continue;
2487 }
2488 let mut text = region_texts[i].clone();
2489 text.push_str(&merge_suffix[i]);
2490 if text.is_empty() {
2491 continue;
2492 }
2493 match region.label {
2494 // docling assembles checkboxes as TEXT_ELEM items (the region's
2495 // cells are the option label, e.g. right_to_left_03's بلی/خير)
2496 // and its Markdown serializer renders them as task-list lines
2497 // (`- [x] …`) — mirrored by [`Node::CheckboxItem`].
2498 "checkbox_selected" | "checkbox_unselected" => nodes.push(Node::CheckboxItem {
2499 checked: region.label == "checkbox_selected",
2500 text: md_escape(&text),
2501 }),
2502 // docling renders both the document title and section headers as
2503 // `##` (it never emits a top-level `#` for PDFs), so match that.
2504 "title" | "section_header" => nodes.push(located(
2505 loc,
2506 Node::Heading {
2507 level: 2,
2508 text: md_escape(&text),
2509 },
2510 )),
2511 // docling drops the rendered bullet glyph; the Markdown serializer
2512 // adds its own `- ` marker. An item whose text opens with an `N.`
2513 // enumeration marker is an ordered item (rendered `N. text`).
2514 // A leading dash stays: it is an ordinary text glyph that
2515 // docling-parse keeps, and docling's items carry it into the
2516 // Markdown (2305's OTSL list renders `- -"C" cell …`) — only the
2517 // symbol-font bullets docling-parse filters out are stripped.
2518 "list_item" => {
2519 let stripped = text
2520 .trim_start_matches(['•', '◦', '▪', '·', '*'])
2521 .trim_start()
2522 .to_string();
2523 if let Some((number, rest)) = parse_ordered_marker(&stripped) {
2524 nodes.push(Node::ListItem {
2525 ordered: true,
2526 number,
2527 first_in_list: false,
2528 text: md_escape(&rest),
2529 level: 0,
2530 marker: None,
2531 location: Some(loc),
2532 dclx: None,
2533 href: None,
2534 layer: None,
2535 });
2536 } else {
2537 nodes.push(Node::ListItem {
2538 ordered: false,
2539 number: 0,
2540 first_in_list: false,
2541 text: md_escape(&stripped),
2542 level: 0,
2543 // docling keeps the bullet as the DocLang list marker
2544 // (`<ldiv><marker>·</marker></ldiv>`); Markdown ignores it.
2545 marker: Some("·".into()),
2546 location: Some(loc),
2547 dclx: None,
2548 href: None,
2549 layer: None,
2550 });
2551 }
2552 }
2553 // TableFormer structure (cells + spans, text matched from word cells)
2554 // when available; otherwise geometric grid reconstruction; finally a
2555 // single cell.
2556 "table" | "document_index" => {
2557 // TableFormer grids carry first-class cells (#240: text +
2558 // page-point bbox + span rectangle + OTSL header roles) into
2559 // the public model, and the DocLang structure overlay derives
2560 // from them so DCLX emits real span/header tokens. The
2561 // geometric fallback has no per-cell records.
2562 let (mut rows, cells, structure) = match table_rows[i].clone() {
2563 Some(grid) => {
2564 let nrows = grid.rows.len();
2565 let ncols = grid.rows.first().map_or(0, Vec::len);
2566 let structure = structure_from_cells(&grid.cells, nrows, ncols);
2567 (grid.rows, Some(grid.cells), Some(structure))
2568 }
2569 None => {
2570 let rows = reconstruct_table(region, &page.cells);
2571 let rows = if rows.iter().any(|r| r.len() > 1) {
2572 rows
2573 } else {
2574 vec![vec![text.clone()]]
2575 };
2576 (rows, None, None)
2577 }
2578 };
2579 // The paired caption (#265) rides on the table — docling's
2580 // TableItem.captions ref; Markdown prints it above the grid,
2581 // the JSON export emits the $ref, DocLang the <caption>.
2582 let caption = table_caption_for[i]
2583 .map(|ci| md_escape(®ion_texts[ci]))
2584 .filter(|t| !t.is_empty());
2585 // Rich cells (docling#3906): the covering cell's blocks are its
2586 // text followed by the nested picture(s). docling's Markdown
2587 // renders a `RichTableCell` through the serializer — the
2588 // group's children joined by blank lines, newlines flattened
2589 // to spaces — so the flat `rows` text becomes
2590 // `text <!-- image -->`; the first-class `cells` (the JSON
2591 // `table_cells` / `grid`) keep the plain text, as upstream.
2592 let mut cell_blocks: Option<Vec<Vec<Vec<Node>>>> = None;
2593 if let (Some(by_cell), Some(fc)) = (rich_cell_pictures.get(&i), cells.as_ref()) {
2594 let nrows = rows.len();
2595 let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
2596 let mut blocks = vec![vec![Vec::<Node>::new(); ncols]; nrows];
2597 for (cell_idx, pics) in by_cell {
2598 let cell = &fc[*cell_idx];
2599 let (r, c) = (cell.start_row, cell.start_col);
2600 if r >= nrows || c >= ncols {
2601 continue;
2602 }
2603 let mut parts: Vec<String> = Vec::new();
2604 let mut cell_nodes: Vec<Node> = Vec::new();
2605 if !cell.text.trim().is_empty() {
2606 parts.push(cell.text.clone());
2607 cell_nodes.push(Node::Paragraph {
2608 text: cell.text.clone(),
2609 });
2610 }
2611 for &p in pics {
2612 parts.push("<!-- image -->".to_string());
2613 let classification = match &enrichments[p] {
2614 Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
2615 _ => None,
2616 };
2617 #[cfg(feature = "ocr-prep")]
2618 let image = crop_region(page, ®ions[p]);
2619 #[cfg(not(feature = "ocr-prep"))]
2620 let image: Option<PictureImage> = None;
2621 cell_nodes.push(located(
2622 norm_loc(®ions[p], page.width, page_h),
2623 Node::Picture {
2624 caption: None,
2625 caption_href: None,
2626 image,
2627 classification,
2628 caption_parent: Default::default(),
2629 },
2630 ));
2631 }
2632 let rendered = parts.join(" ");
2633 for row in rows.iter_mut().skip(r).take(cell.row_span) {
2634 for slot in row.iter_mut().skip(c).take(cell.col_span) {
2635 *slot = rendered.clone();
2636 }
2637 }
2638 blocks[r][c] = cell_nodes;
2639 }
2640 cell_blocks = Some(blocks);
2641 }
2642 nodes.push(located(
2643 loc,
2644 Node::Table(Table {
2645 rows,
2646 location: None,
2647 structure,
2648 cell_blocks,
2649 cells,
2650 caption,
2651 // As for pictures: the caption is the table's child.
2652 caption_parent: CaptionParent::Item,
2653 }),
2654 ));
2655 }
2656 // With formula enrichment the CodeFormula model decodes the region
2657 // to LaTeX; otherwise docling emits a placeholder comment rather
2658 // than the (garbled) raw glyph text.
2659 "formula" => match &enrichments[i] {
2660 Some(Enrichment::Formula { latex }) => nodes.push(Node::Formula {
2661 latex: latex.clone(),
2662 orig: text.clone(),
2663 location: Some(loc),
2664 }),
2665 _ => nodes.push(Node::Paragraph {
2666 text: "<!-- formula-not-decoded -->".into(),
2667 }),
2668 },
2669 // Code blocks: use the space-glyph-only grouping (monospace keeps its
2670 // source spacing) and emit a fenced block, preserving the line breaks
2671 // and indentation of the source (unlike prose, which reflows). pdfium
2672 // still inserts spaces around tight punctuation (`console .log`,
2673 // `add (3 , 5)`); tighten them to match docling-parse's source spacing.
2674 "code" => {
2675 // `code_region_text` preserves line breaks/indentation and tightens
2676 // each line itself; the fallback prose `text` is tightened here.
2677 let code = code_region_text(region, &page.code_cells);
2678 let code = if code.is_empty() {
2679 tighten_code_punct(&text)
2680 } else {
2681 code
2682 };
2683 // With code enrichment the CodeFormula model rewrites the block
2684 // (and names its language); `orig` keeps the raw extraction in
2685 // docling's shape — its parser has no line-preserving code
2686 // path, so its `orig` is the same code with the lines joined
2687 // by single spaces (indentation collapsed).
2688 // docling's parser has no line-preserving code path — its code
2689 // items carry the lines joined by single spaces. That flat
2690 // form is what every byte-conformance surface serializes
2691 // (legacy Markdown, JSON, DocLang); the line-preserving
2692 // extraction rides in `pretty` for strict Markdown only.
2693 let flat = code
2694 .lines()
2695 .map(str::trim)
2696 .filter(|l| !l.is_empty())
2697 .collect::<Vec<_>>()
2698 .join(" ");
2699 let node = match &enrichments[i] {
2700 Some(Enrichment::Code {
2701 language,
2702 text: enriched,
2703 }) => Node::Code {
2704 language: language.clone(),
2705 text: enriched.clone(),
2706 orig: Some(flat),
2707 pretty: None,
2708 },
2709 _ => Node::Code {
2710 language: None,
2711 text: flat,
2712 orig: None,
2713 pretty: Some(code),
2714 },
2715 };
2716 nodes.push(located(loc, node));
2717 // docling emits the `Listing N:` caption after the code block.
2718 if let Some(ci) = code_caption_for[i] {
2719 let cap = md_escape(®ion_texts[ci]);
2720 if !cap.is_empty() {
2721 nodes.push(Node::Paragraph { text: cap });
2722 }
2723 }
2724 }
2725 // text, caption, footnote → paragraph
2726 _ => {
2727 // docling parity (`PageAssembleModel._match_hyperlink`): when
2728 // link annotations cover ≥ half of the region's box, the
2729 // hyperlink attaches to the item and the legacy Markdown
2730 // serializer wraps its full text — 2206.01062's footnote URLs
2731 // render as `[1 https://…](https://…)`. Sparse in-paragraph
2732 // citation links stay below the 0.5 coverage threshold and
2733 // remain plain text, exactly like docling.
2734 //
2735 // Scope: **footnote regions only.** Upstream's page_assemble
2736 // matches every TEXT_ELEM label, but published docling
2737 // observably carries the hyperlink into the document only for
2738 // footnote items — in both committed groundtruth generations
2739 // (docling-JSON and Markdown, independent runs) the fully
2740 // covered plain-text DOI line of 2206.01062 page 1 has
2741 // `hyperlink: None` while the equally covered footnotes carry
2742 // theirs. The corpus is the conformance reference, so match
2743 // the observed behavior; widen the label set if a future
2744 // groundtruth refresh starts linking plain text too.
2745 let escaped = md_escape(&text);
2746 let hyperlink = (region.label == "footnote")
2747 .then(|| region_hyperlink(region, &page.links))
2748 .flatten();
2749 let text = match hyperlink {
2750 Some(uri) => {
2751 // The strict-mode anchor pairs this item covers are
2752 // superseded by the baked whole-item link.
2753 links.retain(|(anchor, href)| {
2754 !(href == &uri && region_texts[i].contains(anchor.as_str()))
2755 });
2756 format!("[{escaped}]({uri})")
2757 }
2758 None => escaped,
2759 };
2760 nodes.push(located(loc, Node::Paragraph { text }))
2761 }
2762 }
2763 }
2764 // A `/Rotate`-normalized scanned page (see `pdfium_backend`) was assembled
2765 // in upright space; rotate the finished geometry back so locations and the
2766 // page size are display-space, like docling and every viewer report them.
2767 if page.rotation != 0 {
2768 rotate_nodes_to_display(&mut nodes, page.rotation);
2769 }
2770 (nodes, links)
2771}
2772
2773/// Rotate one 0–511 location bbox 90° clockwise on the grid (top-left origin):
2774/// `(x, y) → (511 - y, x)`.
2775fn rot_loc_cw(l: [u16; 4]) -> [u16; 4] {
2776 [511 - l[3], l[0], 511 - l[1], l[2]]
2777}
2778
2779/// Map upright-space geometry back to display space for a page whose `/Rotate`
2780/// was normalized away before inference: every `<location>` rotates `rot`°
2781/// clockwise on the 0–511 grid (the grid is per-axis normalized, so no page
2782/// dims are needed), and the `PageInfo` size returns to the display box. Node
2783/// text and order are untouched — reading order was decided upright, which is
2784/// the whole point.
2785fn rotate_nodes_to_display(nodes: &mut [Node], rot: u16) {
2786 let quarter_turns = (rot / 90) as usize;
2787 let rot_loc = |l: &mut [u16; 4]| {
2788 for _ in 0..quarter_turns {
2789 *l = rot_loc_cw(*l);
2790 }
2791 };
2792 fn walk(node: &mut Node, rot_loc: &impl Fn(&mut [u16; 4]), swap_dims: bool) {
2793 match node {
2794 Node::PageInfo { width, height, .. } => {
2795 if swap_dims {
2796 std::mem::swap(width, height);
2797 }
2798 }
2799 Node::Located { location, inner } => {
2800 rot_loc(location);
2801 walk(inner, rot_loc, swap_dims);
2802 }
2803 Node::Furniture { inner, .. } => walk(inner, rot_loc, swap_dims),
2804 Node::Group { children, .. } => {
2805 for c in children {
2806 walk(c, rot_loc, swap_dims);
2807 }
2808 }
2809 Node::ListItem { location, .. }
2810 | Node::Formula { location, .. }
2811 | Node::Chart { location, .. } => {
2812 if let Some(l) = location {
2813 rot_loc(l);
2814 }
2815 }
2816 Node::PageFurniture { location, .. } => rot_loc(location),
2817 Node::Table(t) => {
2818 if let Some(l) = &mut t.location {
2819 rot_loc(l);
2820 }
2821 }
2822 _ => {}
2823 }
2824 }
2825 let swap_dims = quarter_turns % 2 == 1;
2826 for node in nodes {
2827 walk(node, &rot_loc, swap_dims);
2828 }
2829}
2830
2831/// Merge paragraph fragments split across a column or page break. docling joins a
2832/// paragraph whose previous fragment ends mid-sentence (a letter, not sentence
2833/// punctuation) with a lowercase continuation: `…definition of` + `lists in…` →
2834/// `…definition of lists in…`. The fragments are consecutive paragraphs, or
2835/// separated only by figure(s) the text wraps around: a column whose body flows
2836/// past a figure resumes below it (`…The wing type that is` ⟶[figure]⟶ `the most
2837/// common…`), and docling emits the whole paragraph before the figure. A heading,
2838/// table, or list between them ends the paragraph (no merge).
2839/// A paragraph that is really a figure/table caption (`Fig. 1. …`, `Table 2 …`).
2840/// Used to skip an unpaired caption when stitching a paragraph that wraps around
2841/// a figure.
2842fn looks_like_caption(text: &str) -> bool {
2843 let head: String = text.trim_start().chars().take(14).collect();
2844 (head.starts_with("Fig") || head.starts_with("Table"))
2845 && head.contains(|c: char| c.is_ascii_digit())
2846}
2847
2848/// A paragraph fragment is "open" — i.e. it might continue into the next
2849/// paragraph — when it ends mid-word (a letter) or with a wrap hyphen/dash.
2850/// docling joins `vocab-` + `ulary` → `vocab- ulary`.
2851fn paragraph_is_open(text: &str) -> bool {
2852 // docling's merge head test (`.+([a-z,\-\u00AD])\s*`): at least two chars,
2853 // ending in an ASCII lowercase letter, a comma, a hyphen, or a soft
2854 // hyphen. The comma matters: 2206's "…In phase four," resumes across the
2855 // page break. Uppercase/non-Latin endings do not merge, exactly as
2856 // upstream (the dash family is already `-` here — clean_text normalized).
2857 let t = text.trim_end();
2858 t.chars().count() >= 2
2859 && t.chars()
2860 .next_back()
2861 .is_some_and(|c| matches!(c, 'a'..='z' | ',' | '-' | '\u{ad}'))
2862}
2863
2864/// The paragraph text inside a node, looking through a [`Node::Located`]
2865/// provenance wrapper (PDF body paragraphs are wrapped since they carry a
2866/// `<location>`). Returns `None` for non-paragraph nodes.
2867fn as_paragraph(n: &Node) -> Option<&str> {
2868 match n {
2869 Node::Paragraph { text } => Some(text),
2870 Node::Located { inner, .. } => match inner.as_ref() {
2871 Node::Paragraph { text } => Some(text),
2872 _ => None,
2873 },
2874 _ => None,
2875 }
2876}
2877
2878/// Whether a node is a picture, looking through a [`Node::Located`] wrapper.
2879fn is_picture_node(n: &Node) -> bool {
2880 match n {
2881 Node::Picture { .. } => true,
2882 Node::Located { inner, .. } => matches!(inner.as_ref(), Node::Picture { .. }),
2883 _ => false,
2884 }
2885}
2886
2887/// A node a forward paragraph merge looks straight past: a figure or *table*
2888/// the text wraps around, or a page header/footer that falls between the two
2889/// fragments of a paragraph continuing across a page break (docling's merge
2890/// skip-labels: page_header, page_footer, table, picture, caption, footnote —
2891/// 2206's "…In phase four," resumes after a full caption+table+figure block).
2892fn is_merge_trailer(n: &Node) -> bool {
2893 is_picture_node(n)
2894 || matches!(
2895 n,
2896 Node::PageFurniture { .. } | Node::PageInfo { .. } | Node::Table(_)
2897 )
2898 || matches!(n, Node::Located { inner, .. } if matches!(inner.as_ref(), Node::Table(_)))
2899 || as_paragraph(n).is_some_and(looks_like_caption)
2900}
2901
2902/// Rebuild node `i` as a paragraph with `text`, preserving its `<location>`
2903/// wrapper (and thus provenance) if it had one.
2904fn reparagraph(node: &Node, text: String) -> Node {
2905 match node {
2906 Node::Located { location, .. } => located(*location, Node::Paragraph { text }),
2907 _ => Node::Paragraph { text },
2908 }
2909}
2910
2911pub(crate) fn merge_continuations(nodes: &mut Vec<Node>) {
2912 let mut i = 0;
2913 while i + 1 < nodes.len() {
2914 let Some(a) = as_paragraph(&nodes[i]) else {
2915 i += 1;
2916 continue;
2917 };
2918 // A figure/table caption is a self-contained unit; body text resuming
2919 // after a figure is the continuation case, not the caption itself. Never
2920 // stitch *from* a caption — otherwise a caption that ends in a lone glyph
2921 // (`Fig. 5. … PubTabNet. μ`) would swallow a following stray figure label
2922 // (a standalone `μ`) into `… μ μ`.
2923 if looks_like_caption(a) {
2924 i += 1;
2925 continue;
2926 }
2927 if !paragraph_is_open(a) {
2928 i += 1;
2929 continue;
2930 }
2931 // The continuation is the next paragraph, looking past any figures the
2932 // text wraps around — and a figure/table caption that was emitted as its
2933 // own paragraph (an above-the-figure caption that didn't pair), since the
2934 // body text resumes after the whole figure+caption block.
2935 let mut j = i + 1;
2936 while nodes.get(j).is_some_and(is_merge_trailer) {
2937 j += 1;
2938 }
2939 // docling's continuation regex allows either case, but its merge runs
2940 // over the pre-assembly element stream; at node level an uppercase
2941 // start is overwhelmingly a new sentence/heading fragment (allowing it
2942 // swallowed 2305's formula blocks and redp's chapter openers), so the
2943 // continuation stays lowercase-start here.
2944 let cont = nodes.get(j).and_then(as_paragraph).is_some_and(|b| {
2945 b.trim_start()
2946 .chars()
2947 .next()
2948 .is_some_and(char::is_lowercase)
2949 });
2950 if cont {
2951 let a = as_paragraph(&nodes[i]).unwrap().trim_end().to_string();
2952 let b = as_paragraph(&nodes[j]).unwrap().trim_start().to_string();
2953 // A soft hyphen -- or a hard hyphen followed by a lowercase
2954 // continuation (guaranteed lowercase by the `cont` gate above) --
2955 // is a word split across the break: strip it and join without a
2956 // space, docling#3888 ("vocab-" + "ulary" -> "vocabulary");
2957 // docling's older serializer kept the artifact ("vocab- ulary").
2958 // Everything else joins with the space, as before.
2959 let merged = match a.strip_suffix('\u{ad}').or_else(|| a.strip_suffix('-')) {
2960 Some(stem) => format!("{stem}{b}"),
2961 None => format!("{a} {b}"),
2962 };
2963 // Keep node i's provenance wrapper; docling's merged paragraph keeps
2964 // the first fragment's geometry as its primary location.
2965 nodes[i] = reparagraph(&nodes[i], merged);
2966 nodes.remove(j);
2967 // Re-check i: the merged paragraph may continue further.
2968 } else {
2969 i += 1;
2970 }
2971 }
2972}
2973
2974/// How many leading nodes of `nodes` are safe to flush now — i.e. cannot be
2975/// rewritten by a future [`merge_continuations`] once more pages are appended.
2976///
2977/// A forward merge can only start from an "open" paragraph (ends mid-word) and
2978/// only reaches across trailing pictures and figure/table captions. So we scan
2979/// from the end past those skippable trailers: if the first non-skippable node is
2980/// an open paragraph, it (and the trailers after it) must be held; anything else —
2981/// a closed paragraph, a heading, a table, a list — blocks any forward merge, so
2982/// the whole buffer is safe to flush.
2983fn hold_start(nodes: &[Node]) -> usize {
2984 for k in (0..nodes.len()).rev() {
2985 // Skippable trailers (figures, page furniture, captions): a forward merge
2986 // looks straight past them.
2987 if is_merge_trailer(&nodes[k]) {
2988 continue;
2989 }
2990 match as_paragraph(&nodes[k]) {
2991 // An open body paragraph might still pull a continuation off the next
2992 // page — hold from here to the end.
2993 Some(text) if paragraph_is_open(text) => return k,
2994 // A closed paragraph, heading, table, list, etc. ends the paragraph:
2995 // nothing after it can merge backwards across it. Flush everything.
2996 _ => return nodes.len(),
2997 }
2998 }
2999 // Only skippable trailers (or empty) and no open paragraph to anchor a merge.
3000 nodes.len()
3001}
3002
3003/// Streaming counterpart of [`merge_continuations`]: feed per-page node batches in
3004/// document order and get back the prefix that is final (its cross-page merges are
3005/// resolved and no future page can change it), holding back only the small tail
3006/// that might still merge into the next page. Concatenating every flushed batch
3007/// (then [`finish`](Self::finish)) yields exactly the same nodes as running
3008/// [`merge_continuations`] once over the whole document.
3009pub(crate) struct StreamAssembler {
3010 pending: Vec<Node>,
3011}
3012
3013impl StreamAssembler {
3014 pub(crate) fn new() -> Self {
3015 Self {
3016 pending: Vec::new(),
3017 }
3018 }
3019
3020 /// Append one page's nodes, resolve merges within the buffer, and return the
3021 /// now-final prefix to emit (possibly empty).
3022 pub(crate) fn push(&mut self, mut nodes: Vec<Node>) -> Vec<Node> {
3023 self.pending.append(&mut nodes);
3024 merge_continuations(&mut self.pending);
3025 let cut = hold_start(&self.pending);
3026 let tail = self.pending.split_off(cut);
3027 std::mem::replace(&mut self.pending, tail)
3028 }
3029
3030 /// Flush whatever is left after the last page (the held tail is final once no
3031 /// more pages can follow).
3032 pub(crate) fn finish(self) -> Vec<Node> {
3033 self.pending
3034 }
3035}
3036
3037#[cfg(test)]
3038mod tests {
3039 use super::{cells_text, clean_text};
3040 use super::{code_region_text, merge_continuations, resolve_link_anchors, StreamAssembler};
3041 use crate::layout::Region;
3042 use crate::pdfium_backend::{LinkAnnot, PdfPage, TextCell};
3043 use docling_core::Node;
3044
3045 /// The int8-layout guard's coverage metric: cells under detections count,
3046 /// cells outside don't, whitespace cells are ignored, and a cell-less page
3047 /// reads as fully covered (nothing to rescue).
3048 #[test]
3049 fn layout_cell_coverage_counts_claimed_text_cells() {
3050 let cell = |text: &str, l: f32, t: f32| TextCell {
3051 text: text.into(),
3052 l,
3053 t,
3054 r: l + 40.0,
3055 b: t + 10.0,
3056 };
3057 let region = Region {
3058 label: "text",
3059 score: 0.9,
3060 l: 0.0,
3061 t: 0.0,
3062 r: 100.0,
3063 b: 50.0,
3064 };
3065 let cells = vec![
3066 cell("inside", 10.0, 10.0),
3067 cell("also inside", 10.0, 30.0),
3068 cell("outside", 10.0, 200.0),
3069 cell(" ", 10.0, 210.0), // whitespace: not counted at all
3070 ];
3071 let cov = super::layout_cell_coverage(std::slice::from_ref(®ion), &cells);
3072 assert!((cov - 2.0 / 3.0).abs() < 1e-6, "got {cov}");
3073 assert_eq!(super::layout_cell_coverage(&[], &[]), 1.0);
3074 assert_eq!(super::layout_cell_coverage(&[], &cells), 0.0);
3075 }
3076
3077 /// #165: a picture no longer claims cells at 0.2 intersection-over-self.
3078 /// A line straddling the figure border (≤80 % contained) becomes an orphan
3079 /// region and survives the contained-regulars drop — before the fix its
3080 /// cells were silently erased. A line fully inside the picture is still
3081 /// re-dropped, matching docling's Markdown (a picture's children never
3082 /// reach its serializer's output).
3083 #[test]
3084 fn border_straddling_lines_survive_picture_interior_is_still_dropped() {
3085 let pic = Region {
3086 label: "picture",
3087 score: 0.9,
3088 l: 0.0,
3089 t: 0.0,
3090 r: 100.0,
3091 b: 100.0,
3092 };
3093 // ~35 % of this cell overlaps the picture (l=90..120 of 0..100): above
3094 // the old 0.2 claim (was swallowed), below full containment (survives).
3095 let straddler = TextCell {
3096 text: "axis label".into(),
3097 l: 90.0,
3098 t: 40.0,
3099 r: 120.0,
3100 b: 48.0,
3101 };
3102 let interior = TextCell {
3103 text: "in-figure callout".into(),
3104 l: 10.0,
3105 t: 10.0,
3106 r: 60.0,
3107 b: 18.0,
3108 };
3109 let mut regions = vec![pic];
3110 super::add_orphan_regions(&mut regions, &[straddler, interior]);
3111 assert_eq!(
3112 regions.iter().filter(|r| r.label == "text").count(),
3113 2,
3114 "both unclaimed lines become orphans"
3115 );
3116 super::drop_contained_regulars(&mut regions);
3117 let texts: Vec<(f32, f32)> = regions
3118 .iter()
3119 .filter(|r| r.label == "text")
3120 .map(|r| (r.l, r.r))
3121 .collect();
3122 assert_eq!(
3123 texts,
3124 [(90.0, 120.0)],
3125 "the straddler is emitted, the fully-contained callout is not"
3126 );
3127 }
3128
3129 /// docling#3906's concern, pinned on our side: a picture detected fully
3130 /// inside a table region must survive the containment drop (upstream now
3131 /// attaches it to the table's cell; we keep it as a body sibling — either
3132 /// way it must not vanish). The text region inside the same table is the
3133 /// control: regulars are the ones the drop swallows.
3134 #[test]
3135 fn picture_inside_a_table_region_survives_the_containment_drop() {
3136 let mut regions = vec![
3137 region("table", 0.9, 0.0, 0.0, 200.0, 200.0),
3138 region("picture", 0.9, 20.0, 20.0, 120.0, 120.0),
3139 region("text", 0.9, 20.0, 140.0, 180.0, 180.0),
3140 ];
3141 super::drop_contained_regulars(&mut regions);
3142 let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
3143 assert_eq!(
3144 labels,
3145 ["table", "picture"],
3146 "the in-table picture stays; the in-table regular is the special's child"
3147 );
3148 }
3149
3150 /// Table–caption pairing (#265) is reading-order adjacency, docling's
3151 /// `_find_to_captions`: a caption binds the table directly next to it in
3152 /// the region sequence — above-caption and below-caption both work, and
3153 /// geometry is irrelevant (a same-page caption in the other column of a
3154 /// two-column layout is *not* adjacent, however close its box is). A
3155 /// caption with media on both sides, or separated from the table by a
3156 /// text paragraph, stays unattached.
3157 #[test]
3158 fn table_captions_pair_by_reading_order_adjacency() {
3159 // caption → table (above-caption), then table → caption (below-caption),
3160 // then a caption fenced off by a paragraph, then one between two tables.
3161 let regions = vec![
3162 region("text", 0.9, 0.0, 0.0, 100.0, 10.0), // 0 body text
3163 region("caption", 0.9, 0.0, 12.0, 60.0, 20.0), // 1 above-caption
3164 region("table", 0.9, 20.0, 22.0, 90.0, 60.0), // 2 ← pairs with 1
3165 region("table", 0.9, 0.0, 70.0, 100.0, 110.0), // 3 ← pairs with 4
3166 region("caption", 0.9, 0.0, 112.0, 60.0, 120.0), // 4 below-caption
3167 region("text", 0.9, 0.0, 130.0, 100.0, 140.0), // 5 body text
3168 region("caption", 0.9, 0.0, 142.0, 60.0, 150.0), // 6 fenced by 5/7
3169 region("text", 0.9, 0.0, 152.0, 100.0, 162.0), // 7 body text
3170 region("table", 0.9, 0.0, 170.0, 100.0, 200.0), // 8 unpaired
3171 region("caption", 0.9, 0.0, 202.0, 60.0, 210.0), // 9 ambiguous
3172 region("table", 0.9, 0.0, 212.0, 100.0, 240.0), // 10 unpaired
3173 ];
3174 let mut taken = vec![false; regions.len()];
3175 let pairs = super::pair_table_captions(®ions, &mut taken);
3176 assert_eq!(pairs[2], Some(1), "caption directly above its table pairs");
3177 assert_eq!(pairs[3], Some(4), "caption directly below its table pairs");
3178 assert_eq!(
3179 pairs[8], None,
3180 "a text paragraph between caption and table breaks the bond"
3181 );
3182 assert_eq!(
3183 pairs[10], None,
3184 "a caption between two tables is ambiguous and stays loose"
3185 );
3186 assert!(taken[1] && taken[4] && !taken[6] && !taken[9]);
3187 }
3188
3189 /// A colored terms-and-conditions panel detected as `picture` demotes into
3190 /// per-paragraph `text` regions (the blank line between C.7 and C.8 splits
3191 /// them); a chart whose only text is a few narrow axis labels keeps its
3192 /// crop untouched.
3193 #[test]
3194 fn text_panels_demote_to_paragraphs_but_charts_keep_their_crop() {
3195 let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3196 text: text.to_string(),
3197 l,
3198 t,
3199 r,
3200 b,
3201 };
3202 let panel = Region {
3203 label: "picture",
3204 score: 0.9,
3205 l: 0.0,
3206 t: 0.0,
3207 r: 100.0,
3208 b: 100.0,
3209 };
3210 // Three tight lines, a blank-line gap, two more: two paragraphs.
3211 let cells = vec![
3212 cell(
3213 "C.7. Wenn Sie diesen Vertrag widerrufen,",
3214 5.0,
3215 10.0,
3216 95.0,
3217 18.0,
3218 ),
3219 cell(
3220 "haben wir Ihnen alle Zahlungen, die wir",
3221 5.0,
3222 20.0,
3223 95.0,
3224 28.0,
3225 ),
3226 cell(
3227 "von Ihnen erhalten haben, zurückzuzahlen.",
3228 5.0,
3229 30.0,
3230 90.0,
3231 38.0,
3232 ),
3233 cell(
3234 "C.8. Wir können die Rückzahlung verweigern,",
3235 5.0,
3236 52.0,
3237 95.0,
3238 60.0,
3239 ),
3240 cell(
3241 "bis wir die Waren wieder zurückerhalten haben.",
3242 5.0,
3243 62.0,
3244 92.0,
3245 70.0,
3246 ),
3247 ];
3248 let mut regions = vec![panel.clone()];
3249 super::recover_text_panels(&mut regions, &cells);
3250 assert_eq!(
3251 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3252 ["text", "text"],
3253 "dense panel must demote into one text region per paragraph"
3254 );
3255 assert!(regions[0].b < regions[1].t, "paragraphs split at the gap");
3256 // Sparse narrow labels (a chart): picture survives.
3257 let labels = vec![
3258 cell("0", 5.0, 90.0, 8.0, 95.0),
3259 cell("50", 5.0, 50.0, 10.0, 55.0),
3260 cell("100", 5.0, 10.0, 12.0, 15.0),
3261 cell("t, s", 45.0, 96.0, 55.0, 100.0),
3262 ];
3263 let mut regions = vec![panel];
3264 super::recover_text_panels(&mut regions, &labels);
3265 assert_eq!(
3266 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3267 ["picture"]
3268 );
3269 }
3270
3271 /// An uncaptioned chart on a scanned page whose title, axis labels, and
3272 /// OCR boxes over the plot area are dense and wide enough to pass the
3273 /// coverage/width gates still keeps its crop: its line heights are ragged
3274 /// (title face vs tick labels vs bar-area OCR), failing the uniform-leading
3275 /// gate — a real text panel is set with constant leading (#173).
3276 #[test]
3277 fn dense_titled_chart_keeps_its_crop() {
3278 let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3279 text: text.to_string(),
3280 l,
3281 t,
3282 r,
3283 b,
3284 };
3285 let chart = Region {
3286 label: "picture",
3287 score: 0.9,
3288 l: 0.0,
3289 t: 0.0,
3290 r: 100.0,
3291 b: 100.0,
3292 };
3293 // Five wide lines at wildly different heights: a 12-pt title, 20-pt OCR
3294 // boxes over the bars, 4–5-pt tick/axis labels. Coverage and median
3295 // width both clear the panel thresholds.
3296 let cells = vec![
3297 cell("Underground Water Storage", 10.0, 5.0, 90.0, 17.0),
3298 cell("aquifer recharge zone", 15.0, 30.0, 75.0, 50.0),
3299 cell("confined | unconfined | perched", 12.0, 55.0, 80.0, 59.0),
3300 cell("saturated thickness", 8.0, 70.0, 60.0, 90.0),
3301 cell("distance from well, km", 20.0, 92.0, 85.0, 97.0),
3302 ];
3303 let mut regions = vec![chart];
3304 super::recover_text_panels(&mut regions, &cells);
3305 assert_eq!(
3306 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3307 ["picture"],
3308 "ragged line heights mark a figure, not a text panel"
3309 );
3310 }
3311
3312 /// docling serializes a cluster's cells in docling-parse index order
3313 /// (`_sort_cells`) and joins them with `PageAssembleModel.sanitize_text`:
3314 /// a space after every line except one ending in `-`, which either fuses a
3315 /// wrapped word (alnum on both sides — dash dropped) or glues verbatim (a
3316 /// bare `-` cell: `[0000` `-` `0002` → `[0000 -0002`, the 2305 ORCID line;
3317 /// `-` + `"C" cell -` + `a new table cell` → `-"C" cell a new table cell`,
3318 /// its OTSL list). Verified against the corpus: pure index order beats any
3319 /// geometric re-sort (normal_4pages' heading numerals paint after their
3320 /// text and belong last: `## 들어가며 1`).
3321 #[test]
3322 fn cells_join_in_index_order_with_sanitize_text_rules() {
3323 let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3324 text: text.to_string(),
3325 l,
3326 t,
3327 r,
3328 b,
3329 };
3330 let region = Region {
3331 label: "text",
3332 score: 1.0,
3333 l: 0.0,
3334 t: 95.0,
3335 r: 200.0,
3336 b: 130.0,
3337 };
3338 // ORCID superscript: a bare dash cell is a *detached* dash — kept, and
3339 // since docling#4052 (2.122) it joins with the ordinary space on both
3340 // sides (`[0000 -0002 -6960]` before that fix).
3341 let orcid = vec![
3342 cell("[0000", 10.0, 100.0, 30.0, 110.0),
3343 cell("−", 30.0, 100.0, 34.0, 110.0),
3344 cell("0002", 34.0, 100.0, 50.0, 110.0),
3345 cell("−", 50.0, 100.0, 54.0, 110.0),
3346 cell("6960]", 54.0, 100.0, 70.0, 110.0),
3347 ];
3348 assert_eq!(super::region_text(®ion, &orcid), "[0000 - 0002 - 6960]");
3349 // Wrapped word: dash dropped, lines fused (both boundary words alnum).
3350 let wrapped = vec![
3351 cell("platforms-", 10.0, 100.0, 60.0, 110.0),
3352 cell("reflects the design", 10.0, 112.0, 90.0, 122.0),
3353 ];
3354 assert_eq!(
3355 super::region_text(®ion, &wrapped),
3356 "platformsreflects the design"
3357 );
3358 // Dash-ending lines that are *detached* dashes (a bare bullet cell, a
3359 // `cell -` separator): the dash stays and the lines join with a space
3360 // — docling#4052; before it they glued (`-"C" cell a new table cell`,
3361 // 2305's OTSL list bullets).
3362 let otsl = vec![
3363 cell("–", 10.0, 100.0, 14.0, 110.0),
3364 cell("\"C\" cell -", 16.0, 100.0, 60.0, 110.0),
3365 cell("a new table cell", 10.0, 112.0, 80.0, 122.0),
3366 ];
3367 assert_eq!(
3368 super::region_text(®ion, &otsl),
3369 "- \"C\" cell - a new table cell"
3370 );
3371 // Index order is authoritative — no geometric re-sort.
3372 let numeral = vec![
3373 cell("들어가며", 30.0, 100.0, 80.0, 110.0),
3374 cell("1", 10.0, 98.0, 25.0, 112.0), // big numeral painted last
3375 ];
3376 assert_eq!(super::region_text(®ion, &numeral), "들어가며 1");
3377 }
3378
3379 /// The geometric-reliability gate, on the two shapes it has to tell apart.
3380 #[test]
3381 fn geometric_reliability_rejects_split_column_grids() {
3382 let g = |rows: &[&[&str]]| -> Vec<Vec<String>> {
3383 rows.iter()
3384 .map(|r| r.iter().map(|c| c.to_string()).collect())
3385 .collect()
3386 };
3387 // A genuine grid: dense, every column carrying entries. Nothing for
3388 // TableFormer to improve, so geometry is used as-is.
3389 assert!(super::geometric_table_is_reliable(&g(&[
3390 &["Datum", "Leistung", "Anzahl", "Kosten"],
3391 &["04.07", "Internet", "1", "40.30"],
3392 &["04.07", "Telefon", "2", "8.06"],
3393 ])));
3394 // The left-edge split artefact (the shape a scanned invoice produced):
3395 // one real label column plus values scattered across three sparse ones.
3396 assert!(!super::geometric_table_is_reliable(&g(&[
3397 &["www.magenta.at/faq", "", "", ""],
3398 &["Serviceteam", "", "", ""],
3399 &["Telefon", "0676/2000", "", ""],
3400 &["Kundennummer", "", "", "1.21699482"],
3401 &["Rechnungsnummer", "", "922769430725", ""],
3402 &["Rechnungsdatum", "", "", "04.07.2025"],
3403 ])));
3404 // A column only one row ever uses is a split artefact even when the
3405 // grid is otherwise dense.
3406 assert!(!super::geometric_table_is_reliable(&g(&[
3407 &["a", "b", ""],
3408 &["c", "d", ""],
3409 &["e", "f", "g"],
3410 ])));
3411 // Degenerate shapes are never vouched for — TableFormer may recover
3412 // structure a collapsed reconstruction lost.
3413 assert!(!super::geometric_table_is_reliable(&g(&[&[
3414 "only one column"
3415 ]])));
3416 assert!(!super::geometric_table_is_reliable(&[]));
3417 }
3418
3419 /// A `picture` region is cropped out of the rendered page, whatever built
3420 /// that page. The browser pipeline (#157) has no pdfium but does hand over
3421 /// the rasterized bitmap through `from_cells_with_image`, so it must get
3422 /// the same figure bytes the native path does — that is what makes
3423 /// `images = "embedded"` inline real pixels instead of a placeholder.
3424 #[cfg(feature = "ocr-prep")]
3425 #[test]
3426 fn picture_regions_are_cropped_from_a_host_supplied_page_image() {
3427 let mut img = image::RgbImage::new(200, 200);
3428 // Paint the figure area so the crop is distinguishable from the page.
3429 for y in 100..160 {
3430 for x in 20..120 {
3431 img.put_pixel(x, y, image::Rgb([255, 0, 0]));
3432 }
3433 }
3434 // scale 2.0: the region is in page points, the bitmap in pixels.
3435 let page = PdfPage::from_cells_with_image(100.0, 100.0, 2.0, Vec::new(), img);
3436 let region = Region {
3437 label: "picture",
3438 score: 0.9,
3439 l: 10.0,
3440 t: 50.0,
3441 r: 60.0,
3442 b: 80.0,
3443 };
3444 let (nodes, _) = super::assemble_page(&page, vec![region], &[None], &[None]);
3445 // Layout-derived nodes carry provenance, so the picture arrives wrapped.
3446 let image = nodes
3447 .iter()
3448 .find_map(|n| match n {
3449 Node::Located { inner, .. } => match &**inner {
3450 Node::Picture { image, .. } => image.as_ref(),
3451 _ => None,
3452 },
3453 Node::Picture { image, .. } => image.as_ref(),
3454 _ => None,
3455 })
3456 .expect("a picture node with cropped pixels");
3457 assert_eq!(image.mimetype, "image/png");
3458 assert_eq!((image.width, image.height), (100, 60), "region × scale");
3459 assert!(!image.data.is_empty(), "PNG bytes were encoded");
3460 }
3461
3462 #[test]
3463 fn link_anchors_split_a_shared_word_cell_between_adjacent_links() {
3464 // A common header layout: one text run holds several pipe-separated
3465 // labels, each carrying its own link annotation. Every link must get
3466 // its own label as the anchor (and the "|" separators must belong to
3467 // none), not the whole run.
3468 let annot = |l: f32, r: f32, uri: &str| LinkAnnot {
3469 l,
3470 t: 100.0,
3471 r,
3472 b: 114.0,
3473 uri: uri.into(),
3474 };
3475 let page = PdfPage {
3476 width: 600.0,
3477 height: 800.0,
3478 scale: 2.0,
3479 cells: Vec::new(),
3480 code_cells: Vec::new(),
3481 // "LinkedIn | GitHub | Credly" = 26 chars over x 100..360.
3482 word_cells: vec![cell(
3483 "LinkedIn | GitHub | Credly",
3484 100.0,
3485 100.0,
3486 360.0,
3487 114.0,
3488 )],
3489 image: image::RgbImage::new(1, 1),
3490 image_layout: None,
3491 links: vec![
3492 annot(100.0, 180.0, "https://l"),
3493 annot(200.0, 260.0, "https://g"),
3494 annot(290.0, 360.0, "https://c"),
3495 ],
3496 rotation: 0,
3497 };
3498 assert_eq!(
3499 resolve_link_anchors(&page),
3500 vec![
3501 ("LinkedIn".to_string(), "https://l".to_string()),
3502 ("GitHub".to_string(), "https://g".to_string()),
3503 ("Credly".to_string(), "https://c".to_string()),
3504 ]
3505 );
3506 }
3507
3508 /// A one-line code cell at `[l, r] × [t, b]` (top-left coords).
3509 fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
3510 TextCell {
3511 text: text.into(),
3512 l,
3513 t,
3514 r,
3515 b,
3516 }
3517 }
3518
3519 fn region(label: &'static str, score: f32, l: f32, t: f32, r: f32, b: f32) -> Region {
3520 Region {
3521 label,
3522 score,
3523 l,
3524 t,
3525 r,
3526 b,
3527 }
3528 }
3529
3530 #[test]
3531 fn resolve_collapses_nested_code_keeping_the_larger_box() {
3532 // A tight high-score `code` box and a taller lower-score near-duplicate that
3533 // contains it must collapse to one — the *larger* box, so every cell stays
3534 // covered and nothing leaks out as orphan text.
3535 let tight = region("code", 0.95, 78.0, 292.0, 300.0, 330.0);
3536 let wide = region("code", 0.66, 63.0, 260.0, 320.0, 346.0);
3537 let kept = super::resolve(vec![tight, wide]);
3538 assert_eq!(kept.len(), 1, "nested code boxes must collapse to one");
3539 assert!(
3540 kept[0].l == 63.0 && kept[0].b == 346.0,
3541 "the larger containing box is kept"
3542 );
3543 }
3544
3545 #[test]
3546 fn resolve_keeps_distinct_and_differently_typed_regions() {
3547 // A text box fully inside a lower-score *table* must NOT be collapsed (the
3548 // code dedup is code-only), and two separate code blocks stay separate.
3549 let text = region("text", 0.95, 90.0, 210.0, 200.0, 230.0);
3550 let table = region("table", 0.60, 80.0, 200.0, 400.0, 500.0);
3551 assert_eq!(super::resolve(vec![text, table]).len(), 2);
3552
3553 let code_a = region("code", 0.9, 78.0, 100.0, 300.0, 140.0);
3554 let code_b = region("code", 0.9, 78.0, 300.0, 300.0, 360.0); // far below, no overlap
3555 assert_eq!(super::resolve(vec![code_a, code_b]).len(), 2);
3556 }
3557
3558 #[test]
3559 fn code_language_label_above_code_is_detected() {
3560 // A bare "XML" token directly above a code box is a language label; a real
3561 // heading above the same code is not; a language word with no code below is
3562 // left alone.
3563 let label = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
3564 let code = region("code", 0.7, 77.0, 552.0, 290.0, 640.0);
3565 let heading = region("section_header", 0.9, 76.0, 500.0, 260.0, 512.0);
3566 let cells = vec![
3567 cell("XML", 78.0, 541.0, 94.0, 548.0), // inside `label`
3568 cell("Overview", 78.0, 501.0, 250.0, 511.0), // inside `heading`
3569 ];
3570 let drop = super::code_language_labels(&[label, code, heading], &cells);
3571 assert_eq!(drop, vec![true, false, false], "only the label is consumed");
3572
3573 // Same label with no code region present → not consumed.
3574 let label2 = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
3575 let only = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
3576 assert_eq!(super::code_language_labels(&[label2], &only), vec![false]);
3577
3578 // A label swallowed into the top of a wider code box (negative gap) is still
3579 // recognized.
3580 let inside_lbl = region("text", 0.9, 76.0, 540.0, 96.0, 549.0);
3581 let wide_code = region("code", 0.7, 63.0, 531.0, 320.0, 654.0);
3582 let cells2 = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
3583 assert_eq!(
3584 super::code_language_labels(&[inside_lbl, wide_code], &cells2),
3585 vec![true, false]
3586 );
3587
3588 assert!(super::is_code_language("XML") && super::is_code_language("c#"));
3589 assert!(!super::is_code_language("Configure") && !super::is_code_language("XML schema"));
3590 }
3591
3592 #[test]
3593 fn code_region_text_keeps_lines_and_indentation() {
3594 // Three source lines; each glyph is 6 units wide (width / chars = 6), so the
3595 // `int X;` line indented to x=22 is (22-10)/6 = 2 spaces in.
3596 let region = Region {
3597 label: "code",
3598 score: 1.0,
3599 l: 0.0,
3600 t: -5.0,
3601 r: 100.0,
3602 b: 40.0,
3603 };
3604 let cells = vec![
3605 cell("struct P {", 10.0, 0.0, 70.0, 10.0),
3606 cell("int X;", 22.0, 12.0, 58.0, 22.0),
3607 cell("}", 10.0, 24.0, 16.0, 34.0),
3608 ];
3609 assert_eq!(code_region_text(®ion, &cells), "struct P {\n int X;\n}");
3610 }
3611
3612 #[test]
3613 fn code_region_text_tightens_punctuation_without_eating_indentation() {
3614 // A fluent `.Foo()` line at x=22 (2 chars in). Per-line tightening must not
3615 // consume the leading indent space by matching " ." across it.
3616 let region = Region {
3617 label: "code",
3618 score: 1.0,
3619 l: 0.0,
3620 t: -5.0,
3621 r: 100.0,
3622 b: 40.0,
3623 };
3624 let cells = vec![
3625 cell("builder", 10.0, 0.0, 52.0, 10.0),
3626 // pdfium spaced the call: ".Foo (x)" tightens to ".Foo(x)", still 2-indented.
3627 cell(".Foo (x)", 22.0, 12.0, 70.0, 22.0),
3628 ];
3629 assert_eq!(code_region_text(®ion, &cells), "builder\n .Foo(x)");
3630 }
3631
3632 #[test]
3633 fn code_region_text_orders_out_of_order_cells_and_ignores_blank_lines() {
3634 let region = Region {
3635 label: "code",
3636 score: 1.0,
3637 l: 0.0,
3638 t: -5.0,
3639 r: 100.0,
3640 b: 60.0,
3641 };
3642 // Fed bottom-up and with a whitespace-only cell; output is top-down, no blank.
3643 let cells = vec![
3644 cell("b();", 10.0, 24.0, 34.0, 34.0),
3645 cell(" ", 10.0, 12.0, 20.0, 22.0),
3646 cell("a();", 10.0, 0.0, 34.0, 10.0),
3647 ];
3648 assert_eq!(code_region_text(®ion, &cells), "a();\nb();");
3649 // No code cells → empty, so the caller falls back to the prose text.
3650 assert_eq!(code_region_text(®ion, &[]), "");
3651 }
3652
3653 fn para(text: &str) -> Node {
3654 Node::Paragraph { text: text.into() }
3655 }
3656
3657 /// Run a node sequence through [`StreamAssembler`] with the given page splits
3658 /// and assert the flushed result equals one-shot [`merge_continuations`].
3659 fn assert_stream_eq(nodes: &[Node], splits: &[usize]) {
3660 let mut want = nodes.to_vec();
3661 merge_continuations(&mut want);
3662
3663 let mut asm = StreamAssembler::new();
3664 let mut got = Vec::new();
3665 let mut start = 0;
3666 for &end in splits {
3667 got.extend(asm.push(nodes[start..end].to_vec()));
3668 start = end;
3669 }
3670 got.extend(asm.push(nodes[start..].to_vec()));
3671 got.extend(asm.finish());
3672 assert_eq!(got, want, "stream assembly diverged (splits={splits:?})");
3673 }
3674
3675 #[test]
3676 fn stream_assembler_matches_merge_continuations() {
3677 // Open fragment + lowercase continuation split across a page boundary.
3678 let cross = [para("the definition of"), para("lists in scope")];
3679 assert_stream_eq(&cross, &[1]);
3680 assert_stream_eq(&cross, &[]);
3681
3682 // Continuation that wraps around a figure (+ its caption) on the boundary.
3683 let wrap = [
3684 para("the wing type that is"),
3685 Node::Picture {
3686 caption: None,
3687 caption_href: None,
3688 image: None,
3689 classification: None,
3690 caption_parent: Default::default(),
3691 },
3692 para("Fig. 1. a diagram"),
3693 para("the most common kind"),
3694 ];
3695 for splits in [&[][..], &[1][..], &[2][..], &[3][..], &[1, 3][..]] {
3696 assert_stream_eq(&wrap, splits);
3697 }
3698
3699 // A heading between fragments blocks the merge (must still flush correctly).
3700 let blocked = [
3701 para("ends mid word and"),
3702 Node::Heading {
3703 level: 2,
3704 text: "New Section".into(),
3705 },
3706 para("more body here"),
3707 ];
3708 for splits in [&[][..], &[1][..], &[2][..]] {
3709 assert_stream_eq(&blocked, splits);
3710 }
3711
3712 // A chain across three pages: each page is one open lowercase fragment.
3713 let chain = [
3714 para("alpha beta"),
3715 para("gamma delta"),
3716 para("epsilon zeta"),
3717 ];
3718 assert_stream_eq(&chain, &[1, 2]);
3719 }
3720
3721 #[test]
3722 fn clean_text_dehyphenates_and_normalizes_typography() {
3723 // U+0002 line-wrap hyphen + the join space → merged word (like docling).
3724 assert_eq!(clean_text("com\u{2} pact"), "compact");
3725 assert_eq!(clean_text("end-to\u{2} end deep"), "end-toend deep");
3726 // A stray wrap hyphen (no following join) is dropped.
3727 assert_eq!(clean_text("word\u{2}"), "word");
3728 // Typographic punctuation → ASCII: every curly quote becomes `'`
3729 // (docling-parse's sanitizer table), a literal `"` stays.
3730 assert_eq!(
3731 clean_text("Graph\u{2019}s \u{201c}x\u{201d} \"y\""),
3732 "Graph's 'x' \"y\""
3733 );
3734 assert_eq!(clean_text("a\u{2026}"), "a...");
3735 // The dp default (the docling-parse sanitizer) preserves internal spacing
3736 // it placed deliberately; line breaks/tabs normalize to a space, ends trim.
3737 assert_eq!(clean_text("a b\nc"), "a b c");
3738 }
3739
3740 /// docling#4064: a form's children are emitted together where the form
3741 /// sits in the top-level order, not interleaved with surrounding text.
3742 #[test]
3743 fn form_children_stay_together_in_reading_order() {
3744 let reg = |label: &'static str, l: f32, t: f32, r: f32, b: f32| Region {
3745 label,
3746 score: 0.9,
3747 l,
3748 t,
3749 r,
3750 b,
3751 };
3752 // Page: intro text, then a form spanning the left column with two
3753 // fields and a table inside, while a right-column paragraph sits
3754 // level with the form's first field (it would otherwise be read
3755 // between the form's children).
3756 let mut items = vec![
3757 reg("text", 50.0, 50.0, 550.0, 70.0), // 0 intro
3758 reg("form", 50.0, 100.0, 300.0, 400.0), // 1 container
3759 reg("text", 60.0, 110.0, 290.0, 130.0), // 2 field A (child)
3760 reg("text", 320.0, 110.0, 550.0, 130.0), // 3 right column paragraph
3761 reg("table", 60.0, 150.0, 290.0, 300.0), // 4 table (child)
3762 reg("text", 60.0, 320.0, 290.0, 340.0), // 5 field B (child)
3763 reg("text", 50.0, 450.0, 550.0, 470.0), // 6 outro
3764 ];
3765 let cids = super::cluster_cids(&items, &[]);
3766 super::order_with_containers(&mut items, &cids, 600.0, 800.0, |r| r);
3767 let order: Vec<(&str, f32)> = items.iter().map(|r| (r.label, r.t)).collect();
3768 // The form block (container, then its children top-down) is one unit.
3769 let form_pos = order.iter().position(|(l, _)| *l == "form").unwrap();
3770 assert_eq!(
3771 &order[form_pos..form_pos + 4],
3772 &[
3773 ("form", 100.0),
3774 ("text", 110.0),
3775 ("table", 150.0),
3776 ("text", 320.0)
3777 ]
3778 );
3779 assert_eq!(order[0], ("text", 50.0));
3780 assert_eq!(order[order.len() - 1], ("text", 450.0));
3781 // Without a container the plain order interleaves by geometry.
3782 let mut flat: Vec<Region> = items
3783 .iter()
3784 .filter(|r| r.label != "form")
3785 .cloned()
3786 .collect();
3787 let cids = super::cluster_cids(&flat, &[]);
3788 super::order_regions(&mut flat, &cids, 600.0, 800.0, |r| r);
3789 assert_ne!(
3790 flat.iter().map(|r| r.t).collect::<Vec<_>>(),
3791 order
3792 .iter()
3793 .filter(|(l, _)| *l != "form")
3794 .map(|(_, t)| *t)
3795 .collect::<Vec<_>>()
3796 );
3797 }
3798
3799 /// docling#3906: a picture inside a table lands in the covering cell,
3800 /// chosen by the picture's inferred grid position when cell boxes overlap.
3801 #[test]
3802 fn picture_matches_the_cell_at_its_grid_position() {
3803 let cell = |r: usize, c: usize, bbox: [f32; 4]| docling_core::TableCell {
3804 text: format!("r{r}c{c}"),
3805 bbox: Some(bbox),
3806 start_row: r,
3807 start_col: c,
3808 row_span: 1,
3809 col_span: 1,
3810 column_header: false,
3811 row_header: false,
3812 row_section: false,
3813 };
3814 // 2×2 grid; the (1,0) cell box is generous and also covers the picture.
3815 let cells = vec![
3816 cell(0, 0, [0.0, 0.0, 100.0, 50.0]),
3817 cell(0, 1, [100.0, 0.0, 200.0, 50.0]),
3818 cell(1, 0, [0.0, 50.0, 100.0, 100.0]),
3819 cell(1, 1, [100.0, 50.0, 200.0, 100.0]),
3820 ];
3821 let pic = Region {
3822 label: "picture",
3823 score: 0.9,
3824 l: 110.0,
3825 t: 60.0,
3826 r: 190.0,
3827 b: 95.0,
3828 };
3829 assert_eq!(super::match_picture_to_cell(&pic, &cells), Some((1.0, 3)));
3830 // A picture only half inside any cell is not nested.
3831 let straddling = Region {
3832 label: "picture",
3833 score: 0.9,
3834 l: 60.0,
3835 t: 60.0,
3836 r: 160.0,
3837 b: 95.0,
3838 };
3839 assert_eq!(super::match_picture_to_cell(&straddling, &cells), None);
3840 }
3841
3842 /// docling#4052 (2.122): a line-final dash fuses the wrapped word only
3843 /// when attached to it; a detached dash is a literal and the lines join
3844 /// with a space.
3845 #[test]
3846 fn line_final_hyphen_fuses_only_when_attached_to_a_word() {
3847 let line = |text: &str, t: f32| TextCell {
3848 text: text.to_string(),
3849 l: 0.0,
3850 t,
3851 r: 100.0,
3852 b: t + 10.0,
3853 };
3854 // `algo-` / `rithms`: attached hyphen, alnum on both sides → fused.
3855 assert_eq!(
3856 cells_text(vec![&line("algo-", 0.0), &line("rithms", 12.0)]),
3857 "algorithms"
3858 );
3859 // `pp. 545-` / `561`: attached, digits count as alnum → `545561` (upstream).
3860 assert_eq!(
3861 cells_text(vec![&line("pp. 545-", 0.0), &line("561", 12.0)]),
3862 "pp. 545561"
3863 );
3864 // A dash after whitespace — a separator or a lone `-` cell — is kept and
3865 // the lines take the ordinary joining space.
3866 assert_eq!(
3867 cells_text(vec![&line("range -", 0.0), &line("wide", 12.0)]),
3868 "range - wide"
3869 );
3870 assert_eq!(
3871 cells_text(vec![&line("-", 0.0), &line("item", 12.0)]),
3872 "- item"
3873 );
3874 // Attached but the next line opens with no word (`x-` / `...`): dash
3875 // kept and, as before, no separating space.
3876 assert_eq!(
3877 cells_text(vec![&line("x-", 0.0), &line("...", 12.0)]),
3878 "x-..."
3879 );
3880 }
3881
3882 #[test]
3883 fn lam_alef_only_swaps_a_genuinely_reversed_ligature() {
3884 // A mid-word `alef-variant + lam` is pdfium's reversed lam-alef ligature and
3885 // is swapped back to logical `lam + alef-variant` (`ب أ ل` → `ب ل أ`).
3886 assert_eq!(
3887 clean_text("\u{0628}\u{0623}\u{0644}"),
3888 "\u{0628}\u{0644}\u{0623}"
3889 );
3890 // But when the alef-variant is *already* preceded by a lam it is the logical
3891 // ligature `لآ`; the following lam is the next syllable's letter and must not
3892 // move. `التعلم الآلي` must stay `الآلي`, not become `اللآي`.
3893 assert_eq!(
3894 clean_text("\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"),
3895 "\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"
3896 );
3897 }
3898
3899 /// The #419 page, in points: three layout boxes over one paragraph, two of
3900 /// them ending partway through a line. The sliced lines miss the 0.2 claim
3901 /// and become orphans; the third model box starts above the second orphan,
3902 /// so unfitted the reading order emits that box first and strands the line.
3903 fn sliced_paragraph() -> (Vec<Region>, Vec<TextCell>) {
3904 let line = |text: &str, t: f32, r: f32| cell(text, 60.0, t, r, t + 11.0);
3905 let cells = vec![
3906 line("The mission of this series is to improve", 135.0, 458.0),
3907 line("The books in this series are technical,", 147.0, 458.0),
3908 line("substantial. The authors are", 159.0, 458.0),
3909 line("highly experienced craftsmen and", 171.5, 458.0), // sliced: 1.5/11 under box A
3910 line("actually works in practice, as opposed", 185.0, 458.0),
3911 line("about what the author has done, not", 197.0, 458.0),
3912 line("about programming, there will be lots", 210.5, 458.0), // sliced: 1.5/11 under box B
3913 line("will be lots of case studies from real", 223.0, 206.0), // C's line
3914 ];
3915 let regions = vec![
3916 region("text", 0.9, 60.0, 132.0, 458.0, 173.0), // A: three lines + a sliver of the 4th
3917 region("text", 0.9, 60.0, 184.0, 458.0, 212.0), // B: two lines + a sliver of the 7th
3918 region("text", 0.9, 60.0, 216.0, 206.0, 227.0), // C: last line, box opening 5.5pt too early
3919 ];
3920 (regions, cells)
3921 }
3922
3923 fn ordered_texts(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
3924 let mut items: Vec<Region> = regions.to_vec();
3925 let cids = super::cluster_cids(&items, cells);
3926 super::order_regions(&mut items, &cids, 500.0, 700.0, |r| r);
3927 super::region_texts_exclusive(&items, cells)
3928 .into_iter()
3929 .map(|t| t.chars().take(9).collect())
3930 .collect()
3931 }
3932
3933 /// #419: fitted to its cells, a model box that cut a line in half no longer
3934 /// overlaps the orphan that line became, so the orphan orders where it
3935 /// reads; unfitted, the same page strands the line after the paragraph.
3936 #[test]
3937 fn fitting_boxes_to_cells_puts_a_sliced_line_back_in_order() {
3938 let (mut regions, cells) = sliced_paragraph();
3939 super::add_orphan_regions(&mut regions, &cells);
3940 assert_eq!(regions.len(), 5, "two orphan lines");
3941 // The defect, for the record: C (top 216) is not strictly below the
3942 // orphan at 210.5–221.5, so the graph orders C first.
3943 assert_eq!(
3944 ordered_texts(®ions, &cells).last().map(String::as_str),
3945 Some("about pro")
3946 );
3947
3948 super::fit_regions_to_cells(&mut regions, &cells);
3949 assert_eq!(regions.len(), 5);
3950 // A ends on its last claimed line, C starts on its only one.
3951 assert_eq!((regions[0].t, regions[0].b), (135.0, 170.0));
3952 assert_eq!((regions[2].t, regions[2].b), (223.0, 234.0));
3953 assert_eq!(
3954 ordered_texts(®ions, &cells),
3955 [
3956 "The missi",
3957 "highly ex",
3958 "actually ",
3959 "about pro",
3960 "will be l"
3961 ]
3962 );
3963 }
3964
3965 /// An orphan the fitted paragraph box surrounds (a short middle line the
3966 /// narrow model box missed while claiming the lines around it) is folded
3967 /// into the paragraph; an empty regular box goes away, a formula stays, a
3968 /// picture is never refitted, and a page with no cells is left untouched.
3969 #[test]
3970 fn fitting_folds_surrounded_orphans_and_drops_empty_regulars() {
3971 let wide = |text: &str, t: f32| cell(text, 60.0, t, 400.0, t + 11.0);
3972 let cells = vec![
3973 wide("first line of the paragraph", 100.0),
3974 cell("stray", 250.0, 112.0, 400.0, 123.0), // clear of the narrow box
3975 wide("third line of the paragraph", 124.0),
3976 ];
3977 let mut regions = vec![
3978 // Narrow box: claims the wide lines at 0.41, misses the short one.
3979 region("text", 0.9, 60.0, 98.0, 200.0, 136.0),
3980 region("section_header", 0.8, 60.0, 300.0, 200.0, 320.0), // no cells
3981 region("formula", 0.8, 60.0, 340.0, 200.0, 360.0), // no cells, kept
3982 region("picture", 0.8, 0.0, 400.0, 500.0, 600.0),
3983 ];
3984 super::add_orphan_regions(&mut regions, &cells);
3985 assert_eq!(regions.len(), 5, "the short line became an orphan");
3986 super::fit_regions_to_cells(&mut regions, &cells);
3987 let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
3988 assert_eq!(labels, ["text", "formula", "picture"]);
3989 let para = ®ions[0];
3990 assert_eq!(
3991 (para.l, para.t, para.r, para.b),
3992 (60.0, 100.0, 400.0, 135.0)
3993 );
3994 assert_eq!(
3995 super::region_texts_exclusive(®ions, &cells)[0],
3996 "first line of the paragraph stray third line of the paragraph"
3997 );
3998 assert_eq!(
3999 (regions[2].t, regions[2].b),
4000 (400.0, 600.0),
4001 "picture untouched"
4002 );
4003
4004 let mut untouched = vec![region("text", 0.9, 0.0, 0.0, 10.0, 10.0)];
4005 super::fit_regions_to_cells(&mut untouched, &[]);
4006 assert_eq!(untouched.len(), 1, "no cells yet: nothing dropped");
4007 }
4008}