docling_pdf/assemble.rs
1//! Layout-driven assembly: map detected [`Region`]s + text cells to a
2//! [`DoclingDocument`], mirroring docling's page-assembly + reading-order.
3//!
4//! Overlapping detections are resolved greedily by score, each text cell is
5//! assigned to its best-containing region, regions are ordered in reading order
6//! (two-column aware), and each becomes a typed node by its layout label.
7
8use docling_core::{CaptionParent, Node, PictureClass, PictureImage, Table};
9#[cfg(any(feature = "ml", feature = "ocr-prep"))]
10use image::RgbImage;
11
12use crate::layout::Region;
13use crate::pdfium_backend::{PdfPage, TextCell};
14
15fn area(l: f32, t: f32, r: f32, b: f32) -> f32 {
16 ((r - l).max(0.0)) * ((b - t).max(0.0))
17}
18
19/// Intersection area of two boxes.
20fn inter(a: &Region, l: f32, t: f32, r: f32, b: f32) -> f32 {
21 let il = a.l.max(l);
22 let it = a.t.max(t);
23 let ir = a.r.min(r);
24 let ib = a.b.min(b);
25 area(il, it, ir, ib)
26}
27
28/// Wrapper (structured-region) labels, ported from docling
29/// `LayoutPostprocessor.WRAPPER_TYPES`: a region that *contains* other regions
30/// and renders as a structured block (a table / table-of-contents index), not as
31/// its own flat text.
32fn is_wrapper(label: &str) -> bool {
33 matches!(
34 label,
35 "table" | "document_index" | "form" | "key_value_region"
36 )
37}
38
39/// Labels docling's table-structure (TableFormer) model runs on and that render
40/// as a Markdown table: a plain `table` and a `document_index` (a table of
41/// contents), which docling assembles as a `TableItem` too.
42pub fn is_table_like(label: &str) -> bool {
43 matches!(label, "table" | "document_index")
44}
45
46/// Greedily keep regions by descending score, dropping a region that is mostly
47/// covered by an already-kept one (RT-DETR emits overlapping duplicates).
48fn greedy(mut regions: Vec<Region>) -> Vec<Region> {
49 regions.sort_by(|a, b| b.score.total_cmp(&a.score));
50 let mut kept: Vec<Region> = Vec::new();
51 for r in regions {
52 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
53 let covered = kept.iter().any(|k| {
54 let i = inter(&r, k.l, k.t, k.r, k.b);
55 let ka = area(k.l, k.t, k.r, k.b).max(1.0);
56 // drop if most of r is inside k, or they strongly mutually overlap
57 i / ra > 0.7 || i / (ra + ka - i) > 0.5
58 });
59 if !covered {
60 kept.push(r);
61 }
62 }
63 kept
64}
65
66/// Resolve overlapping RT-DETR detections, ported from the bucket structure of
67/// docling's `LayoutPostprocessor`: regular, picture and wrapper clusters live in
68/// **separate** spatial indexes and are de-overlapped independently, so a
69/// high-score picture never suppresses a lower-score table or table-of-contents
70/// index (the redp5110 TOC that was otherwise replaced by a picture box). A
71/// cross-type pass first drops a picture that nearly coincides with a table
72/// (`_handle_cross_type_overlaps`), keeping the structured table.
73/// docling's `_remove_overlapping_clusters("picture")`: same-label picture
74/// detections whose boxes heavily overlap (IoU > 0.8, or either box > 80 %
75/// contained in the other) form one group, and a single survivor is kept per
76/// group. Survivor selection ports `_should_prefer_cluster` /
77/// `_select_best_cluster_from_group` with the picture params
78/// (`area_threshold` 2.0, `conf_threshold` 0.3): a candidate is rejected only
79/// when a rival is both comparable in size (candidate ≤ 2× its area) and
80/// clearly more confident (> 0.3); among the survivors the *larger* box wins
81/// unless it is > 0.3 less confident. Net effect on the corpus: a figure the
82/// detector proposes both whole and as its sub-panels (2206's four-thumbnail
83/// Figure 1) collapses to the whole-figure box, exactly like docling.
84pub(crate) fn dedup_pictures(regions: &mut Vec<Region>) {
85 remove_overlapping_specials(regions, |l| l == "picture", 2.0, 0.3);
86}
87
88/// docling's `_remove_overlapping_clusters` for one special bucket (the
89/// regions whose label satisfies `in_bucket`), with that bucket's
90/// `OVERLAP_PARAMS`: picture (2.0, 0.3) — see [`dedup_pictures`] — or wrapper
91/// (2.0, 0.2) for the table bucket in [`resolve`]. Regions outside the bucket
92/// are untouched; a bucket member overlapping nothing is always kept.
93fn remove_overlapping_specials(
94 regions: &mut Vec<Region>,
95 in_bucket: impl Fn(&str) -> bool,
96 area_threshold: f32,
97 conf_threshold: f32,
98) {
99 let idx: Vec<usize> = (0..regions.len())
100 .filter(|&i| in_bucket(regions[i].label))
101 .collect();
102 if idx.len() < 2 {
103 return;
104 }
105 // Union-find over the bucket.
106 let mut parent: Vec<usize> = (0..idx.len()).collect();
107 fn find(parent: &mut [usize], i: usize) -> usize {
108 let mut root = i;
109 while parent[root] != root {
110 root = parent[root];
111 }
112 let mut cur = i;
113 while parent[cur] != root {
114 let next = parent[cur];
115 parent[cur] = root;
116 cur = next;
117 }
118 root
119 }
120 let boxed = |r: &Region| (r.l, r.t, r.r, r.b);
121 for a in 0..idx.len() {
122 for b in (a + 1)..idx.len() {
123 let (ra, rb) = (®ions[idx[a]], ®ions[idx[b]]);
124 let (al, at, ar, ab_) = boxed(ra);
125 let (bl, bt, br, bb) = boxed(rb);
126 let ix = (ar.min(br) - al.max(bl)).max(0.0);
127 let iy = (ab_.min(bb) - at.max(bt)).max(0.0);
128 let inter = ix * iy;
129 let aa = area(al, at, ar, ab_).max(f32::EPSILON);
130 let ba = area(bl, bt, br, bb).max(f32::EPSILON);
131 let iou = inter / (aa + ba - inter).max(f32::EPSILON);
132 if iou > 0.8 || inter / aa > 0.8 || inter / ba > 0.8 {
133 let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
134 if pa != pb {
135 parent[pa] = pb;
136 }
137 }
138 }
139 }
140 // Per group, run docling's pairwise preference + larger-wins selection.
141 let mut groups: std::collections::HashMap<usize, Vec<usize>> = std::collections::HashMap::new();
142 for i in 0..idx.len() {
143 let root = find(&mut parent, i);
144 groups.entry(root).or_default().push(i);
145 }
146 let mut drop = vec![false; regions.len()];
147 for group in groups.values() {
148 if group.len() < 2 {
149 continue;
150 }
151 let area_of = |i: usize| {
152 let r = ®ions[idx[i]];
153 area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
154 };
155 let mut best: Option<usize> = None;
156 for &cand in group {
157 let passes = group.iter().all(|&other| {
158 if other == cand {
159 return true;
160 }
161 let area_ratio = area_of(cand) / area_of(other);
162 let conf_diff = regions[idx[other]].score - regions[idx[cand]].score;
163 !(area_ratio <= area_threshold && conf_diff > conf_threshold)
164 });
165 if passes {
166 best = Some(match best {
167 None => cand,
168 Some(cur) => {
169 if area_of(cand) > area_of(cur)
170 && regions[idx[cur]].score - regions[idx[cand]].score <= conf_threshold
171 {
172 cand
173 } else {
174 cur
175 }
176 }
177 });
178 }
179 }
180 // Every candidate rejected can't happen with docling's rule (rejection
181 // needs a strictly better rival); guard with highest score anyway.
182 let keep = best.unwrap_or_else(|| {
183 *group
184 .iter()
185 .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(®ions[idx[b]].score))
186 .expect("non-empty group")
187 });
188 for &i in group {
189 if i != keep {
190 drop[idx[i]] = true;
191 }
192 }
193 }
194 let mut keep_iter = drop.into_iter();
195 regions.retain(|_| !keep_iter.next().expect("aligned"));
196}
197
198/// docling's `_remove_overlapping_clusters("regular")` for what [`greedy`]
199/// leaves standing, run on OCR'd pages before the region-scoped OCR (found
200/// with #471's synthetic scan: every line of the page came out twice).
201///
202/// `greedy` keeps regions by descending score and drops a candidate mostly
203/// inside an already-kept one — so a *lower*-score block that contains
204/// several higher-score line boxes (RT-DETR's favourite reading of a sparse
205/// scanned page: every line, plus the paragraph) survives alongside them.
206/// On a digital page that is harmless: [`fit_regions_to_cells`] hands each
207/// text cell to one owner and drops the regions left empty. On a scanned
208/// page the cells don't exist yet — they come from OCR of *each* region's
209/// crop — so the block and its lines were each recognized, and the block,
210/// which then owned both cell sets, read every line twice.
211///
212/// Upstream never has this problem because its OCR runs over the bitmap
213/// before layout postprocessing and each cell is assigned once; the
214/// postprocessor's regular pass then groups clusters that overlap (IoU > 0.8,
215/// or either > 80 % contained in the other) with a union-find, keeps one
216/// survivor per group via `_should_prefer_cluster` /
217/// `_select_best_cluster_from_group` (`area_threshold` 1.3, `conf_threshold`
218/// 0.05; a LIST_ITEM beats a same-sized TEXT, a CODE box beats what it
219/// contains) and merges the losers' cells into it, whose box is then fitted
220/// to those cells. The equivalent for region-scoped OCR: one survivor per
221/// group, keeping its label and score, with the group's **union** box so the
222/// single crop still covers every merged line. Pictures and wrappers are not
223/// regulars and are left alone ([`dedup_pictures`], [`resolve`]).
224pub(crate) fn merge_overlapping_regulars(regions: &mut Vec<Region>) {
225 let idx: Vec<usize> = (0..regions.len())
226 .filter(|&i| regions[i].label != "picture" && !is_wrapper(regions[i].label))
227 .collect();
228 if idx.len() < 2 {
229 return;
230 }
231 let mut parent: Vec<usize> = (0..idx.len()).collect();
232 fn find(parent: &mut [usize], i: usize) -> usize {
233 let mut root = i;
234 while parent[root] != root {
235 root = parent[root];
236 }
237 let mut cur = i;
238 while parent[cur] != root {
239 let next = parent[cur];
240 parent[cur] = root;
241 cur = next;
242 }
243 root
244 }
245 for a in 0..idx.len() {
246 for b in (a + 1)..idx.len() {
247 let (ra, rb) = (®ions[idx[a]], ®ions[idx[b]]);
248 let i = inter(ra, rb.l, rb.t, rb.r, rb.b);
249 let aa = area(ra.l, ra.t, ra.r, ra.b).max(f32::EPSILON);
250 let ba = area(rb.l, rb.t, rb.r, rb.b).max(f32::EPSILON);
251 if i / (aa + ba - i).max(f32::EPSILON) > 0.8 || i / aa > 0.8 || i / ba > 0.8 {
252 let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
253 if pa != pb {
254 parent[pa] = pb;
255 }
256 }
257 }
258 }
259 let mut groups: std::collections::BTreeMap<usize, Vec<usize>> =
260 std::collections::BTreeMap::new();
261 for i in 0..idx.len() {
262 let root = find(&mut parent, i);
263 groups.entry(root).or_default().push(i);
264 }
265 const AREA_THRESHOLD: f32 = 1.3;
266 const CONF_THRESHOLD: f32 = 0.05;
267 let area_of = |i: usize| {
268 let r = ®ions[idx[i]];
269 area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
270 };
271 // `_should_prefer_cluster(candidate, other)` with the regular params.
272 let prefer = |cand: usize, other: usize| -> bool {
273 let (c, o) = (®ions[idx[cand]], ®ions[idx[other]]);
274 let area_ratio = area_of(cand) / area_of(other);
275 if c.label == "list_item" && o.label == "text" && (1.0 - area_ratio).abs() < 0.2 {
276 return true;
277 }
278 if c.label == "code" && inter(o, c.l, c.t, c.r, c.b) / area_of(other) > 0.8 {
279 return true;
280 }
281 !(area_ratio <= AREA_THRESHOLD && o.score - c.score > CONF_THRESHOLD)
282 };
283 let mut drop = vec![false; regions.len()];
284 let mut unions: Vec<(usize, (f32, f32, f32, f32))> = Vec::new();
285 for group in groups.values() {
286 if group.len() < 2 {
287 continue;
288 }
289 let mut best: Option<usize> = None;
290 for &cand in group {
291 if group
292 .iter()
293 .all(|&other| other == cand || prefer(cand, other))
294 {
295 best = Some(match best {
296 None => cand,
297 Some(cur)
298 if area_of(cand) > area_of(cur)
299 && regions[idx[cur]].score - regions[idx[cand]].score
300 <= CONF_THRESHOLD =>
301 {
302 cand
303 }
304 Some(cur) => cur,
305 });
306 }
307 }
308 // docling falls back to the group's first cluster; the highest score
309 // is the deterministic equivalent for a set with no insertion order.
310 let keep = best.unwrap_or_else(|| {
311 *group
312 .iter()
313 .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(®ions[idx[b]].score))
314 .expect("non-empty group")
315 });
316 let mut u = (
317 f32::INFINITY,
318 f32::INFINITY,
319 f32::NEG_INFINITY,
320 f32::NEG_INFINITY,
321 );
322 for &i in group {
323 let r = ®ions[idx[i]];
324 u = (u.0.min(r.l), u.1.min(r.t), u.2.max(r.r), u.3.max(r.b));
325 if i != keep {
326 drop[idx[i]] = true;
327 }
328 }
329 unions.push((idx[keep], u));
330 }
331 for (i, (l, t, r, b)) in unions {
332 let k = &mut regions[i];
333 (k.l, k.t, k.r, k.b) = (l, t, r, b);
334 }
335 let mut keep_iter = drop.into_iter();
336 regions.retain(|_| !keep_iter.next().expect("aligned"));
337}
338
339/// `intersection_over_union` of two regions.
340fn iou(a: &Region, b: &Region) -> f32 {
341 let i = inter(a, b.l, b.t, b.r, b.b);
342 let u = area(a.l, a.t, a.r, a.b) + area(b.l, b.t, b.r, b.b) - i;
343 if u > 0.0 {
344 i / u
345 } else {
346 0.0
347 }
348}
349
350/// docling's `_resolve_coincident_pairs` (#4059, 2.122): for every (loser,
351/// winner) pair at a near-identical box (IoU > 0.8) whose confidences are
352/// within 0.1 (`loser.score - winner.score < 0.1`), the loser label is dropped
353/// so the label with the richer downstream semantic survives. Nothing else —
354/// containment, area — is considered; a clearly more confident loser stays.
355fn coincident_losers(regions: &[Region], losers: &[usize], winners: &[usize]) -> Vec<usize> {
356 let mut out = Vec::new();
357 for &li in losers {
358 for &wi in winners {
359 if iou(®ions[li], ®ions[wi]) > 0.8 && regions[li].score - regions[wi].score < 0.1
360 {
361 out.push(li);
362 break;
363 }
364 }
365 }
366 out
367}
368
369/// docling's `_handle_cross_type_overlaps` (2.122/2.123 shape): the layout
370/// model can emit one grounded region under several labels, and the picture /
371/// table / container buckets are de-overlapped independently, so such a region
372/// survives twice. Elect a winner for the near-identical pairs:
373///
374/// | pair | loser | winner |
375/// |---------------------------------------|-----------|---------------------|
376/// | TABLE vs DOCUMENT_INDEX | table | document_index |
377/// | PICTURE vs TABLE / DOCUMENT_INDEX | picture | the table-like |
378/// | FORM / KEY_VALUE_REGION vs *surviving* TABLE / DOCUMENT_INDEX / PICTURE | container | structured element |
379///
380/// IoU (not containment) so a genuine small figure inside a large table region
381/// is not removed; the confidence tolerance keeps a clearly more confident
382/// loser (an earlier port dropped every coincident picture regardless).
383fn handle_cross_type_overlaps(regions: Vec<Region>) -> Vec<Region> {
384 let by = |pred: &dyn Fn(&str) -> bool| -> Vec<usize> {
385 (0..regions.len())
386 .filter(|&i| pred(regions[i].label))
387 .collect()
388 };
389 let tables = by(&|l| l == "table");
390 let doc_indices = by(&|l| l == "document_index");
391 let pictures = by(&|l| l == "picture");
392 let containers = by(&|l| matches!(l, "form" | "key_value_region"));
393 let mut drop = vec![false; regions.len()];
394 for i in coincident_losers(®ions, &tables, &doc_indices) {
395 drop[i] = true;
396 }
397 let table_like: Vec<usize> = tables.iter().chain(&doc_indices).copied().collect();
398 for i in coincident_losers(®ions, &pictures, &table_like) {
399 drop[i] = true;
400 }
401 let structured: Vec<usize> = table_like
402 .iter()
403 .chain(&pictures)
404 .copied()
405 .filter(|&i| !drop[i])
406 .collect();
407 for i in coincident_losers(®ions, &containers, &structured) {
408 drop[i] = true;
409 }
410 let mut drop = drop.into_iter();
411 let mut regions = regions;
412 regions.retain(|_| !drop.next().expect("aligned"));
413 regions
414}
415
416pub fn resolve(regions: Vec<Region>) -> Vec<Region> {
417 let regions = handle_cross_type_overlaps(regions);
418 // De-overlap each bucket on its own.
419 let pictures = greedy(
420 regions
421 .iter()
422 .filter(|r| r.label == "picture")
423 .cloned()
424 .collect(),
425 );
426 // Tables and containers are separate buckets since docling 2.123
427 // (`TABLE_TYPES` vs `CONTAINER_TYPES`, docling#4064): a form drawn around a
428 // table no longer competes with it for survival — the table nests inside
429 // the container instead (`order_with_containers`).
430 let mut tables = greedy(
431 regions
432 .iter()
433 .filter(|r| is_table_like(r.label))
434 .cloned()
435 .collect(),
436 );
437 // `greedy` only drops a table mostly inside a *more* confident one, so a
438 // low-score whole-page table proposed over the column tables it contains
439 // (a two-column glossary page: 0.53 over 0.71/0.67/0.66) survived next to them and every
440 // cell was emitted twice. docling's `_remove_overlapping_clusters(tables,
441 // "wrapper")` groups tables whose boxes overlap (IoU > 0.8, or either one
442 // > 80 % inside the other) and keeps one per group: run it on what
443 // `greedy` leaves, like `merge_overlapping_regulars` does for regulars.
444 remove_overlapping_specials(&mut tables, |_| true, 2.0, 0.2);
445 let containers = greedy(
446 regions
447 .iter()
448 .filter(|r| matches!(r.label, "form" | "key_value_region"))
449 .cloned()
450 .collect(),
451 );
452 let mut kept = greedy(
453 regions
454 .iter()
455 .filter(|r| r.label != "picture" && !is_wrapper(r.label))
456 .cloned()
457 .collect(),
458 );
459 dedup_nested_code(&mut kept);
460 kept.extend(pictures);
461 kept.extend(tables);
462 kept.extend(containers);
463 kept
464}
465
466/// Drop a regular region that is >80% contained in a surviving special region we
467/// render **as a single unit** — a table/table-of-contents index — ported from
468/// docling's "Remove regular clusters that are included in wrappers" step: the
469/// special absorbs it as a child (a table cell), so it must not also be emitted
470/// as its own paragraph/list-item. This stops the survey list-items from
471/// appearing both inside the detected table and again as bullets
472/// (`table_mislabeled_as_picture`).
473///
474/// `picture` regions are **not** in the swallow set: docling keeps a picture's
475/// contained clusters as the `PictureItem`'s *children* in the document JSON
476/// (`_set_cluster_children`, `ReadingOrderModel._add_child_elements`), while
477/// its `MarkdownPictureSerializer` prints only the caption and the image — the
478/// children never reach the Markdown (verified against the corpus groundtruth:
479/// `amt_handbook`'s in-figure callout labels are absent). So the regulars
480/// inside a picture stay in the region list, claim their cells and are fitted
481/// like any regular, and [`assemble_page`] lifts them out of the page's
482/// reading order into a [`Node::PictureChildren`] after their picture (see
483/// [`picture_parents`]). A line only partially under a figure box (straddling
484/// its border, ≤80 % contained) is no picture's child and is emitted, as
485/// docling emits it (#165).
486///
487/// `form` / `key_value_region` wrappers are deliberately **excluded**: this
488/// pipeline does not render them as a structured block (they are skipped), so
489/// their textual content comes precisely from the contained regular regions —
490/// A `page_footer` that is really the body of the last heading on the page
491/// becomes `text`.
492///
493/// The layout model labels the bottom margin by position as much as by
494/// content: a one-line paragraph that happens to sit where a running footer
495/// would — a CV's `Languages` line under its `## Languages` heading, the last
496/// entry of a section that runs to the page edge — comes back as
497/// `page_footer` (0.88 against 0.49 for `text` on the reporting file, with
498/// the fp32 model and either renderer alike), and Markdown drops furniture,
499/// so the document loses real content while the JSON keeps it under the
500/// wrong label. docling fails the same way. Deliberate deviation, gated so a
501/// real footer keeps its label: the nearest `section_header` above the footer
502/// must overlap it horizontally, end within 2.5 footer heights of its top,
503/// and be the **last body element** of its column — no other non-furniture
504/// region starts at or below the heading's bottom edge over the same span —
505/// and the footer must be a line of text, not a page number: at least 40 %
506/// of the page width. A running footer never follows a heading that has no
507/// body of its own, so the combination is the misread heading body.
508pub fn reclaim_heading_body_footers(regions: &mut [Region], page_w: f32) {
509 let n = regions.len();
510 let overlap_x = |a: &Region, b: &Region| a.r.min(b.r) - a.l.max(b.l) > 0.0;
511 for fi in 0..n {
512 let f = regions[fi].clone();
513 if f.label != "page_footer" || f.r - f.l < 0.4 * page_w {
514 continue;
515 }
516 let fh = (f.b - f.t).max(1.0);
517 // The nearest heading above the footer, over the footer's span.
518 let heading = (0..n)
519 .filter(|&j| {
520 let h = ®ions[j];
521 j != fi && h.label == "section_header" && h.b <= f.t + 0.5 * fh && overlap_x(h, &f)
522 })
523 .min_by(|&a, &b| regions[b].b.total_cmp(®ions[a].b));
524 let Some(hi) = heading else {
525 continue;
526 };
527 let h = regions[hi].clone();
528 if f.t - h.b > 2.5 * fh {
529 continue;
530 }
531 // The heading must have no body of its own: nothing but the footer
532 // starts at or below its bottom edge over the heading's or footer's
533 // span (a heading whose paragraph follows is not this case, and a
534 // heading with the footer far below it was filtered above).
535 let has_body = (0..n).any(|j| {
536 let r = ®ions[j];
537 j != fi
538 && j != hi
539 && !matches!(r.label, "page_footer" | "page_header")
540 && r.t >= h.b - 0.5 * fh
541 && (overlap_x(r, &h) || overlap_x(r, &f))
542 });
543 if has_body {
544 continue;
545 }
546 regions[fi].label = "text";
547 }
548}
549
550/// dropping those would erase the page (e.g. `right_to_left_03`'s form-heavy
551/// pages). Runs *after* [`drop_false_pictures`] so a phantom picture can't
552/// swallow real text on its way out.
553pub fn drop_contained_regulars(regions: &mut Vec<Region>) {
554 let specials: Vec<(f32, f32, f32, f32)> = regions
555 .iter()
556 .filter(|r| is_table_like(r.label))
557 .map(|r| (r.l, r.t, r.r, r.b))
558 .collect();
559 if specials.is_empty() {
560 return;
561 }
562 regions.retain(|r| {
563 if r.label == "picture" || is_wrapper(r.label) {
564 return true;
565 }
566 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
567 !specials
568 .iter()
569 .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.8)
570 });
571}
572
573/// docling's `_set_cluster_children` for pictures: a regular region (one that
574/// claims cells) whose box is > 80 % inside a picture's box
575/// (`intersection_over_self`) is that picture's child — not a page element
576/// (it leaves the reading order and the layout score), but an item under the
577/// `PictureItem` in the JSON. Returns, per region, the index of its parent
578/// picture: the smallest containing one when pictures nest (upstream would
579/// attach it to each).
580pub fn picture_parents(regions: &[Region]) -> Vec<Option<usize>> {
581 regions
582 .iter()
583 .map(|r| {
584 if !claims_cells(r) {
585 return None;
586 }
587 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
588 regions
589 .iter()
590 .enumerate()
591 .filter(|(_, p)| p.label == "picture" && inter(r, p.l, p.t, p.r, p.b) / ra > 0.8)
592 .min_by(|(_, a), (_, b)| {
593 area(a.l, a.t, a.r, a.b).total_cmp(&area(b.l, b.t, b.r, b.b))
594 })
595 .map(|(i, _)| i)
596 })
597 .collect()
598}
599
600/// True for a bare, single-token source-code language label (`XML`, `C#`, `JSON`,
601/// `bash`, …) — the little header the docs render above a code block. Matched
602/// case-insensitively; anything with whitespace or longer than a token is out.
603fn is_code_language(t: &str) -> bool {
604 let t = t.trim();
605 if t.is_empty() || t.chars().any(char::is_whitespace) || t.chars().count() > 12 {
606 return false;
607 }
608 const LANGS: &[&str] = &[
609 "xml",
610 "html",
611 "xhtml",
612 "json",
613 "jsonc",
614 "yaml",
615 "yml",
616 "toml",
617 "ini",
618 "c#",
619 "csharp",
620 "f#",
621 "fsharp",
622 "vb",
623 "c",
624 "c++",
625 "cpp",
626 "java",
627 "kotlin",
628 "scala",
629 "go",
630 "golang",
631 "rust",
632 "swift",
633 "javascript",
634 "js",
635 "typescript",
636 "ts",
637 "jsx",
638 "tsx",
639 "python",
640 "py",
641 "ruby",
642 "rb",
643 "php",
644 "perl",
645 "lua",
646 "r",
647 "dart",
648 "bash",
649 "sh",
650 "shell",
651 "powershell",
652 "zsh",
653 "batch",
654 "cmd",
655 "sql",
656 "tsql",
657 "plsql",
658 "graphql",
659 "dockerfile",
660 "makefile",
661 "css",
662 "scss",
663 "sass",
664 "less",
665 "markdown",
666 "md",
667 "tex",
668 "latex",
669 "diff",
670 "proto",
671 "razor",
672 "cshtml",
673 "xaml",
674 "aspx",
675 "http",
676 ];
677 let lower = t.to_ascii_lowercase();
678 LANGS.contains(&lower.as_str())
679}
680
681/// Mark the region indices that are a code block's **language label** — a bare
682/// `XML`/`C#`/… token sitting directly above a `code` region — so they are consumed
683/// rather than emitted as their own stray paragraph/heading. The label may also be
684/// captured inside a wider code box (rendered as the fence's first line); dropping
685/// the standalone copy just removes the duplicate.
686fn code_language_labels(regions: &[Region], cells: &[TextCell]) -> Vec<bool> {
687 let mut drop = vec![false; regions.len()];
688 for (i, r) in regions.iter().enumerate() {
689 if matches!(r.label, "code" | "picture" | "table") {
690 continue;
691 }
692 if !is_code_language(®ion_text(r, cells)) {
693 continue;
694 }
695 // The label sits just above the code (a blank line's gap) or is swallowed
696 // into the top of a wider code box; either way it is that block's label.
697 // The window is generous because the label's own font is small, so a
698 // one-line gap is several times its height.
699 let line_h = (r.b - r.t).abs().max(1.0);
700 let window = (line_h * 4.0).max(28.0);
701 let labels_code = regions.iter().enumerate().any(|(j, c)| {
702 if j == i || c.label != "code" {
703 return false;
704 }
705 let gap = c.t - r.b; // >0 when the code is below the label
706 let h_overlap = (r.r.min(c.r) - r.l.max(c.l)).max(0.0);
707 gap > -line_h * 3.0 && gap < window && h_overlap > 0.0
708 });
709 if labels_code {
710 drop[i] = true;
711 }
712 }
713 drop
714}
715
716/// Collapse `code` regions where one is nested inside another, keeping the larger.
717///
718/// RT-DETR sometimes emits a tight code box *and* a wider near-duplicate that also
719/// captures the block's language label (`XML`, `C#`, …). When the tight box scores
720/// higher it is kept first, and the wider container — not "mostly inside" the tight
721/// box — survives [`resolve`]'s greedy pass, so the block is emitted twice. Keeping
722/// the **larger** box (rather than dropping it) collapses the pair without leaking
723/// the container's extra cells back out as orphan text, since the larger box still
724/// covers every cell. Restricted to `code` so genuinely distinct nested regions of
725/// other kinds are untouched.
726fn dedup_nested_code(kept: &mut Vec<Region>) {
727 let mut drop = vec![false; kept.len()];
728 for i in 0..kept.len() {
729 if kept[i].label != "code" {
730 continue;
731 }
732 let ai = area(kept[i].l, kept[i].t, kept[i].r, kept[i].b).max(1.0);
733 for j in 0..kept.len() {
734 if i == j || drop[j] || kept[j].label != "code" {
735 continue;
736 }
737 let aj = area(kept[j].l, kept[j].t, kept[j].r, kept[j].b).max(1.0);
738 // Drop i when it is mostly inside a strictly larger code box j.
739 let overlap = inter(&kept[i], kept[j].l, kept[j].t, kept[j].r, kept[j].b);
740 if aj > ai && overlap / ai > 0.7 {
741 drop[i] = true;
742 break;
743 }
744 }
745 }
746 let mut keep = drop.iter();
747 kept.retain(|_| !*keep.next().unwrap());
748}
749
750/// Fraction of the page's non-empty text cells that some detected region
751/// claims (>0.2 intersection-over-self, docling's assignment rule). 1.0 for a
752/// page without text cells.
753///
754/// The int8-layout guard keys off this: a dense digital page whose detections
755/// cover almost none of its text is the signature of quantized confidences
756/// flipping under the 0.5 label thresholds on this CPU's kernels — not of a
757/// genuinely empty layout — and is worth re-running on the fp32 graph.
758pub fn layout_cell_coverage(regions: &[Region], cells: &[TextCell]) -> f32 {
759 let mut total = 0usize;
760 let mut covered = 0usize;
761 for c in cells {
762 if c.text.trim().is_empty() {
763 continue;
764 }
765 total += 1;
766 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
767 if regions
768 .iter()
769 .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
770 {
771 covered += 1;
772 }
773 }
774 if total == 0 {
775 1.0
776 } else {
777 covered as f32 / total as f32
778 }
779}
780
781/// Append `text` regions for cells the layout left uncovered ("orphan cells"),
782/// the way docling's `LayoutPostprocessor` does (`create_orphan_clusters`): any
783/// non-empty cell that no kept region covers (>50% of the cell's area) becomes a
784/// text region of its own, so text the detector missed (a stray `.`, a small
785/// label) is still emitted instead of silently dropped. Adjacent orphan cells on a
786/// line are merged so a missed paragraph doesn't shatter into one block per line.
787pub fn add_orphan_regions(regions: &mut Vec<Region>, cells: &[TextCell]) {
788 // docling assigns each cell to its single best-overlapping cluster at
789 // intersection-over-self > 0.2 and serializes exactly the assigned cells —
790 // and since [`region_texts_exclusive`] now emits under that very rule, the
791 // claim test here matches it: any cell over 0.2 will actually render in
792 // its best region, everything else becomes an orphan. Completeness by
793 // construction, with no (0.2, 0.5] hole (the old > 0.5 serializer needed
794 // the claim test raised to > 0.5 to keep right_to_left_03's `20300` from
795 // vanishing; the exclusive port closes that structurally).
796 //
797 // Only *regular* clusters claim cells: docling's `_find_unassigned_cells`
798 // walks `regular_clusters` alone, so a cell under a `picture` or a wrapper
799 // (`table`/`document_index`/`form`/`key_value_region`) that no regular
800 // cluster covers still becomes an orphan text cluster (#165). The orphans
801 // that end up *fully* inside the special are re-dropped by
802 // [`drop_contained_regulars`] (docling's Markdown drops them the same way
803 // — a picture's children never reach its `MarkdownPictureSerializer`
804 // output, a table's text renders through the reconstructed grid). The
805 // observable fix is the border-straddlers: a line only partially under a
806 // figure box used to lose its cells to the picture's 0.2 claim and vanish
807 // — now it forms an orphan region and is emitted, as docling does.
808 let assigned = |c: &TextCell| {
809 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
810 regions
811 .iter()
812 .filter(|r| r.label != "picture" && !is_wrapper(r.label))
813 .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
814 };
815 // Collect orphan cells (non-empty, unassigned), in page order.
816 let mut orphans: Vec<&TextCell> = cells
817 .iter()
818 .filter(|c| !c.text.trim().is_empty() && !assigned(c))
819 .collect();
820 if orphans.is_empty() {
821 return;
822 }
823 orphans.sort_by(|a, b| a.t.total_cmp(&b.t).then(a.l.total_cmp(&b.l)));
824 // Merge cells that sit on the same line and nearly touch into one region, so a
825 // dropped multi-word line stays one block (docling's refinement merges these).
826 let mut merged: Vec<Region> = Vec::new();
827 for c in orphans {
828 let h = (c.b - c.t).abs().max(1.0);
829 if let Some(last) = merged.last_mut() {
830 let same_line = (last.t - c.t).abs() < h * 0.5;
831 let touching = c.l <= last.r + h && c.l >= last.l - h;
832 if same_line && touching {
833 last.l = last.l.min(c.l);
834 last.r = last.r.max(c.r);
835 last.t = last.t.min(c.t);
836 last.b = last.b.max(c.b);
837 continue;
838 }
839 }
840 merged.push(Region {
841 label: "text",
842 score: 0.0,
843 l: c.l,
844 t: c.t,
845 r: c.r,
846 b: c.b,
847 });
848 }
849 regions.extend(merged);
850}
851
852/// Demote a `picture` region that is really a **text panel** — a paragraph block
853/// the layout model boxed as a figure because it is typeset on a colored
854/// background (terms-and-conditions callouts, quote boxes) — into ordinary
855/// `text` regions, one per paragraph, so its words are read instead of shipped
856/// as pixels. docling loses this text the same way (cells assigned to a picture
857/// cluster are never serialized); this is a deliberate improvement, not parity.
858///
859/// The gate is conservative so a genuine figure keeps its crop: the region must
860/// contain at least three text lines whose median width spans most of the panel
861/// (axis labels and chat bubbles are narrow and varied) and whose cells cover a
862/// substantial fraction of its area (a photo or chart with sparse labels does
863/// not). Paragraph boundaries are re-derived from the line pitch: a vertical gap
864/// clearly larger than the panel's own leading starts a new `text` region, so
865/// the panel doesn't collapse into one giant paragraph.
866///
867/// Works on any cell source — the digital text layer or OCR lines recognized
868/// from the picture crop — so the native and browser paths, with or without
869/// force-OCR, demote identically.
870pub fn recover_text_panels(regions: &mut Vec<Region>, cells: &[TextCell]) {
871 // A *captioned* picture is a genuine figure whatever it contains — the
872 // corpus is full of document screenshots ("Figure 3: …" above a page
873 // image) that are exactly as dense and wide as a text panel. Only an
874 // uncaptioned picture is a demotion candidate.
875 let captioned: Vec<bool> = regions
876 .iter()
877 .map(|r| {
878 r.label == "picture"
879 && regions.iter().any(|c| {
880 c.label == "caption" && c.r.min(r.r) - c.l.max(r.l) > 0.0 && {
881 let gap = if c.t >= r.b {
882 c.t - r.b
883 } else if r.t >= c.b {
884 r.t - c.b
885 } else {
886 f32::MAX // vertically overlapping: not a caption
887 };
888 gap <= 25.0
889 }
890 })
891 })
892 .collect();
893 let mut out: Vec<Region> = Vec::with_capacity(regions.len());
894 // Synthesized paragraphs and the demoted panels' boxes are kept separate
895 // from `out` until the end: the dedup filter below must not confuse a
896 // paragraph we just built with a pre-existing region inside the panel.
897 let mut demoted_paras: Vec<Region> = Vec::new();
898 let mut demoted_boxes: Vec<(f32, f32, f32, f32)> = Vec::new();
899 for (i, r) in regions.drain(..).enumerate() {
900 if r.label != "picture" || captioned[i] {
901 out.push(r);
902 continue;
903 }
904 let inside: Vec<&TextCell> = cells
905 .iter()
906 .filter(|c| {
907 !c.text.trim().is_empty() && {
908 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
909 inter(&r, c.l, c.t, c.r, c.b) / ca > 0.5
910 }
911 })
912 .collect();
913 // Group the contained cells into lines by vertical overlap (the same
914 // rule region_text orders by), tracking each line's union box.
915 let mut lines: Vec<(f32, f32, f32, f32)> = Vec::new(); // (t, b, l, r)
916 for c in &inside {
917 let (ct, cb) = (c.t.min(c.b), c.t.max(c.b));
918 match lines.iter_mut().find(|(lt, lb, _, _)| {
919 let ov = cb.min(*lb) - ct.max(*lt);
920 ov > 0.5 * (cb - ct).min(*lb - *lt).max(1.0)
921 }) {
922 Some((lt, lb, ll, lr)) => {
923 *lt = lt.min(ct);
924 *lb = lb.max(cb);
925 *ll = ll.min(c.l);
926 *lr = lr.max(c.r);
927 }
928 None => lines.push((ct, cb, c.l, c.r)),
929 }
930 }
931 if lines.len() < 3 {
932 out.push(r);
933 continue;
934 }
935 let panel_w = (r.r - r.l).max(1.0);
936 let coverage = inside.iter().map(|c| area(c.l, c.t, c.r, c.b)).sum::<f32>()
937 / area(r.l, r.t, r.r, r.b).max(1.0);
938 let mut widths: Vec<f32> = lines.iter().map(|(_, _, l, rr)| rr - l).collect();
939 widths.sort_by(f32::total_cmp);
940 // A figure's text is ragged: a title line, small axis/tick labels, and
941 // OCR boxes over the plot area come out at wildly different heights,
942 // whereas a real text panel is set in one face with constant leading.
943 // Require near-uniform line heights (median absolute deviation ≤ 35%
944 // of the median) so an uncaptioned chart keeps its crop even when its
945 // labels are dense enough to pass the coverage gate (#173) — garbled
946 // OCR of its bars is not content.
947 let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
948 heights.sort_by(f32::total_cmp);
949 let h_med = heights[heights.len() / 2].max(1.0);
950 let mut devs: Vec<f32> = heights.iter().map(|h| (h - h_med).abs()).collect();
951 devs.sort_by(f32::total_cmp);
952 let uniform = devs[devs.len() / 2] <= 0.35 * h_med;
953 let text_panel = coverage >= 0.2 && widths[widths.len() / 2] >= 0.45 * panel_w && uniform;
954 if !text_panel {
955 out.push(r);
956 continue;
957 }
958 lines.sort_by(|a, b| a.0.total_cmp(&b.0));
959 let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
960 heights.sort_by(f32::total_cmp);
961 let h = heights[heights.len() / 2].max(1.0);
962 let mut gaps: Vec<f32> = lines
963 .windows(2)
964 .map(|w| (w[1].0 - w[0].1).max(0.0))
965 .collect();
966 gaps.sort_by(f32::total_cmp);
967 let leading = if gaps.is_empty() {
968 0.0
969 } else {
970 gaps[gaps.len() / 2]
971 };
972 let brk = (1.8 * leading).max(0.75 * h);
973 let mut para: Option<(f32, f32, f32, f32)> = None; // (l, t, r, b) union
974 for (t, b, l, rr) in &lines {
975 match &mut para {
976 Some((pl, _, pr, pb)) if *t - *pb <= brk => {
977 *pl = pl.min(*l);
978 *pr = pr.max(*rr);
979 *pb = pb.max(*b);
980 }
981 _ => {
982 if let Some((pl, pt, pr, pb)) = para.take() {
983 demoted_paras.push(Region {
984 label: "text",
985 score: r.score,
986 l: pl,
987 t: pt,
988 r: pr,
989 b: pb,
990 });
991 }
992 para = Some((*l, *t, *rr, *b));
993 }
994 }
995 }
996 if let Some((pl, pt, pr, pb)) = para {
997 demoted_paras.push(Region {
998 label: "text",
999 score: r.score,
1000 l: pl,
1001 t: pt,
1002 r: pr,
1003 b: pb,
1004 });
1005 }
1006 demoted_boxes.push((r.l, r.t, r.r, r.b));
1007 }
1008 // The paragraphs are rebuilt from *all* of the panel's cells, so any
1009 // surviving text region inside a demoted panel (an orphan cluster or a
1010 // layout-detected fragment — pictures no longer swallow them, #165) would
1011 // say the same words twice. Consume those; wrappers and pictures stay.
1012 if !demoted_boxes.is_empty() {
1013 out.retain(|r| {
1014 r.label == "picture" || is_wrapper(r.label) || {
1015 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
1016 !demoted_boxes
1017 .iter()
1018 .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.5)
1019 }
1020 });
1021 }
1022 // docling's "Remove regular clusters that are included in wrappers" (a
1023 // regular > 80 % inside a table is absorbed by it) already ran as
1024 // [`drop_contained_regulars`], but before this demotion created new
1025 // regulars. Apply it to them too: a panel that coincides with a table (a
1026 // dense data table detected as picture 0.80 and table 0.62 on one box;
1027 // `_handle_cross_type_overlaps` keeps both once the picture is ≥ 0.1 more
1028 // confident) rebuilds the table's words as a paragraph the grid already
1029 // renders. A panel inside another picture is left as it was.
1030 demoted_paras.retain(|p| {
1031 let pa = area(p.l, p.t, p.r, p.b).max(1.0);
1032 !out.iter()
1033 .any(|s| is_table_like(s.label) && inter(p, s.l, s.t, s.r, s.b) / pa > 0.8)
1034 });
1035 out.extend(demoted_paras);
1036 *regions = out;
1037}
1038
1039/// Drop a `picture` detection covering more than 90 % of the page — docling's
1040/// `LayoutPostprocessor._process_special_clusters` "Filter out full-page
1041/// pictures" (upstream since 2.15), applied to the thresholded detections
1042/// before overlap resolution. A box that big is the page itself, not a figure
1043/// on it: the layout model emits one for a whole-page diagram (a LaTeX figure
1044/// PDF cropped to its drawing), a plate, or a scan, and keeping it would swallow
1045/// every text cell on the page as picture children — the diagram's labels and
1046/// caption vanish behind a lone `<!-- image -->`, where docling reads them out as
1047/// text. `page_w`/`page_h` is the display-frame page box.
1048pub fn drop_full_page_pictures(regions: &mut Vec<Region>, page_w: f32, page_h: f32) {
1049 let page_area = (page_w * page_h).max(1.0);
1050 regions.retain(|r| r.label != "picture" || area(r.l, r.t, r.r, r.b) / page_area <= 0.90);
1051}
1052
1053/// Drop a `picture` detection that is a small, empty, low-confidence margin box on
1054/// a **text page** — a false positive the RT-DETR layout sometimes emits (e.g.
1055/// `right_to_left_02`'s phantom right-column picture, score 0.40); docling does not
1056/// emit it. The gate is deliberately narrow so a genuine figure is never dropped:
1057/// (1) only on pages with a digital text layer — image/scanned/figure pages have
1058/// no `cells` yet at this point (OCR runs later), so their pictures, which *are*
1059/// the content, are kept; (2) only a box covering < 25 % of the page (a margin
1060/// artifact, not a dominant figure); (3) only when it contains no text and scores
1061/// below 0.5 (real empty figures in the corpus all score ≥ 0.86).
1062pub fn drop_false_pictures(
1063 regions: &mut Vec<Region>,
1064 cells: &[TextCell],
1065 page_w: f32,
1066 page_h: f32,
1067) {
1068 if cells.iter().all(|c| c.text.trim().is_empty()) {
1069 return; // no digital text layer (image/scanned page) — keep all pictures
1070 }
1071 // A text-document page carries several text-bearing non-picture regions (so a
1072 // spurious margin picture is clearly extra). A slide / figure page has at most
1073 // one — there the picture is the content, so never drop it.
1074 let content_regions = regions
1075 .iter()
1076 .filter(|r| r.label != "picture" && !region_text(r, cells).trim().is_empty())
1077 .count();
1078 if content_regions < 2 {
1079 return;
1080 }
1081 let page_area = (page_w * page_h).max(1.0);
1082 regions.retain(|r| {
1083 if r.label != "picture" || r.score >= 0.5 {
1084 return true;
1085 }
1086 if area(r.l, r.t, r.r, r.b) / page_area >= 0.25 {
1087 return true; // a dominant figure, not a margin artifact
1088 }
1089 // Keep it if any text cell falls mostly inside (a real captioned/labelled
1090 // figure); drop only the genuinely empty low-confidence boxes.
1091 cells.iter().any(|c| {
1092 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1093 !c.text.trim().is_empty() && inter(r, c.l, c.t, c.r, c.b) / ca > 0.5
1094 })
1095 });
1096}
1097
1098/// A small digit-only region in the top/bottom margin: a page number. docling
1099/// emits `right_to_left_02`'s bottom `11` as the page's *first* text item (its
1100/// reading-order model floats the page number to the front), whereas our
1101/// position-based ordering would place a bottom region last.
1102fn is_page_number(region: &Region, cells: &[TextCell], page_h: f32) -> bool {
1103 let t = region_text(region, cells);
1104 let t = t.trim();
1105 !t.is_empty()
1106 && t.chars().all(|c| c.is_ascii_digit())
1107 && (region.b - region.t).abs() < 30.0
1108 && (region.t < page_h * 0.12 || region.b > page_h * 0.88)
1109}
1110
1111/// docling's `form` / `key_value_region` *containers* (2.123, docling#4064):
1112/// every region sitting > 0.8 inside one — text, list items, and since #4064
1113/// tables and pictures too — is that container's child. Children are
1114/// reading-ordered among themselves and emitted as one block where the
1115/// container falls in the page's top-level order (a `form_area` /
1116/// `key_value_area` group upstream), instead of interleaving with the text
1117/// around the form. A child inside several containers belongs to the smallest
1118/// (then most confident, then first); a container with children shrinks to
1119/// their union for the top-level ordering, like upstream's bbox adjustment.
1120///
1121/// The containers themselves are still not emitted (`is_skipped`), so the
1122/// Markdown is exactly upstream's — a group prints only its children.
1123///
1124/// `cids` are the items' positions in docling's assembly order
1125/// ([`cluster_cids`]) — the reading-order predictor's same-row rule (#424)
1126/// pairs consecutive ones, within the top level and within each container.
1127fn order_with_containers<T: Clone>(
1128 items: &mut Vec<T>,
1129 cids: &[usize],
1130 page_w: f32,
1131 page_h: f32,
1132 reg: impl Fn(&T) -> &Region,
1133) {
1134 let is_container = |r: &Region| matches!(r.label, "form" | "key_value_region");
1135 let containers: Vec<usize> = (0..items.len())
1136 .filter(|&i| is_container(reg(&items[i])))
1137 .collect();
1138 if containers.is_empty() {
1139 order_regions(items, cids, page_w, page_h, reg);
1140 return;
1141 }
1142 // Parent container per item (containers never nest in each other here —
1143 // upstream assigns regulars and tables/pictures only).
1144 let mut parent: Vec<Option<usize>> = vec![None; items.len()];
1145 for i in 0..items.len() {
1146 let r = reg(&items[i]);
1147 if is_container(r) {
1148 continue;
1149 }
1150 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
1151 let mut best: Option<(usize, f32, f32)> = None; // (idx, area, -score)
1152 for &c in &containers {
1153 let cr = reg(&items[c]);
1154 if inter(r, cr.l, cr.t, cr.r, cr.b) / ra > 0.8 {
1155 let key = (area(cr.l, cr.t, cr.r, cr.b), -cr.score);
1156 if best.is_none_or(|(_, a, s)| key.0 < a || (key.0 == a && key.1 < s)) {
1157 best = Some((c, key.0, key.1));
1158 }
1159 }
1160 }
1161 parent[i] = best.map(|(c, _, _)| c);
1162 }
1163 // Top-level pass: non-children plus the containers, the latter shrunk to
1164 // their children's union.
1165 let mut top: Vec<(usize, Region)> = Vec::new();
1166 for i in 0..items.len() {
1167 if parent[i].is_some() {
1168 continue;
1169 }
1170 let mut r = reg(&items[i]).clone();
1171 if is_container(&r) {
1172 let kids: Vec<&Region> = (0..items.len())
1173 .filter(|&k| parent[k] == Some(i))
1174 .map(|k| reg(&items[k]))
1175 .collect();
1176 if !kids.is_empty() {
1177 r.l = kids.iter().map(|k| k.l).fold(f32::INFINITY, f32::min);
1178 r.t = kids.iter().map(|k| k.t).fold(f32::INFINITY, f32::min);
1179 r.r = kids.iter().map(|k| k.r).fold(f32::NEG_INFINITY, f32::max);
1180 r.b = kids.iter().map(|k| k.b).fold(f32::NEG_INFINITY, f32::max);
1181 }
1182 }
1183 top.push((i, r));
1184 }
1185 let top_cids: Vec<usize> = top.iter().map(|(i, _)| cids[*i]).collect();
1186 order_regions(&mut top, &top_cids, page_w, page_h, |it| &it.1);
1187 let mut out: Vec<T> = Vec::with_capacity(items.len());
1188 for (i, _) in top {
1189 if is_container(reg(&items[i])) {
1190 let kid_idx: Vec<usize> = (0..items.len()).filter(|&k| parent[k] == Some(i)).collect();
1191 let mut kids: Vec<T> = kid_idx.iter().map(|&k| items[k].clone()).collect();
1192 let kid_cids: Vec<usize> = kid_idx.iter().map(|&k| cids[k]).collect();
1193 order_regions(&mut kids, &kid_cids, page_w, page_h, ®);
1194 out.push(items[i].clone());
1195 out.extend(kids);
1196 } else {
1197 out.push(items[i].clone());
1198 }
1199 }
1200 *items = out;
1201}
1202
1203/// Furniture / not-yet-emitted labels.
1204fn is_skipped(label: &str) -> bool {
1205 matches!(
1206 label,
1207 "page_header" | "page_footer" | "form" | "key_value_region"
1208 )
1209}
1210
1211/// Reading-order sort of a page's regions, via the ported rule-based
1212/// [`reading_order`](crate::reading_order) predictor (docling's
1213/// `ReadingOrderPredictor`): an up/down geometry graph with same-row links
1214/// between `cids`-consecutive elements (#424), horizontal dilation and a
1215/// depth-first traversal, with `page_header`/`page_footer` ordered as their own
1216/// groups (first/last) as docling does.
1217fn order_regions<T: Clone>(
1218 items: &mut Vec<T>,
1219 cids: &[usize],
1220 page_w: f32,
1221 page_h: f32,
1222 reg: impl Fn(&T) -> &Region,
1223) {
1224 let boxes: Vec<(f32, f32, f32, f32)> = items
1225 .iter()
1226 .map(|it| {
1227 let r = reg(it);
1228 (r.l, r.t, r.r, r.b)
1229 })
1230 .collect();
1231 let is_header: Vec<bool> = items
1232 .iter()
1233 .map(|it| reg(it).label == "page_header")
1234 .collect();
1235 let is_footer: Vec<bool> = items
1236 .iter()
1237 .map(|it| reg(it).label == "page_footer")
1238 .collect();
1239 let order =
1240 crate::reading_order::order_page(&boxes, cids, &is_header, &is_footer, page_w, page_h);
1241 *items = order.iter().map(|&i| items[i].clone()).collect();
1242}
1243
1244/// docling's assembly order of a page's clusters (`LayoutPostprocessor`'s
1245/// final `_sort_clusters(mode="id")`, #424): each region's rank when sorted by
1246/// its first source cell, then by top edge, then left edge; a region with no
1247/// cells sorts after every one that has some. docling numbers its page
1248/// elements (`cid`) in this order, and the reading-order predictor's same-row
1249/// rule pairs elements with consecutive numbers, so the ranks are what
1250/// [`order_with_containers`] hands the predictor.
1251///
1252/// A regular region's first cell is the smallest index among the cells it
1253/// claims. A table, picture or container has no cells of its own upstream
1254/// either — its cells are its *children's*: the regular clusters > 0.8 inside
1255/// it, and upstream every cell no regular cluster claimed is an orphan cluster
1256/// of its own, so a table's interior text (which no regular cluster claims)
1257/// reaches the table through those orphans. Here that is the cells > 0.8
1258/// inside the region plus the claimed cells of the regular regions > 0.8
1259/// inside it. Without the interior cells every table would sort last, and two
1260/// side-by-side tables would then be consecutive and row-linked — reading the
1261/// right table's caption ahead of the left column's headings (2206 page 8).
1262pub fn cluster_cids(regions: &[Region], cells: &[TextCell]) -> Vec<usize> {
1263 let owned = assign_cells(regions, cells);
1264 let first_cell: Vec<usize> = regions
1265 .iter()
1266 .enumerate()
1267 .map(|(i, r)| {
1268 if claims_cells(r) {
1269 return owned[i].iter().copied().min().unwrap_or(usize::MAX);
1270 }
1271 let interior = cells
1272 .iter()
1273 .enumerate()
1274 .filter(|(_, c)| {
1275 !c.text.trim().is_empty()
1276 && inter(r, c.l, c.t, c.r, c.b) / area(c.l, c.t, c.r, c.b).max(1.0) > 0.8
1277 })
1278 .map(|(ci, _)| ci)
1279 .min();
1280 let children = regions
1281 .iter()
1282 .enumerate()
1283 .filter(|(j, child)| {
1284 *j != i && claims_cells(child) && {
1285 let ca = area(child.l, child.t, child.r, child.b).max(1.0);
1286 inter(r, child.l, child.t, child.r, child.b) / ca > 0.8
1287 }
1288 })
1289 .filter_map(|(j, _)| owned[j].iter().copied().min())
1290 .min();
1291 interior
1292 .into_iter()
1293 .chain(children)
1294 .min()
1295 .unwrap_or(usize::MAX)
1296 })
1297 .collect();
1298 let mut by_source: Vec<usize> = (0..regions.len()).collect();
1299 // Stable, like Python's `sorted`: full ties keep the layout order.
1300 by_source.sort_by(|&a, &b| {
1301 first_cell[a]
1302 .cmp(&first_cell[b])
1303 .then(regions[a].t.total_cmp(®ions[b].t))
1304 .then(regions[a].l.total_cmp(®ions[b].l))
1305 });
1306 let mut cids = vec![0; regions.len()];
1307 for (rank, &i) in by_source.iter().enumerate() {
1308 cids[i] = rank;
1309 }
1310 cids
1311}
1312
1313/// Clean a region's assembled text: undo soft-hyphen line wraps, map curly
1314/// quotes and the ellipsis to ASCII (matching docling), and collapse runs of
1315/// whitespace. pdfium emits the line-wrap hyphen as U+0002 in this corpus
1316/// (U+00AD elsewhere), so `word\u{2} continuation` is one hyphenated word —
1317/// drop the hyphen + the joining space and merge (`com\u{2} pact` → `compact`,
1318/// `end-to\u{2} end` → `end-toend`), exactly as docling does.
1319///
1320/// Token spacing is otherwise left as the geometric join produced it. We do not
1321/// tighten punctuation spacing: docling preserves the PDF's own spaces (it keeps
1322/// `{ ahn }`, `Name 1 .`, `[ 9 ]`), and a geometric gap heuristic diverges from
1323/// it more than a plain single-space join does.
1324/// An ordered-list enumeration marker at the start of a list item: leading ASCII
1325/// digits followed by `.`, e.g. `1. Undo/Redo` → `(1, "Undo/Redo")`. Returns
1326/// `None` when the text doesn't start with `digits.`.
1327/// docling's `ListItemMarkerProcessor` bullet patterns
1328/// (`docling/models/postprocessing/list_marker_processor.py`), one glyph each.
1329const LIST_BULLET_MARKERS: &str = "\u{2022}\u{2023}\u{25E6}\u{2043}\u{204C}\u{204D}\u{2219}\u{25AA}\u{25AB}\u{25CF}\u{25CB}-*+•·‣⁃►▶▸➤➢✓✔✗✘";
1330
1331/// docling's numbered-marker patterns as byte-length scanners over the start
1332/// of the text, in its first-wins order (the compound ones first, as they are
1333/// the more specific). Each returns the marker's candidate lengths, longest
1334/// (greedy) first — the alternatives Python's regex would backtrack through
1335/// before the `\s(.+)` tail. Hand-rolled rather than `regex` because this file
1336/// is part of the `pdf-text` (wasm) build, where the `regex` crate is not.
1337/// `\d` is Unicode-aware in Python, hence `char::is_numeric`; the letters are
1338/// ASCII classes in both.
1339const LIST_NUMBERED_MARKERS: &[fn(&str) -> Vec<usize>] = &[
1340 // `\d+(?:\.\d+)+\.?` — 1.1 1.2.3 1.1.
1341 |s| {
1342 let mut i = digits(s, 0);
1343 if i == 0 {
1344 return Vec::new();
1345 }
1346 let mut groups = 0;
1347 while s[i..].starts_with('.') && digits(s, i + 1) > i + 1 {
1348 i = digits(s, i + 1);
1349 groups += 1;
1350 }
1351 if groups == 0 {
1352 return Vec::new();
1353 }
1354 if s[i..].starts_with('.') {
1355 vec![i + 1, i]
1356 } else {
1357 vec![i]
1358 }
1359 },
1360 // `\d+\.?[a-zA-Z]\.` — 9a. 3.a.
1361 |s| digits_dot_letter(s, 0, '.').into_iter().collect(),
1362 // `\d+\.?[a-zA-Z]\)` — 9a) 3.a)
1363 |s| digits_dot_letter(s, 0, ')').into_iter().collect(),
1364 // `\(\d+\.?[a-zA-Z]\)` — (9a) (3.a)
1365 |s| {
1366 if !s.starts_with('(') {
1367 return Vec::new();
1368 }
1369 digits_dot_letter(s, 1, ')').into_iter().collect()
1370 },
1371 // `\d+\.` — 1. 2. 3.
1372 |s| digits_then(s, 0, '.').into_iter().collect(),
1373 // `\d+\)` — 1) 2) 3)
1374 |s| digits_then(s, 0, ')').into_iter().collect(),
1375 // `\(\d+\)` — (1) (2) (3)
1376 |s| {
1377 if !s.starts_with('(') {
1378 return Vec::new();
1379 }
1380 digits_then(s, 1, ')').into_iter().collect()
1381 },
1382 // `\[\d+\]` — [1] [2] [3]
1383 |s| {
1384 if !s.starts_with('[') {
1385 return Vec::new();
1386 }
1387 digits_then(s, 1, ']').into_iter().collect()
1388 },
1389 // `[ivxlcdm]+\.` — i. ii. iii.
1390 |s| class_run_then(s, "ivxlcdm", '.').into_iter().collect(),
1391 // `[IVXLCDM]+\.` — I. II. III.
1392 |s| class_run_then(s, "IVXLCDM", '.').into_iter().collect(),
1393 // `[a-z]\.` / `[A-Z]\.` / `[a-z]\)` / `[A-Z]\)`
1394 |s| {
1395 letter_then(s, char::is_ascii_lowercase, '.')
1396 .into_iter()
1397 .collect()
1398 },
1399 |s| {
1400 letter_then(s, char::is_ascii_uppercase, '.')
1401 .into_iter()
1402 .collect()
1403 },
1404 |s| {
1405 letter_then(s, char::is_ascii_lowercase, ')')
1406 .into_iter()
1407 .collect()
1408 },
1409 |s| {
1410 letter_then(s, char::is_ascii_uppercase, ')')
1411 .into_iter()
1412 .collect()
1413 },
1414];
1415
1416/// Byte offset just past the run of `\d` characters starting at `from`
1417/// (`from` itself when there is none).
1418fn digits(s: &str, from: usize) -> usize {
1419 s[from..]
1420 .char_indices()
1421 .find(|(_, c)| !c.is_numeric())
1422 .map_or(s.len(), |(i, _)| from + i)
1423}
1424
1425/// `\d+<close>` from `from`: the length through `close`, if it matches.
1426fn digits_then(s: &str, from: usize, close: char) -> Option<usize> {
1427 let end = digits(s, from);
1428 (end > from && s[end..].starts_with(close)).then(|| end + close.len_utf8())
1429}
1430
1431/// `\d+\.?[a-zA-Z]<close>` from `from`.
1432fn digits_dot_letter(s: &str, from: usize, close: char) -> Option<usize> {
1433 let mut i = digits(s, from);
1434 if i == from {
1435 return None;
1436 }
1437 if s[i..].starts_with('.') {
1438 i += 1;
1439 }
1440 let letter = s[i..].chars().next().filter(char::is_ascii_alphabetic)?;
1441 i += letter.len_utf8();
1442 s[i..].starts_with(close).then(|| i + close.len_utf8())
1443}
1444
1445/// `[<class>]+<close>` at the start.
1446fn class_run_then(s: &str, class: &str, close: char) -> Option<usize> {
1447 let end = s
1448 .char_indices()
1449 .find(|(_, c)| !class.contains(*c))
1450 .map_or(s.len(), |(i, _)| i);
1451 (end > 0 && s[end..].starts_with(close)).then(|| end + close.len_utf8())
1452}
1453
1454/// `[<letter class>]<close>` at the start.
1455fn letter_then(s: &str, class: fn(&char) -> bool, close: char) -> Option<usize> {
1456 let letter = s.chars().next().filter(class)?;
1457 let i = letter.len_utf8();
1458 s[i..].starts_with(close).then(|| i + close.len_utf8())
1459}
1460
1461/// docling's `ListItemMarkerProcessor.process_list_item`: the item's original
1462/// text is matched against `^(marker)\s(.+)` (DOTALL) for the bullet patterns,
1463/// then the numbered ones in order; a hit splits it into
1464/// `(marker, text, enumerated)`. `\s` is one Unicode whitespace character and
1465/// `.+` everything after it, which must be non-empty.
1466fn split_list_marker(text: &str) -> Option<(&str, &str, bool)> {
1467 let tail_after = |len: usize| -> Option<&str> {
1468 let ws = text[len..].chars().next()?;
1469 if !ws.is_whitespace() {
1470 return None;
1471 }
1472 let rest = &text[len + ws.len_utf8()..];
1473 (!rest.is_empty()).then_some(rest)
1474 };
1475 let first = text.chars().next()?;
1476 if LIST_BULLET_MARKERS.contains(first) {
1477 if let Some(rest) = tail_after(first.len_utf8()) {
1478 return Some((&text[..first.len_utf8()], rest, false));
1479 }
1480 }
1481 for matcher in LIST_NUMBERED_MARKERS {
1482 for len in matcher(text) {
1483 if let Some(rest) = tail_after(len) {
1484 return Some((&text[..len], rest, true));
1485 }
1486 }
1487 }
1488 None
1489}
1490
1491/// A PDF `list_item` region as docling emits it: `ListItemMarkerProcessor`
1492/// splits the marker off (see [`split_list_marker`]), and docling-core's
1493/// Markdown list serializer (default `orig_list_item_marker_mode = AUTO`,
1494/// `ensure_valid_list_item_marker`) prints it as `N. text` for an `N.` marker
1495/// (`case_already_valid`: the marker verbatim); as `- text` for a bullet glyph
1496/// (no letter or digit in the marker: only the `-` the serializer adds); and
1497/// as `- a) text` / `- 1.2 text` / `- [3] text` for any other marker
1498/// (`case_auto`: the serializer's `-`, then the original marker) — spelled
1499/// here as a bullet item whose text carries the marker, the way the DOCX and
1500/// DOC backends already spell theirs. An item without a recognizable marker is
1501/// a plain bullet. The symbol-font bullets docling-parse filters out of its
1502/// cells (`•◦▪·*` glued to the text) are stripped before the match, as before.
1503fn list_item_node(text: &str, loc: [u16; 4], first_in_list: bool) -> Node {
1504 // docling's match runs on the text as docling-parse hands it over; the
1505 // glued symbol-font bullets it never sees are stripped only when the raw
1506 // text matches no marker (`•Text` → `Text`, but `• Text` → marker `•`).
1507 let stripped = text
1508 .trim_start_matches(['•', '◦', '▪', '·', '*'])
1509 .trim_start();
1510 let split = split_list_marker(text).or_else(|| split_list_marker(stripped));
1511 let bullet = |text: String, marker: &str| Node::ListItem {
1512 ordered: false,
1513 number: 0,
1514 first_in_list,
1515 text: md_escape(&text),
1516 level: 0,
1517 // docling keeps the marker as the DocLang list marker
1518 // (`<ldiv><marker>·</marker></ldiv>`); Markdown prints its own `-`.
1519 marker: Some(marker.to_string()),
1520 location: Some(loc),
1521 dclx: None,
1522 href: None,
1523 layer: None,
1524 };
1525 // docling-core's `case_already_valid` is a *full* `\d+\.` match: `3.a.`
1526 // and `1.2` are `case_auto` markers (`- 3.a. text`), not item numbers.
1527 let is_number_dot = |m: &str| {
1528 m.strip_suffix('.')
1529 .is_some_and(|d| !d.is_empty() && d.chars().all(char::is_numeric))
1530 };
1531 match split {
1532 Some((marker, body, true)) if is_number_dot(marker) => {
1533 let number = parse_ordered_marker(marker).map_or(0, |(n, _)| n);
1534 Node::ListItem {
1535 ordered: true,
1536 number,
1537 first_in_list,
1538 text: md_escape(body),
1539 level: 0,
1540 marker: Some(marker.to_string()),
1541 location: Some(loc),
1542 dclx: None,
1543 href: None,
1544 layer: None,
1545 }
1546 }
1547 // `case_auto`: a marker holding a letter or digit rides in the text.
1548 Some((marker, body, true)) => bullet(format!("{marker} {body}"), marker),
1549 Some((marker, body, false)) => bullet(body.to_string(), marker),
1550 None => bullet(stripped.to_string(), "·"),
1551 }
1552}
1553
1554fn parse_ordered_marker(s: &str) -> Option<(u64, String)> {
1555 let digits: String = s.chars().take_while(|c| c.is_ascii_digit()).collect();
1556 if digits.is_empty() {
1557 return None;
1558 }
1559 let rest = s[digits.len()..].strip_prefix('.')?;
1560 let number = digits.parse().ok()?;
1561 Some((number, rest.trim_start().to_string()))
1562}
1563
1564/// Escape markdown special characters the way docling-core's markdown serializer
1565/// does (`markdown.py` post_process): `_` → `\_`, then HTML-escape `&`, `<`, `>`
1566/// (quote=False, so quotes are left). Applied to prose (headings, list items,
1567/// paragraphs); code blocks, the formula placeholder, and table cells are left raw.
1568fn md_escape(text: &str) -> String {
1569 text.replace('_', "\\_")
1570 .replace('&', "&")
1571 .replace('<', "<")
1572 .replace('>', ">")
1573}
1574
1575fn clean_text(text: &str) -> String {
1576 // Typographic-quote normalization follows docling-parse's sanitizer table
1577 // (`pdf_sanitators/constants.h`): every curly quote — single *and double* —
1578 // becomes the ASCII apostrophe `'`, and `‚` a comma. A `"` in docling's
1579 // output only ever comes from a literal `quotedbl` glyph, never from `“ ”`
1580 // (2206's `'text in the wild"` pairs a curly open with a literal-quote
1581 // close). This replaces an earlier Hangul-only special case that patched
1582 // one symptom of mapping `“ ”` to `"`.
1583 let replaced = text
1584 .replace("\u{2} ", "")
1585 .replace("\u{ad} ", "")
1586 .replace(['\u{2}', '\u{ad}'], "") // any stray wrap hyphens not at a join
1587 .replace(
1588 [
1589 '\u{2018}', '\u{2019}', '\u{201b}', '\u{201c}', '\u{201d}', '\u{201e}', '\u{201f}',
1590 ],
1591 "'",
1592 ) // ‘ ’ ‛ “ ” „ ‟ → '
1593 .replace('\u{201a}', ",") // ‚ → ,
1594 .replace(
1595 [
1596 '\u{2010}', '\u{2011}', '\u{2012}', '\u{2013}', '\u{2014}', '\u{2015}', '\u{2212}',
1597 ],
1598 "-",
1599 ) // hyphen/dash family → -
1600 .replace('\u{2044}', "/") // ⁄ fraction slash → /
1601 .replace('\u{2022}', "\u{b7}") // • → · (docling never emits •; inline CCS-concept separators)
1602 .replace('\u{2026}', "..."); // … → ...
1603 // The docling-parse sanitizer already placed the correct spacing (e.g.
1604 // justified double spaces); preserve internal runs of spaces, only
1605 // normalizing line breaks/tabs and trimming the ends.
1606 let out = replaced.replace(['\n', '\r', '\t'], " ").trim().to_string();
1607 fix_arabic_lam_alef(&out)
1608}
1609
1610/// pdfium decomposes the Arabic lam-alef ligature (لا / لإ / لأ / لآ) into its
1611/// glyph constituents in *visual* order — `alef-variant, lam` — but docling keeps
1612/// logical order, `lam, alef-variant`. Swap a mid-word `alef-variant + lam` back
1613/// to `lam + alef-variant`. "Mid-word" (the previous char is an Arabic letter)
1614/// distinguishes the ligature from the definite article `ال` (word-initial
1615/// `alef + lam`), which must stay. No-op for non-Arabic text.
1616fn fix_arabic_lam_alef(s: &str) -> String {
1617 let is_arabic_letter = |c: char| ('\u{0620}'..='\u{064A}').contains(&c);
1618 let chars: Vec<char> = s.chars().collect();
1619 if !chars.iter().any(|&c| is_arabic_letter(c)) {
1620 return s.to_string(); // no-op for non-Arabic text
1621 }
1622 // Pass 1: swap mid-word `alef-variant + lam` → `lam + alef-variant`. Only the
1623 // hamza/madda alef variants (إ أ آ) are safe: the definite article is always
1624 // plain `ا + ل`, so plain `alef + lam` is ambiguous (a legitimate `فعالة` vs a
1625 // reversed `لا` ligature look identical) — leaving plain alef alone avoids
1626 // corrupting legitimate words.
1627 let mut a: Vec<char> = Vec::with_capacity(chars.len());
1628 let mut i = 0;
1629 while i < chars.len() {
1630 let c = chars[i];
1631 if matches!(c, '\u{0622}' | '\u{0623}' | '\u{0625}')
1632 && chars.get(i + 1) == Some(&'\u{0644}')
1633 && i > 0
1634 && is_arabic_letter(chars[i - 1])
1635 // A preceding lam means this alef-variant is *already* the logical
1636 // `lam + alef` ligature; the following lam is the next syllable's
1637 // letter, not a reversed ligature — swapping it corrupts `لآل` → `للآ`
1638 // (e.g. التعلم الآلي → الآلي, not اللآي).
1639 && chars[i - 1] != '\u{0644}'
1640 {
1641 a.push('\u{0644}');
1642 a.push(c);
1643 i += 2;
1644 continue;
1645 }
1646 a.push(c);
1647 i += 1;
1648 }
1649 // Pass 2: insert a space at Arabic↔Latin boundaries (bidi script switch) that
1650 // pdfium runs together — docling separates the embedded Latin run (`وPython`
1651 // → `و Python`).
1652 let mut out: Vec<char> = Vec::with_capacity(a.len());
1653 for (j, &c) in a.iter().enumerate() {
1654 if j > 0 {
1655 let p = a[j - 1];
1656 if (is_arabic_letter(p) && c.is_ascii_alphabetic())
1657 || (p.is_ascii_alphabetic() && is_arabic_letter(c))
1658 {
1659 out.push(' ');
1660 }
1661 }
1662 out.push(c);
1663 }
1664 out.into_iter().collect()
1665}
1666
1667/// docling's `PageAssembleModel._match_hyperlink`: the URI whose link
1668/// annotations cover at least half of the region's box, or `None`. Coverage is
1669/// intersection-over-region-area, **accumulated per URI** — a URL that wraps
1670/// across lines carries several annotation rects that sum toward the same
1671/// target. Ties resolve to the first-seen URI (Python's `max` over dict
1672/// insertion order); the winner still needs `>= 0.5`
1673/// (`_HYPERLINK_COVERAGE_THRESHOLD`).
1674pub(crate) fn region_hyperlink(
1675 region: &Region,
1676 links: &[crate::pdfium_backend::LinkAnnot],
1677) -> Option<String> {
1678 if links.is_empty() {
1679 return None;
1680 }
1681 let area = (region.r - region.l).max(0.0) * (region.b - region.t).max(0.0);
1682 if area <= 0.0 {
1683 return None;
1684 }
1685 let mut coverage: Vec<(&str, f32)> = Vec::new();
1686 for link in links {
1687 let ix = (region.r.min(link.r) - region.l.max(link.l)).max(0.0);
1688 let iy = (region.b.min(link.b) - region.t.max(link.t)).max(0.0);
1689 let c = ix * iy / area;
1690 match coverage.iter_mut().find(|(uri, _)| *uri == link.uri) {
1691 Some((_, acc)) => *acc += c,
1692 None => coverage.push((&link.uri, c)),
1693 }
1694 }
1695 let mut best: Option<(&str, f32)> = None;
1696 for (uri, c) in coverage {
1697 // Strictly greater keeps the first-seen URI on ties, like Python's max.
1698 if best.is_none_or(|(_, bc)| c > bc) {
1699 best = Some((uri, c));
1700 }
1701 }
1702 let (uri, c) = best?;
1703 (c >= 0.5).then(|| normalize_uri(uri))
1704}
1705
1706/// The pydantic-`AnyUrl` normalization docling's hyperlink value passes
1707/// through on its way to the serializer: a URL with an authority but no path
1708/// gains a trailing `/` (`https://arxiv.org` → `https://arxiv.org/`). Other
1709/// AnyUrl canonicalizations (scheme/host lowercasing, percent-encoding) don't
1710/// occur in PDF link annotations in practice, so they are not reproduced.
1711fn normalize_uri(uri: &str) -> String {
1712 if let Some((_, rest)) = uri.split_once("://") {
1713 if !rest.is_empty() && !rest.contains(['/', '?', '#']) {
1714 return format!("{uri}/");
1715 }
1716 }
1717 uri.to_string()
1718}
1719
1720/// Resolve each page hyperlink to the visible text it covers, as `(anchor, uri)`
1721/// in reading order. The anchor is the cells whose centre falls in the link rect,
1722/// joined left-to-right and cleaned the same way prose is (so it matches the
1723/// serialized text), deduped against the immediately-preceding link so pdfium's
1724/// occasional duplicate annotation doesn't double-list. Empty anchors are dropped.
1725pub(crate) fn resolve_link_anchors(page: &PdfPage) -> Vec<(String, String)> {
1726 let mut out: Vec<(String, String)> = Vec::new();
1727 // Use per-word cells, not the line-merged `cells`: a link rect covers a few
1728 // words on a line, and a whole merged line cell would over-capture (its centre
1729 // lands in one link's rect, grabbing the entire line as that link's anchor).
1730 let words = if page.word_cells.is_empty() {
1731 &page.cells
1732 } else {
1733 &page.word_cells
1734 };
1735 for link in &page.links {
1736 // A cell participates when its centre row is inside the rect and it
1737 // overlaps the rect horizontally. A cell can be *wider* than the rect:
1738 // PDFs often draw a whole header line as one text run ("LinkedIn |
1739 // GitHub | Credly"), which docling-parse's word grouping keeps as one
1740 // cell even though each label carries its own link annotation —
1741 // centre-in-rect alone would hand the entire line to every link.
1742 // [`cell_text_in_rect`] clips such a cell to the tokens under the rect.
1743 let mut inside: Vec<(&TextCell, String)> = words
1744 .iter()
1745 .filter(|c| {
1746 let cy = (c.t + c.b) / 2.0;
1747 cy >= link.t && cy <= link.b && c.r.min(link.r) > c.l.max(link.l)
1748 })
1749 .filter_map(|c| {
1750 let text = cell_text_in_rect(c, link.l, link.r);
1751 (!text.is_empty()).then_some((c, text))
1752 })
1753 .collect();
1754 // Reading order: top band then left-to-right (link anchors are LTR).
1755 let band = inside
1756 .iter()
1757 .map(|(c, _)| (c.b - c.t).abs())
1758 .fold(0.0f32, f32::max)
1759 .max(1.0);
1760 inside.sort_by_key(|(c, _)| ((c.t / band).round() as i64, (c.l * 10.0) as i64));
1761 let anchor = clean_text(
1762 &inside
1763 .iter()
1764 .map(|(_, t)| t.trim())
1765 .filter(|t| !t.is_empty())
1766 .collect::<Vec<_>>()
1767 .join(" "),
1768 );
1769 if anchor.is_empty() {
1770 continue;
1771 }
1772 if out
1773 .last()
1774 .is_some_and(|(a, u)| a == &anchor && u == &link.uri)
1775 {
1776 continue;
1777 }
1778 out.push((anchor, link.uri.clone()));
1779 }
1780 out
1781}
1782
1783/// The part of a cell's text that lies under a link rect's x-range. A cell
1784/// fully inside the rect (by centre) returns its whole text. A wider cell is
1785/// split into whitespace tokens whose x-spans are estimated proportionally to
1786/// their character positions (kerning makes this approximate, so selection
1787/// snaps to whole tokens, never characters); tokens whose estimated centre
1788/// falls inside the rect are kept. Returns "" when nothing falls inside.
1789fn cell_text_in_rect(c: &TextCell, l: f32, r: f32) -> String {
1790 let cx = (c.l + c.r) / 2.0;
1791 if cx >= l && cx <= r && c.l >= l - (c.r - c.l) * 0.25 && c.r <= r + (c.r - c.l) * 0.25 {
1792 return c.text.trim().to_string();
1793 }
1794 let chars: Vec<char> = c.text.chars().collect();
1795 let n = chars.len();
1796 if n == 0 || c.r <= c.l {
1797 return String::new();
1798 }
1799 let per = (c.r - c.l) / n as f32;
1800 let mut out: Vec<String> = Vec::new();
1801 let mut token = String::new();
1802 let mut start = 0usize;
1803 // A trailing sentinel space flushes the last token.
1804 for (i, &ch) in chars.iter().enumerate().chain(std::iter::once((n, &' '))) {
1805 if ch.is_whitespace() {
1806 if !token.is_empty() {
1807 let mid = c.l + (start as f32 + (i - start) as f32 / 2.0) * per;
1808 if mid >= l && mid <= r {
1809 out.push(std::mem::take(&mut token));
1810 } else {
1811 token.clear();
1812 }
1813 }
1814 } else {
1815 if token.is_empty() {
1816 start = i;
1817 }
1818 token.push(ch);
1819 }
1820 }
1821 out.join(" ")
1822}
1823
1824/// Cells assigned to a region (best container), in reading order, joined.
1825fn region_text(region: &Region, cells: &[TextCell]) -> String {
1826 let inside: Vec<&TextCell> = cells
1827 .iter()
1828 .filter(|c| {
1829 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1830 inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1831 })
1832 .collect();
1833 cells_text(inside)
1834}
1835
1836/// docling's exclusive cell assignment (`_assign_cells_to_clusters`): every
1837/// non-empty cell goes to the single best-overlapping *regular* region at
1838/// intersection-over-self > 0.2, and each region serializes exactly its
1839/// assigned cells. A cell under two overlapping boxes is emitted once (by the
1840/// better-covering one), and a cell only partially under its region — e.g.
1841/// normal_4pages' big section numeral, ~30 % inside the heading box — still
1842/// joins it (`## 들어가며 1`) instead of leaking as an orphan. Pictures and
1843/// wrappers never claim (docling walks regular clusters only); ties go to the
1844/// first region, like docling's strict `>` best-overlap scan.
1845pub fn region_texts_exclusive(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
1846 let owned = assign_cells(regions, cells);
1847 // Non-claimers (tables/wrappers/pictures) keep the inclusive > 0.5 text:
1848 // docling fills a special cluster's cells from its contained children, and
1849 // downstream table assembly gates on that text being non-empty.
1850 regions
1851 .iter()
1852 .zip(owned)
1853 .map(|(r, cs)| {
1854 if claims_cells(r) {
1855 cells_text(cs.iter().map(|&i| &cells[i]).collect())
1856 } else {
1857 region_text(r, cells)
1858 }
1859 })
1860 .collect()
1861}
1862
1863/// A *regular* region in docling's sense — one that claims cells. Pictures and
1864/// the wrappers (`table`, `document_index`, `form`, `key_value_region`) fill
1865/// their cells from contained children instead.
1866fn claims_cells(r: &Region) -> bool {
1867 r.label != "picture" && !is_wrapper(r.label)
1868}
1869
1870/// docling's `_assign_cells_to_clusters`: each non-empty cell's index goes to
1871/// the single best-overlapping regular region at intersection-over-self > 0.2
1872/// (ties to the first region, like docling's strict `>` scan). One entry per
1873/// region, in region order.
1874fn assign_cells(regions: &[Region], cells: &[TextCell]) -> Vec<Vec<usize>> {
1875 let mut owned: Vec<Vec<usize>> = vec![Vec::new(); regions.len()];
1876 for (ci, c) in cells.iter().enumerate() {
1877 if c.text.trim().is_empty() {
1878 continue;
1879 }
1880 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1881 let mut best: Option<(usize, f32)> = None;
1882 for (i, r) in regions.iter().enumerate() {
1883 if !claims_cells(r) {
1884 continue;
1885 }
1886 let ov = inter(r, c.l, c.t, c.r, c.b) / ca;
1887 if ov > 0.2 && best.is_none_or(|(_, b)| ov > b) {
1888 best = Some((i, ov));
1889 }
1890 }
1891 if let Some((i, _)) = best {
1892 owned[i].push(ci);
1893 }
1894 }
1895 owned
1896}
1897
1898/// docling's regular-cluster refinement after cell assignment
1899/// (`LayoutPostprocessor._process_regular_clusters`, #419), run once the page's
1900/// cells are final and before reading order:
1901///
1902/// 1. every regular region's box becomes the union of the cells it claimed
1903/// (`_adjust_cluster_bboxes` — a regular cluster's bbox *is* its cells'
1904/// bbox; a table's is the union with the model box, and pictures keep
1905/// theirs, so neither is touched here);
1906/// 2. a regular region that claimed no cell is dropped (`keep_empty_clusters`
1907/// is off; a `formula` is kept, as upstream keeps it);
1908/// 3. an orphan text region (`score == 0.0`, from [`add_orphan_regions`]) that
1909/// now sits > 0.8 inside another regular region's fitted box is folded into
1910/// it (`_remove_overlapping_clusters` at containment 0.8, the larger box
1911/// winning the group) — up to three rounds, like upstream's loop.
1912///
1913/// Why it matters: the layout model's box can end partway through a line. That
1914/// line fails the 0.2 claim and becomes an orphan — recoverable — but the
1915/// *model* box still overlaps the orphan's line by a few points, so the
1916/// reading-order graph, which links only strictly-above pairs, gets no edge
1917/// between them and may emit the next paragraph first, stranding the line
1918/// after the paragraph it belongs in (1540 of 6050 text blocks on the #419
1919/// book began mid-sentence). Fitted to its cells, the box ends on a line
1920/// boundary and the orphan slots in between; an orphan the fitted box
1921/// swallows joins the paragraph outright. Cell assignment is untouched: a
1922/// region's fitted box contains every cell it claimed, so
1923/// [`region_texts_exclusive`] hands it the same cells afterwards.
1924///
1925/// A page with no cells yet (a scan before OCR) is left alone: dropping every
1926/// text region for want of cells would be wrong, and the OCR paths call this
1927/// again once the cells exist.
1928pub fn fit_regions_to_cells(regions: &mut Vec<Region>, cells: &[TextCell]) {
1929 if !cells.iter().any(|c| !c.text.trim().is_empty()) {
1930 return;
1931 }
1932 for _ in 0..3 {
1933 let owned = assign_cells(regions, cells);
1934 let mut fitted: Vec<Region> = Vec::with_capacity(regions.len());
1935 for (r, own) in regions.iter().zip(&owned) {
1936 if !claims_cells(r) {
1937 fitted.push(r.clone());
1938 continue;
1939 }
1940 if own.is_empty() {
1941 if r.label == "formula" {
1942 fitted.push(r.clone());
1943 }
1944 continue;
1945 }
1946 let mut f = r.clone();
1947 f.l = own
1948 .iter()
1949 .map(|&i| cells[i].l)
1950 .fold(f32::INFINITY, f32::min);
1951 f.t = own
1952 .iter()
1953 .map(|&i| cells[i].t)
1954 .fold(f32::INFINITY, f32::min);
1955 f.r = own
1956 .iter()
1957 .map(|&i| cells[i].r)
1958 .fold(f32::NEG_INFINITY, f32::max);
1959 f.b = own
1960 .iter()
1961 .map(|&i| cells[i].b)
1962 .fold(f32::NEG_INFINITY, f32::max);
1963 fitted.push(f);
1964 }
1965 let mut changed = fitted.len() != regions.len();
1966 // Fold orphans into the regular region whose fitted box holds them.
1967 let mut drop = vec![false; fitted.len()];
1968 for i in 0..fitted.len() {
1969 let o = &fitted[i];
1970 if !(o.score == 0.0 && o.label == "text") {
1971 continue;
1972 }
1973 let oa = area(o.l, o.t, o.r, o.b).max(1.0);
1974 let mut best: Option<(usize, f32)> = None;
1975 for (j, r) in fitted.iter().enumerate() {
1976 if j == i || drop[j] || r.score == 0.0 || !claims_cells(r) {
1977 continue;
1978 }
1979 let ov = inter(r, o.l, o.t, o.r, o.b) / oa;
1980 if ov > 0.8 && best.is_none_or(|(_, b)| ov > b) {
1981 best = Some((j, ov));
1982 }
1983 }
1984 if let Some((j, _)) = best {
1985 let (l, t, r, b) = (o.l, o.t, o.r, o.b);
1986 let host = &mut fitted[j];
1987 host.l = host.l.min(l);
1988 host.t = host.t.min(t);
1989 host.r = host.r.max(r);
1990 host.b = host.b.max(b);
1991 drop[i] = true;
1992 changed = true;
1993 }
1994 }
1995 let mut drop = drop.into_iter();
1996 fitted.retain(|_| !drop.next().expect("aligned"));
1997 *regions = fitted;
1998 if !changed {
1999 break;
2000 }
2001 }
2002}
2003
2004/// Join a prefiltered cell list into the region's text (docling's
2005/// `sanitize_text` over the sanitizer's cell order).
2006fn cells_text(inside: Vec<&TextCell>) -> String {
2007 // docling orders a cluster's cells by their docling-parse cell index
2008 // alone (`LayoutPostprocessor._sort_cells`: `sorted(cells, key=c.index)`)
2009 // — the sanitizer's output order, which our `cells` slice already is.
2010 // No geometric re-sort: normal_4pages' big section numerals paint
2011 // *after* their heading text, and docling's `## 들어가며 1` (numeral
2012 // last) only falls out of pure index order — a band sort dragged the
2013 // numeral to the front. The overlap-grouped line restore this replaced
2014 // measured strictly worse on the corpus (it fixed nothing the index
2015 // order broke, and broke the numerals).
2016 let joined = {
2017 // docling's `PageAssembleModel.sanitize_text`, ported verbatim over the
2018 // parse-index-ordered lines: append a separating space to a line —
2019 // unless it ends with `-`. A dash-ending line whose last word and the
2020 // next line's first word are both alphanumeric is a wrapped word: the
2021 // dash is dropped and the lines fuse (`platforms-` + `reflects` →
2022 // `platformsreflects`, `pp. 545-` + `561` → `545561`). Any other
2023 // dash-ending line — e.g. the *bare* `-` cell a superscript ORCID or an
2024 // inline `–` bullet splits off (its word list is empty, so the fuse
2025 // test fails) — keeps its dash and still takes no trailing space:
2026 // `[0000` `-` `0002` joins as docling's `[0000 -0002`, and the OTSL
2027 // list's `-` + `"C" cell -` + `a new table cell` collapses to
2028 // `-"C" cell a new table cell`. Our cells still carry the raw dash
2029 // family (docling-parse normalizes to `-` before this; clean_text does
2030 // it after), so the endswith test matches them all.
2031 let texts: Vec<&str> = inside
2032 .iter()
2033 .map(|c| c.text.trim())
2034 // Skip whitespace-only cells (a justified line's trailing space
2035 // glyph): an empty line would double the separator.
2036 .filter(|t| !t.is_empty())
2037 .collect();
2038 let last_word_alnum = |s: &str| {
2039 s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
2040 .rfind(|w| !w.is_empty())
2041 .is_some_and(|w| w.chars().all(char::is_alphanumeric))
2042 };
2043 let first_word_alnum = |s: &str| {
2044 s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
2045 .find(|w| !w.is_empty())
2046 .is_some_and(|w| w.chars().all(char::is_alphanumeric))
2047 };
2048 let mut out = String::new();
2049 for (i, t) in texts.iter().enumerate() {
2050 if i > 0 {
2051 let prev = texts[i - 1];
2052 let dashish = matches!(
2053 prev.chars().last(),
2054 Some(
2055 '-' | '\u{2010}'
2056 | '\u{2011}'
2057 | '\u{2012}'
2058 | '\u{2013}'
2059 | '\u{2014}'
2060 | '\u{2015}'
2061 | '\u{2212}'
2062 )
2063 );
2064 // docling#4052 (2.122): a dash only splits a word when it is
2065 // *attached* to one — the character before it is alphanumeric.
2066 // A dash that follows whitespace (a separator dash, a bullet
2067 // marker, a wrapped `-prefixed` token, the bare `-` cell an
2068 // ORCID splits off) is a literal character: it is kept and the
2069 // lines join with the ordinary space.
2070 let attached = prev.chars().rev().nth(1).is_some_and(char::is_alphanumeric);
2071 if dashish && attached {
2072 if last_word_alnum(prev) && first_word_alnum(t) {
2073 out.pop(); // wrapped word: fuse without the dash
2074 }
2075 // an attached dash never takes a separating space
2076 } else {
2077 out.push(' ');
2078 }
2079 }
2080 out.push_str(t);
2081 }
2082 out
2083 };
2084 clean_text(&joined)
2085}
2086
2087/// Tighten the spaces pdfium leaves around tight punctuation in a code line
2088/// (`console .log` → `console.log`, `add (3 , 5)` → `add(3, 5)`), matching
2089/// docling-parse's source spacing.
2090fn tighten_code_punct(s: &str) -> String {
2091 s.replace(" .", ".")
2092 .replace(" ,", ",")
2093 .replace(" ;", ";")
2094 .replace(" )", ")")
2095 .replace(" (", "(")
2096}
2097
2098/// Assemble a **code** region's text with its line structure preserved.
2099///
2100/// Unlike [`region_text`] — which joins every cell with a single space, the right
2101/// thing for prose reflow — a code block's line breaks and indentation are
2102/// significant. The `code_cells` are already one physical source line each
2103/// (grouped space-glyph-only, so monospace runs keep their spacing), so this:
2104///
2105/// 1. groups the cells into vertical line bands and orders them top→bottom,
2106/// left→right;
2107/// 2. joins the lines with `\n` (rather than spaces), keeping the carriage
2108/// returns; and
2109/// 3. reconstructs each line's leading indentation from its left offset, in units
2110/// of the block's estimated monospace character width, so nesting survives.
2111///
2112/// Typography is normalized per line via [`clean_text`] (smart quotes, dashes,
2113/// ellipsis), which never merges lines. Returns an empty string if the region has
2114/// no code cells (the caller falls back to the prose text).
2115fn code_region_text(region: &Region, cells: &[TextCell]) -> String {
2116 let mut inside: Vec<&TextCell> = cells
2117 .iter()
2118 .filter(|c| {
2119 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2120 inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
2121 })
2122 .filter(|c| !c.text.trim().is_empty())
2123 .collect();
2124 if inside.is_empty() {
2125 return String::new();
2126 }
2127
2128 // Quantize the top edge into ~line bands (like `region_text`), then order the
2129 // cells by band (top→bottom) and, within a band, by left edge.
2130 let band = inside
2131 .iter()
2132 .map(|c| (c.b - c.t).abs())
2133 .fold(0.0f32, f32::max)
2134 .max(1.0);
2135 let line_of = |c: &TextCell| (c.t / band).round() as i64;
2136 inside.sort_by_key(|c| (line_of(c), (c.l * 10.0) as i64));
2137
2138 // Estimate one monospace character's width (total ink width / total glyphs) to
2139 // convert a line's left offset into a count of leading spaces. Measured over
2140 // all lines so a single short line can't skew it.
2141 let (mut total_w, mut total_chars) = (0.0f32, 0usize);
2142 for c in &inside {
2143 let n = c.text.trim().chars().count();
2144 if n > 0 {
2145 total_w += (c.r - c.l).max(0.0);
2146 total_chars += n;
2147 }
2148 }
2149 let char_w = if total_chars > 0 {
2150 (total_w / total_chars as f32).max(1.0)
2151 } else {
2152 1.0
2153 };
2154 // The block's own left margin is the zero-indent baseline.
2155 let base_l = inside.iter().map(|c| c.l).fold(f32::INFINITY, f32::min);
2156
2157 let mut lines: Vec<String> = Vec::new();
2158 let mut cur: Option<i64> = None;
2159 for c in &inside {
2160 // Tighten pdfium's spaced punctuation per line (on the trimmed content, so
2161 // the reconstructed leading indentation is never nibbled).
2162 let text = tighten_code_punct(&clean_text(c.text.trim()));
2163 if Some(line_of(c)) == cur {
2164 // A second cell sharing this band (rare — e.g. split columns): keep it
2165 // on the same source line, separated by a space.
2166 if let Some(last) = lines.last_mut() {
2167 last.push(' ');
2168 last.push_str(&text);
2169 }
2170 continue;
2171 }
2172 let indent = ((c.l - base_l) / char_w).round().max(0.0) as usize;
2173 lines.push(format!("{}{}", " ".repeat(indent), text));
2174 cur = Some(line_of(c));
2175 }
2176 lines.join("\n")
2177}
2178
2179/// Reconstruct a table's grid geometrically from the text cells inside its
2180/// region: cluster cells into rows (by vertical centre) and columns (by clustered
2181/// left edges), then place each cell. A model-free stand-in for TableFormer that
2182/// recovers grid-aligned tables from the precise PDF text layer (it does not
2183/// resolve row/column spans).
2184pub fn reconstruct_table(region: &Region, cells: &[TextCell]) -> Vec<Vec<String>> {
2185 let mut inside: Vec<&TextCell> = cells
2186 .iter()
2187 .filter(|c| {
2188 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2189 inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
2190 })
2191 .collect();
2192 if inside.is_empty() {
2193 return Vec::new();
2194 }
2195 inside.sort_by(|a, b| a.t.total_cmp(&b.t));
2196
2197 // Rows: consecutive cells whose vertical centre is within ~0.7 line height.
2198 let mut rows: Vec<(f32, Vec<&TextCell>)> = Vec::new();
2199 for c in &inside {
2200 let cyc = (c.t + c.b) / 2.0;
2201 let lh = (c.b - c.t).abs().max(1.0);
2202 if let Some((ryc, row)) = rows.last_mut() {
2203 if (cyc - *ryc).abs() < lh * 0.7 {
2204 row.push(c);
2205 continue;
2206 }
2207 }
2208 rows.push((cyc, vec![c]));
2209 }
2210
2211 // Columns: cluster left edges (merge those within a tolerance).
2212 let tol = {
2213 let mut hs: Vec<f32> = inside.iter().map(|c| (c.b - c.t).abs()).collect();
2214 hs.sort_by(f32::total_cmp);
2215 hs[hs.len() / 2].max(4.0) * 1.5
2216 };
2217 let mut lefts: Vec<f32> = inside.iter().map(|c| c.l).collect();
2218 lefts.sort_by(f32::total_cmp);
2219 let mut col_starts: Vec<f32> = Vec::new();
2220 for l in lefts {
2221 if col_starts.last().is_none_or(|&last| l - last > tol) {
2222 col_starts.push(l);
2223 }
2224 }
2225 let ncols = col_starts.len().max(1);
2226 let col_of = |l: f32| -> usize {
2227 col_starts
2228 .iter()
2229 .rposition(|&s| l + tol * 0.5 >= s)
2230 .unwrap_or(0)
2231 .min(ncols - 1)
2232 };
2233
2234 let mut grid = Vec::with_capacity(rows.len());
2235 for (_, mut row) in rows {
2236 row.sort_by(|a, b| a.l.total_cmp(&b.l));
2237 let mut cols = vec![String::new(); ncols];
2238 for c in row {
2239 let ci = col_of(c.l);
2240 // Strip the wrap-hyphen control char so it never lands in a cell.
2241 let t = c.text.trim().replace(['\u{2}', '\u{ad}'], "");
2242 if cols[ci].is_empty() {
2243 cols[ci] = t;
2244 } else {
2245 cols[ci].push(' ');
2246 cols[ci].push_str(&t);
2247 }
2248 }
2249 grid.push(cols);
2250 }
2251 grid
2252}
2253
2254/// Does the geometric reconstruction of a table look trustworthy enough to use
2255/// as-is, instead of paying for TableFormer?
2256///
2257/// [`reconstruct_table`] derives columns by clustering cell **left edges**. On a
2258/// clean grid that is exact, but when a column's entries are not left-aligned
2259/// (or the OCR boxes wobble) the clustering splits one real column into several,
2260/// and the result is a wide, mostly-empty grid — the "spurious empty columns"
2261/// failure TableFormer exists to fix.
2262///
2263/// Two symptoms separate the two cases, and both are properties of the grid
2264/// alone (no model needed):
2265/// * **density** — a real table is mostly full; a split-up one is mostly holes;
2266/// * **thin columns** — a column carrying at most one entry across several rows
2267/// is almost always a split artefact rather than a real column.
2268///
2269/// Deliberately conservative: it answers `true` only for grids that are plainly
2270/// well-formed, so the expensive path stays the default whenever there is doubt.
2271/// A caller that skips TableFormer on `true` trades no quality for the time.
2272pub fn geometric_table_is_reliable(rows: &[Vec<String>]) -> bool {
2273 let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
2274 // Fewer than two columns is not a grid this heuristic can vouch for: it is
2275 // exactly the shape a collapsed table takes, and TableFormer may recover
2276 // real structure from it.
2277 if rows.len() < 2 || ncols < 2 {
2278 return false;
2279 }
2280 let filled = |c: &String| !c.trim().is_empty();
2281 let total = rows.len() * ncols;
2282 let full = rows.iter().flatten().filter(|c| filled(c)).count();
2283 if (full as f32) < MIN_TABLE_FILL * total as f32 {
2284 return false;
2285 }
2286 // A column used by at most one row, when there are rows enough to tell.
2287 if rows.len() >= 3 {
2288 for ci in 0..ncols {
2289 let used = rows
2290 .iter()
2291 .filter(|r| r.get(ci).is_some_and(filled))
2292 .count();
2293 if used <= 1 {
2294 return false;
2295 }
2296 }
2297 }
2298 true
2299}
2300
2301/// Share of a geometric grid's cells that must carry text for it to be trusted
2302/// without TableFormer. Chosen well above the density a left-edge split
2303/// produces (those land nearer a third) and below what a genuine table with a
2304/// few blank cells reaches.
2305const MIN_TABLE_FILL: f32 = 0.6;
2306
2307/// The union bbox of the text cells assigned to a region (same >50%-overlap
2308/// rule as [`region_text`]), or `None` when no cell lands in it. docling's
2309/// LayoutPostprocessor shrinks a regular cluster's bbox to its cells, and the
2310/// enrichment crops are taken from that cell-tight box — cropping the raw
2311/// detector box instead hands the VLM surrounding chrome (e.g. the `Listing N:`
2312/// caption under a code block) that changes its output.
2313pub fn region_cell_bbox(region: &Region, cells: &[TextCell]) -> Option<[f32; 4]> {
2314 let mut bbox: Option<[f32; 4]> = None;
2315 for c in cells {
2316 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2317 if inter(region, c.l, c.t, c.r, c.b) / ca <= 0.5 {
2318 continue;
2319 }
2320 bbox = Some(match bbox {
2321 None => [c.l, c.t, c.r, c.b],
2322 Some([l, t, r, b]) => [l.min(c.l), t.min(c.t), r.max(c.r), b.max(c.b)],
2323 });
2324 }
2325 bbox
2326}
2327
2328/// One region's enrichment-model result, produced by the pipeline's opt-in
2329/// passes (issue #76) and applied during assembly.
2330#[derive(Debug, Clone)]
2331pub enum Enrichment {
2332 /// DocumentPictureClassifier predictions, descending confidence.
2333 PictureClasses(Vec<PictureClass>),
2334 /// CodeFormulaV2 output for a `code` region: the rewritten source text and
2335 /// the `<_language_>` prefix (when the model emitted one).
2336 Code {
2337 language: Option<String>,
2338 text: String,
2339 },
2340 /// CodeFormulaV2 output for a `formula` region: the decoded LaTeX.
2341 Formula { latex: String },
2342}
2343
2344/// Crop a region (page points, already expanded by the caller if needed) from
2345/// the rendered page image and resize it to `target_scale` pixels per point —
2346/// the enrichment-model equivalent of docling's
2347/// `page.get_image(scale=…, cropbox=…)`, sourced from the existing
2348/// [`crate::pdfium_backend::RENDER_SCALE`] render instead of a fresh pdfium
2349/// pass (the page bitmap is already the exact docling render at scale 2).
2350#[cfg(feature = "ml")]
2351pub fn crop_region_scaled(page: &PdfPage, bbox: [f32; 4], target_scale: f32) -> Option<RgbImage> {
2352 let s = page.scale;
2353 let [l, t, r, b] = bbox;
2354 let (iw, ih) = (page.image.width(), page.image.height());
2355 let x = (l * s).max(0.0) as u32;
2356 let y = (t * s).max(0.0) as u32;
2357 if x >= iw || y >= ih {
2358 return None;
2359 }
2360 let w = (((r - l.max(0.0)) * s) as u32).min(iw - x);
2361 let h = (((b - t.max(0.0)) * s) as u32).min(ih - y);
2362 if w == 0 || h == 0 {
2363 return None;
2364 }
2365 let crop = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
2366 // docling renders the crop at `target_scale` directly; from the scale-2
2367 // page render that is a resize to the same pixel geometry
2368 // (`round(width_points * scale)`, PIL's BICUBIC ≙ CatmullRom).
2369 let tw = ((w as f32 / s) * target_scale).round().max(1.0) as u32;
2370 let th = ((h as f32 / s) * target_scale).round().max(1.0) as u32;
2371 if (tw, th) == (w, h) {
2372 return Some(crop);
2373 }
2374 Some(image::imageops::resize(
2375 &crop,
2376 tw,
2377 th,
2378 image::imageops::FilterType::CatmullRom,
2379 ))
2380}
2381
2382/// Resample `img` (rendered at `from` px/pt, covering `w_pt`×`h_pt` points)
2383/// to `to` px/pt — docling's `round(points * scale)` pixel geometry, PIL's
2384/// BICUBIC ≙ CatmullRom. Unchanged when the geometry already matches.
2385#[cfg(feature = "ocr-prep")]
2386fn rescale(img: RgbImage, w_pt: f32, h_pt: f32, to: f32) -> RgbImage {
2387 let tw = (w_pt * to).round().max(1.0) as u32;
2388 let th = (h_pt * to).round().max(1.0) as u32;
2389 if (tw, th) == img.dimensions() {
2390 return img;
2391 }
2392 image::imageops::resize(&img, tw, th, image::imageops::FilterType::CatmullRom)
2393}
2394
2395/// Encode `img` as a PNG [`PictureImage`] rendered at `scale` px/pt.
2396#[cfg(feature = "ocr-prep")]
2397fn png_image(img: &RgbImage, scale: f32) -> Option<PictureImage> {
2398 let mut buf = std::io::Cursor::new(Vec::new());
2399 img.write_to(&mut buf, image::ImageFormat::Png).ok()?;
2400 Some(PictureImage {
2401 mimetype: "image/png".into(),
2402 width: img.width(),
2403 height: img.height(),
2404 data: buf.into_inner(),
2405 dpi: PictureImage::dpi_for_scale(scale),
2406 })
2407}
2408
2409/// The whole page render as docling's `PageItem.image` (#520): at `scale`
2410/// px/pt (`None` = the render's own), `None` when the page has no bitmap.
2411#[cfg(feature = "ocr-prep")]
2412pub fn page_image(page: &PdfPage, scale: Option<f32>) -> Option<PictureImage> {
2413 if page.image.width() == 0 || page.image.height() == 0 || page.scale <= 0.0 {
2414 return None;
2415 }
2416 let scale = scale.unwrap_or(page.scale);
2417 let img = rescale(page.image.clone(), page.width, page.height, scale);
2418 png_image(&img, scale)
2419}
2420
2421/// Crop a layout region from the rendered page image and encode it as PNG (the
2422/// figure bytes docling stores on a `PictureItem`). Region coordinates are page
2423/// points; the image is rendered at `page.scale` and resampled to `scale`
2424/// px/pt when one is given (docling's `images_scale`, #520). The image's `dpi`
2425/// is 72·scale (#519).
2426#[cfg(feature = "ocr-prep")]
2427fn crop_region(page: &PdfPage, region: &Region, scale: Option<f32>) -> Option<PictureImage> {
2428 let s = page.scale;
2429 let (iw, ih) = (page.image.width(), page.image.height());
2430 let x = (region.l * s).max(0.0) as u32;
2431 let y = (region.t * s).max(0.0) as u32;
2432 if x >= iw || y >= ih {
2433 return None;
2434 }
2435 let w = (((region.r - region.l) * s) as u32).min(iw - x);
2436 let h = (((region.b - region.t) * s) as u32).min(ih - y);
2437 if w == 0 || h == 0 {
2438 return None;
2439 }
2440 let sub = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
2441 match scale {
2442 Some(to) if (to - s).abs() > f32::EPSILON => {
2443 // The crop's own point extent (the pixel box, back in points), so
2444 // the resampled geometry is `round(points * scale)`.
2445 let img = rescale(sub, w as f32 / s, h as f32 / s, to);
2446 png_image(&img, to)
2447 }
2448 _ => png_image(&sub, s),
2449 }
2450}
2451
2452/// For each `picture` region, find the `caption` region closest below it (and
2453/// horizontally overlapping); docling pairs them and emits the caption first.
2454/// Each caption is claimed by at most one picture.
2455fn pair_captions(regions: &[Region]) -> Vec<Option<usize>> {
2456 let mut pairs = vec![None; regions.len()];
2457 let mut taken = vec![false; regions.len()];
2458 for (pi, p) in regions.iter().enumerate() {
2459 if p.label != "picture" {
2460 continue;
2461 }
2462 let mut best: Option<(usize, f32)> = None;
2463 for (ci, c) in regions.iter().enumerate() {
2464 if c.label != "caption" || taken[ci] {
2465 continue;
2466 }
2467 let line_h = (c.b - c.t).abs().max(1.0);
2468 let gap = c.t - p.b; // caption sits below the picture
2469 let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
2470 if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
2471 let dist = gap.abs();
2472 if best.is_none_or(|(_, bd)| dist < bd) {
2473 best = Some((ci, dist));
2474 }
2475 }
2476 }
2477 if let Some((ci, _)) = best {
2478 pairs[pi] = Some(ci);
2479 taken[ci] = true;
2480 }
2481 }
2482 pairs
2483}
2484
2485/// Pair each `code` region with the `caption` region just **above** it (a
2486/// `Listing N:` label). docling renders the code block first, then its caption,
2487/// so the caption is consumed from its own (earlier) reading-order slot and
2488/// re-emitted after the code.
2489fn pair_code_captions(regions: &[Region]) -> Vec<Option<usize>> {
2490 let mut pairs = vec![None; regions.len()];
2491 let mut taken = vec![false; regions.len()];
2492 for (pi, p) in regions.iter().enumerate() {
2493 if p.label != "code" {
2494 continue;
2495 }
2496 let mut best: Option<(usize, f32)> = None;
2497 for (ci, c) in regions.iter().enumerate() {
2498 if c.label != "caption" || taken[ci] {
2499 continue;
2500 }
2501 let line_h = (c.b - c.t).abs().max(1.0);
2502 let gap = p.t - c.b; // caption sits above the code
2503 let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
2504 if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
2505 let dist = gap.abs();
2506 if best.is_none_or(|(_, bd)| dist < bd) {
2507 best = Some((ci, dist));
2508 }
2509 }
2510 }
2511 if let Some((ci, _)) = best {
2512 pairs[pi] = Some(ci);
2513 taken[ci] = true;
2514 }
2515 }
2516 pairs
2517}
2518
2519/// Pair each `table`/`document_index` region with its `caption` (#265) the way
2520/// docling's `ReadingOrderPredictor._find_to_captions` does: by **reading-order
2521/// adjacency**, not geometry. A caption claims the media element
2522/// (table/picture/code) immediately next to it in the ordered region sequence,
2523/// and only when exactly one side holds one — a caption sandwiched between two
2524/// media elements stays unattached, and a text paragraph between caption and
2525/// table breaks the bond. This is what lets a flush-left "Table 3: …" label
2526/// bind a centered grid it doesn't horizontally overlap, while a caption in
2527/// the neighbouring column of a two-column page — geometrically close — never
2528/// pairs across the gutter. Runs after the picture and code pairings (the
2529/// picture/code arms of the same upstream matcher), so a caption they claimed
2530/// stays claimed. docling attaches these as `TableItem.captions` refs; the
2531/// paired caption is consumed from its own reading-order slot and rides on the
2532/// table node instead.
2533fn pair_table_captions(regions: &[Region], taken: &mut [bool]) -> Vec<Option<usize>> {
2534 let is_media = |label: &str| is_table_like(label) || matches!(label, "picture" | "code");
2535 let mut pairs: Vec<Option<usize>> = vec![None; regions.len()];
2536 for ci in 0..regions.len() {
2537 if regions[ci].label != "caption" || taken[ci] {
2538 continue;
2539 }
2540 // Furniture (headers/footers, form chrome) is not part of docling's
2541 // body-element sequence, so it neither bonds nor blocks.
2542 let prev = regions[..ci].iter().rposition(|r| !is_skipped(r.label));
2543 let next = regions[ci + 1..]
2544 .iter()
2545 .position(|r| !is_skipped(r.label))
2546 .map(|off| ci + 1 + off);
2547 let prev_media = prev.is_some_and(|j| is_media(regions[j].label));
2548 let next_media = next.is_some_and(|j| is_media(regions[j].label));
2549 let target = match (prev_media, next_media) {
2550 (true, false) => prev,
2551 (false, true) => next,
2552 // Ambiguous (media on both sides) or no media at all: leave the
2553 // caption in its own reading-order slot, as docling does.
2554 _ => None,
2555 };
2556 if let Some(ti) = target {
2557 // A first claim wins (a table with captions above *and* below
2558 // keeps the earlier one — docling's nearest-first tiebreak).
2559 if is_table_like(regions[ti].label) && pairs[ti].is_none() {
2560 pairs[ti] = Some(ci);
2561 taken[ci] = true;
2562 }
2563 }
2564 }
2565 pairs
2566}
2567
2568/// Assemble one page from its (already overlap-resolved) layout regions and
2569/// text cells.
2570/// Normalize a layout region (page points, top-left origin) to DocLang's 0–511
2571/// location grid: `clamp(round(512 · coord / page_dim), 0, 511)`, per axis,
2572/// order `[x0, y0, x1, y1]`. Mirrors docling_core's
2573/// `_doclang_utils._create_location_tokens_for_bbox` (resolution 512) so the
2574/// emitted `<location>` tokens line up with the Python groundtruth. Our heron
2575/// cluster boxes match docling's to within ~1 grid unit; the residual (mainly
2576/// the aspect-ratio-stretch vs letterbox preprocessing difference) is absorbed
2577/// by the conformance harness's geometry tolerance.
2578fn norm_loc(region: &Region, page_w: f32, page_h: f32) -> [u16; 4] {
2579 let q = |v: f32, dim: f32| -> u16 {
2580 if dim <= 0.0 {
2581 return 0;
2582 }
2583 let g = (512.0 * (v as f64) / (dim as f64)).round() as i64;
2584 g.clamp(0, 511) as u16
2585 };
2586 [
2587 q(region.l, page_w),
2588 q(region.t, page_h),
2589 q(region.r, page_w),
2590 q(region.b, page_h),
2591 ]
2592}
2593
2594/// Wrap a node in its layout provenance so the DocLang serializer emits the four
2595/// `<location>` tokens as the element's head (Markdown/JSON render `inner`
2596/// unchanged).
2597fn located(loc: [u16; 4], inner: Node) -> Node {
2598 Node::Located {
2599 location: loc,
2600 inner: Box::new(inner),
2601 }
2602}
2603
2604/// Stamp the real 1-based page number onto a page's leading marker (see
2605/// [`assemble_page`], which emits it with `page_no: 0` because only the
2606/// document-level collector knows the true index — `--pages` windows shift it).
2607pub fn stamp_page_no(nodes: &mut [Node], page_no: usize) {
2608 if let Some(Node::PageInfo { page_no: p, .. }) = nodes.first_mut() {
2609 *p = page_no;
2610 }
2611}
2612
2613/// A dense table grid plus its first-class cells (#240): `rows` is the text
2614/// grid every serializer renders (spans replicate their anchor's text);
2615/// `cells` are the docling-parity per-cell records (text, page-point bbox,
2616/// span rectangle, OTSL header roles). Produced by the TableFormer paths
2617/// (`tf_core`); lives in this always-compiled module so the pure-text (wasm
2618/// `pdf-text`) build sees the type.
2619#[derive(Clone, Debug)]
2620pub struct TableGrid {
2621 pub rows: Vec<Vec<String>>,
2622 pub cells: Vec<docling_core::TableCell>,
2623}
2624
2625/// docling's `_RICH_CELL_PICTURE_COVERAGE_THRESHOLD`.
2626const RICH_CELL_PICTURE_COVERAGE: f32 = 0.8;
2627
2628/// docling `ReadingOrderModel._match_table_pictures` (#3906, 2.118.1): every
2629/// picture ≥ 80 % inside a TableFormer-structured table on the page is matched
2630/// to the cell covering it, and returned per table as `cell index → pictures`.
2631/// A picture that pairs with a caption stays a standalone figure (upstream
2632/// would nest it and lose the caption; keeping the caption is the better
2633/// failure). Tables without first-class cells (geometric fallback) have no cell
2634/// boxes to match against and nest nothing.
2635fn match_table_pictures(
2636 regions: &[Region],
2637 table_rows: &[Option<TableGrid>],
2638 caption_for: &[Option<usize>],
2639) -> std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> {
2640 let mut out: std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> =
2641 std::collections::HashMap::new();
2642 for (p, pic) in regions.iter().enumerate() {
2643 if pic.label != "picture" || caption_for.get(p).is_some_and(Option::is_some) {
2644 continue;
2645 }
2646 let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2647 let mut best: Option<(f32, usize, usize)> = None; // (coverage, table, cell)
2648 for (t, tbl) in regions.iter().enumerate() {
2649 if !is_table_like(tbl.label) {
2650 continue;
2651 }
2652 let Some(grid) = table_rows.get(t).and_then(Option::as_ref) else {
2653 continue;
2654 };
2655 if inter(pic, tbl.l, tbl.t, tbl.r, tbl.b) / pa < RICH_CELL_PICTURE_COVERAGE {
2656 continue;
2657 }
2658 if let Some((cov, cell)) = match_picture_to_cell(pic, &grid.cells) {
2659 if best.is_none_or(|(b, _, _)| cov > b) {
2660 best = Some((cov, t, cell));
2661 }
2662 }
2663 }
2664 if let Some((_, t, cell)) = best {
2665 let entry = out.entry(t).or_default();
2666 match entry.iter_mut().find(|(c, _)| *c == cell) {
2667 Some((_, pics)) => pics.push(p),
2668 None => entry.push((cell, vec![p])),
2669 }
2670 }
2671 }
2672 out
2673}
2674
2675/// docling `_match_picture_to_table_cell`: among the cells covering ≥ 80 % of
2676/// the picture, prefer the one at the picture's inferred grid position (the
2677/// row / column whose median cell center is nearest the picture's center —
2678/// cell boxes can overlap across logical rows and columns), else the best
2679/// coverage. Returns `(coverage, cell index)`.
2680fn match_picture_to_cell(pic: &Region, cells: &[docling_core::TableCell]) -> Option<(f32, usize)> {
2681 let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2682 let cover = |b: &[f32; 4]| inter(pic, b[0], b[1], b[2], b[3]) / pa;
2683 let eligible: Vec<(f32, usize)> = cells
2684 .iter()
2685 .enumerate()
2686 .filter_map(|(i, c)| {
2687 let b = c.bbox.as_ref()?;
2688 let cov = cover(b);
2689 (cov >= RICH_CELL_PICTURE_COVERAGE).then_some((cov, i))
2690 })
2691 .collect();
2692 if eligible.is_empty() {
2693 return None;
2694 }
2695 let mut row_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2696 let mut col_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2697 for c in cells {
2698 let Some(b) = c.bbox.as_ref() else { continue };
2699 for r in c.start_row..c.start_row + c.row_span {
2700 row_centers.entry(r).or_default().push((b[1] + b[3]) / 2.0);
2701 }
2702 for k in c.start_col..c.start_col + c.col_span {
2703 col_centers.entry(k).or_default().push((b[0] + b[2]) / 2.0);
2704 }
2705 }
2706 let median = |v: &mut Vec<f32>| -> f32 {
2707 v.sort_by(f32::total_cmp);
2708 let n = v.len();
2709 if n % 2 == 1 {
2710 v[n / 2]
2711 } else {
2712 (v[n / 2 - 1] + v[n / 2]) / 2.0
2713 }
2714 };
2715 let (px, py) = ((pic.l + pic.r) / 2.0, (pic.t + pic.b) / 2.0);
2716 let nearest = |centers: &mut std::collections::BTreeMap<usize, Vec<f32>>, target: f32| {
2717 centers
2718 .iter_mut()
2719 .map(|(&i, v)| (i, (median(v) - target).abs()))
2720 .min_by(|a, b| a.1.total_cmp(&b.1))
2721 .map(|(i, _)| i)
2722 };
2723 let row = nearest(&mut row_centers, py);
2724 let col = nearest(&mut col_centers, px);
2725 let logical: Vec<(f32, usize)> = eligible
2726 .iter()
2727 .copied()
2728 .filter(|&(_, i)| {
2729 let c = &cells[i];
2730 row.is_some_and(|r| c.start_row <= r && r < c.start_row + c.row_span)
2731 && col.is_some_and(|k| c.start_col <= k && k < c.start_col + c.col_span)
2732 })
2733 .collect();
2734 let pool = if logical.is_empty() {
2735 &eligible
2736 } else {
2737 &logical
2738 };
2739 // Python's `max` over `(coverage, cell_index, cell)` tuples: highest
2740 // coverage, ties to the higher index.
2741 pool.iter()
2742 .copied()
2743 .max_by(|a, b| a.0.total_cmp(&b.0).then(a.1.cmp(&b.1)))
2744}
2745
2746/// The DocLang structure overlay derived from first-class cells: span
2747/// continuations (`lcel`/`ucel`/`xcel`) and per-cell header roles, so the
2748/// PDF path's DCLX carries real spans instead of a flat grid.
2749fn structure_from_cells(
2750 cells: &[docling_core::TableCell],
2751 nrows: usize,
2752 ncols: usize,
2753) -> docling_core::TableStructure {
2754 let grid = || vec![vec![false; ncols]; nrows];
2755 let mut col_cont = grid();
2756 let mut row_cont = grid();
2757 let mut row_header = grid();
2758 let mut col_header = grid();
2759 for c in cells {
2760 for r in c.start_row..(c.start_row + c.row_span).min(nrows) {
2761 for k in c.start_col..(c.start_col + c.col_span).min(ncols) {
2762 col_cont[r][k] = k > c.start_col;
2763 row_cont[r][k] = r > c.start_row;
2764 row_header[r][k] = c.row_header;
2765 col_header[r][k] = c.column_header;
2766 }
2767 }
2768 }
2769 docling_core::TableStructure {
2770 header_row: Vec::new(),
2771 col_continuation: col_cont,
2772 row_continuation: row_cont,
2773 row_header,
2774 col_header,
2775 }
2776}
2777
2778pub fn assemble_page(
2779 page: &PdfPage,
2780 regions: Vec<Region>,
2781 table_rows: &[Option<TableGrid>],
2782 enrichments: &[Option<Enrichment>],
2783 // Picture-crop scale in px/pt (docling's `images_scale`, #520); `None`
2784 // keeps the page render's own scale.
2785 picture_scale: Option<f32>,
2786) -> (Vec<Node>, Vec<(String, String)>) {
2787 // Without pixels (the text-layer-only wasm build) no picture is cropped.
2788 #[cfg(not(feature = "ocr-prep"))]
2789 let _ = picture_scale;
2790 let mut nodes: Vec<Node> = Vec::new();
2791 // Every page opens with an invisible page marker carrying its size in
2792 // points — what the JSON export needs to build docling's `pages` map and
2793 // denormalize the 0–511 `<location>` grid into point bboxes (#171). The
2794 // page *number* is stamped by the document-level collector (which knows
2795 // the real 1-based index, `--pages` windows included); every serializer
2796 // except JSON skips the marker, so Markdown/DocLang stay byte-identical.
2797 nodes.push(Node::PageInfo {
2798 page_no: 0,
2799 width: page.width,
2800 height: page.height,
2801 });
2802 // Recover this page's hyperlinks (anchor-precise pairs for strict
2803 // Markdown; whole-item docling-parity links are baked below and their
2804 // pairs dropped from this list so strict output doesn't double-wrap).
2805 let mut links = resolve_link_anchors(page);
2806 // Pair each region with its precomputed TableFormer grid and enrichment
2807 // (indexed by original order) and order by reading order together, so they
2808 // stay aligned.
2809 // A picture's children (docling's `_set_cluster_children`: the regulars
2810 // > 80 % inside it) are not page elements — they leave the reading order
2811 // here and ride with their picture, to be written under it in the JSON.
2812 let parents = picture_parents(®ions);
2813 let mut kids: Vec<Vec<Region>> = vec![Vec::new(); regions.len()];
2814 let mut top: Vec<(usize, Region)> = Vec::with_capacity(regions.len());
2815 for (i, (r, parent)) in regions.into_iter().zip(parents).enumerate() {
2816 match parent {
2817 Some(p) => kids[p].push(r),
2818 None => top.push((i, r)),
2819 }
2820 }
2821 // docling's assembly order of the regions — what its reading-order
2822 // predictor knows as `cid` (#424) — before they are shuffled.
2823 let top_regions: Vec<Region> = top.iter().map(|(_, r)| r.clone()).collect();
2824 let cids = cluster_cids(&top_regions, &page.cells);
2825 type RegionItem = (Region, Option<TableGrid>, Option<Enrichment>, Vec<Region>);
2826 let mut items: Vec<RegionItem> = top
2827 .into_iter()
2828 .map(|(i, r)| {
2829 (
2830 r,
2831 table_rows.get(i).cloned().flatten(),
2832 enrichments.get(i).cloned().flatten(),
2833 std::mem::take(&mut kids[i]),
2834 )
2835 })
2836 .collect();
2837 order_with_containers(&mut items, &cids, page.width, page.height, |it| &it.0);
2838 // Float a margin page number to the front of reading order (docling parity:
2839 // right_to_left_02's bottom `11` is its first item). Stable, so everything
2840 // else keeps its order; no-op on pages without such a region.
2841 let page_h = page.height;
2842 items.sort_by_key(|(r, _, _, _)| !is_page_number(r, &page.cells, page_h));
2843 let table_rows: Vec<Option<TableGrid>> = items.iter().map(|(_, t, _, _)| t.clone()).collect();
2844 let enrichments: Vec<Option<Enrichment>> = items.iter().map(|(_, _, e, _)| e.clone()).collect();
2845 let mut picture_children: Vec<Vec<Region>> = items
2846 .iter_mut()
2847 .map(|it| std::mem::take(&mut it.3))
2848 .collect();
2849 let regions: Vec<Region> = items.into_iter().map(|(r, _, _, _)| r).collect();
2850 // Children in docling's `_sort_clusters(mode="id")` order: first source
2851 // cell, then top, then left.
2852 for kids in picture_children.iter_mut().filter(|k| k.len() > 1) {
2853 let rank = cluster_cids(kids, &page.cells);
2854 let mut ranked: Vec<(usize, Region)> = rank.into_iter().zip(kids.drain(..)).collect();
2855 ranked.sort_by_key(|(k, _)| *k);
2856 kids.extend(ranked.into_iter().map(|(_, r)| r));
2857 }
2858 // docling emits a figure's caption *before* the image marker. Pair each
2859 // picture with the caption region nearest below it and consume that caption,
2860 // so it isn't also emitted in its own (lower) reading-order position.
2861 let caption_for = pair_captions(®ions);
2862 let code_caption_for = pair_code_captions(®ions);
2863 let mut consumed = vec![false; regions.len()];
2864 for ci in caption_for.iter().flatten() {
2865 consumed[*ci] = true;
2866 }
2867 for ci in code_caption_for.iter().flatten() {
2868 consumed[*ci] = true;
2869 }
2870 // Table captions (#265) claim from what the picture/code pairings left.
2871 let mut caption_taken = consumed.clone();
2872 let table_caption_for = pair_table_captions(®ions, &mut caption_taken);
2873 for ci in table_caption_for.iter().flatten() {
2874 consumed[*ci] = true;
2875 }
2876 // Pictures inside a table become rich-cell content (docling#3906, 2.118.1):
2877 // the picture is nested in the cell it covers and not emitted standalone.
2878 let rich_cell_pictures = match_table_pictures(®ions, &table_rows, &caption_for);
2879 for (_, pics) in rich_cell_pictures.values().flatten() {
2880 for &p in pics {
2881 consumed[p] = true;
2882 }
2883 }
2884 // A code block's language label (`XML`, `C#`, …) is chrome, not content — the
2885 // detector emits it as its own region above the code; consume it.
2886 for (i, is_label) in code_language_labels(®ions, &page.cells)
2887 .into_iter()
2888 .enumerate()
2889 {
2890 if is_label {
2891 consumed[i] = true;
2892 }
2893 }
2894
2895 // docling `ReadingOrderPredictor.predict_merges`: join a text fragment with a
2896 // following text fragment strictly to its right (an author column that wraps
2897 // into the next, a paragraph continuing in the next column) into one block —
2898 // the intra-page half of docling's reading-order merges (cross-page/vertical
2899 // continuations stay with [`merge_continuations`]). Already-consumed regions
2900 // (paired captions, code labels) are excluded.
2901 // Exclusive docling cell assignment: computed once for the ordered region
2902 // list and reused for every serialization below, so a cell can never render
2903 // in two regions. The picture children take part (docling assigns cells to
2904 // every regular cluster before it nests any); their texts are split off.
2905 let with_children: Vec<Region> = regions
2906 .iter()
2907 .chain(picture_children.iter().flatten())
2908 .cloned()
2909 .collect();
2910 let mut region_texts: Vec<String> = region_texts_exclusive(&with_children, &page.cells);
2911 let mut kid_texts = region_texts.split_off(regions.len()).into_iter();
2912 let child_texts: Vec<Vec<String>> = picture_children
2913 .iter()
2914 .map(|k| kid_texts.by_ref().take(k.len()).collect())
2915 .collect();
2916 let is_text: Vec<bool> = regions
2917 .iter()
2918 .enumerate()
2919 .map(|(i, r)| r.label == "text" && !consumed[i])
2920 .collect();
2921 let is_skip: Vec<bool> = regions
2922 .iter()
2923 .enumerate()
2924 .map(|(i, r)| {
2925 consumed[i]
2926 || matches!(
2927 r.label,
2928 "page_header" | "page_footer" | "table" | "picture" | "caption" | "footnote"
2929 )
2930 })
2931 .collect();
2932 let boxes: Vec<(f32, f32, f32, f32)> = regions.iter().map(|r| (r.l, r.t, r.r, r.b)).collect();
2933 if docling_core::env::flag("DOCLING_RS_DEBUG_MERGES") {
2934 for (i, r) in regions.iter().enumerate() {
2935 eprintln!(
2936 "MRG {i:2} {} text={} skip={} [{:.0},{:.0},{:.0},{:.0}] {:?}",
2937 r.label,
2938 is_text[i],
2939 is_skip[i],
2940 r.l,
2941 r.t,
2942 r.r,
2943 r.b,
2944 region_texts[i].chars().take(40).collect::<String>()
2945 );
2946 }
2947 }
2948 let mut merge_suffix: Vec<String> = vec![String::new(); regions.len()];
2949 for (head, children) in
2950 crate::reading_order::predict_merges(&boxes, ®ion_texts, &is_text, &is_skip)
2951 .into_iter()
2952 .enumerate()
2953 {
2954 for c in children {
2955 let t = region_texts[c].trim();
2956 if !t.is_empty() {
2957 merge_suffix[head].push(' ');
2958 merge_suffix[head].push_str(t);
2959 }
2960 consumed[c] = true;
2961 }
2962 }
2963
2964 for (i, region) in regions.iter().enumerate() {
2965 if consumed[i] {
2966 continue;
2967 }
2968 // Page headers/footers: docling emits them as furniture blocks
2969 // (`<page_header>`/`<page_footer>` with a layer + location + text) at
2970 // their reading-order position, not as body — emit them, don't skip.
2971 if matches!(region.label, "page_header" | "page_footer") {
2972 let text = region_texts[i].clone();
2973 if !text.is_empty() {
2974 nodes.push(Node::PageFurniture {
2975 footer: region.label == "page_footer",
2976 location: norm_loc(region, page.width, page_h),
2977 text: md_escape(&text),
2978 });
2979 }
2980 continue;
2981 }
2982 if is_skipped(region.label) {
2983 continue;
2984 }
2985 // Layout provenance for this region, normalized to docling's 0–511 grid.
2986 let loc = norm_loc(region, page.width, page_h);
2987 if region.label == "picture" {
2988 // The figure pixels are cropped from the page render for image export.
2989 // Captions are prose: markdown-escaped like a paragraph (the JSON
2990 // export unescapes back to the raw text, matching docling).
2991 let caption = caption_for[i]
2992 .map(|ci| md_escape(®ion_texts[ci]))
2993 .filter(|t| !t.is_empty());
2994 let classification = match &enrichments[i] {
2995 Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
2996 _ => None,
2997 };
2998 // Without the page render (text-layer-only build) a picture keeps
2999 // its caption/classification but carries no cropped pixels.
3000 #[cfg(feature = "ocr-prep")]
3001 let image =
3002 crate::timing::timed("crop_region", || crop_region(page, region, picture_scale));
3003 #[cfg(not(feature = "ocr-prep"))]
3004 let image: Option<PictureImage> = None;
3005 nodes.push(located(
3006 loc,
3007 Node::Picture {
3008 caption,
3009 caption_href: None,
3010 image,
3011 classification,
3012 // docling's layout pipeline parents a figure's caption to
3013 // the picture itself (#390) — the one backend that does.
3014 caption_parent: CaptionParent::Item,
3015 },
3016 ));
3017 let children: Vec<Node> = picture_children[i]
3018 .iter()
3019 .zip(&child_texts[i])
3020 .filter_map(|(r, text)| {
3021 picture_child_node(r, text, norm_loc(r, page.width, page_h))
3022 })
3023 .collect();
3024 if !children.is_empty() {
3025 nodes.push(Node::PictureChildren(children));
3026 }
3027 continue;
3028 }
3029 let mut text = region_texts[i].clone();
3030 text.push_str(&merge_suffix[i]);
3031 if text.is_empty() {
3032 continue;
3033 }
3034 match region.label {
3035 // docling assembles checkboxes as TEXT_ELEM items (the region's
3036 // cells are the option label, e.g. right_to_left_03's بلی/خير)
3037 // and its Markdown serializer renders them as task-list lines
3038 // (`- [x] …`) — mirrored by [`Node::CheckboxItem`].
3039 "checkbox_selected" | "checkbox_unselected" => nodes.push(Node::CheckboxItem {
3040 checked: region.label == "checkbox_selected",
3041 text: md_escape(&text),
3042 }),
3043 // docling renders both the document title and section headers as
3044 // `##` (it never emits a top-level `#` for PDFs), so match that.
3045 "title" | "section_header" => nodes.push(located(
3046 loc,
3047 Node::Heading {
3048 level: 2,
3049 text: md_escape(&text),
3050 },
3051 )),
3052 // docling's `ListItemMarkerProcessor.process_list_item` runs on
3053 // every PDF list item: a leading bullet glyph or enumeration marker
3054 // followed by whitespace is split off into the item's `marker`, and
3055 // docling-core's Markdown then prints `- text` for a bullet, `N. text`
3056 // for an `N.` marker and `- a) text` for any other marker holding a
3057 // letter or digit (see [`list_item_node`]). The symbol-font bullets
3058 // docling-parse filters out of its cells are stripped first.
3059 "list_item" => nodes.push(list_item_node(&text, loc, false)),
3060 // TableFormer structure (cells + spans, text matched from word cells)
3061 // when available; otherwise geometric grid reconstruction; finally a
3062 // single cell.
3063 "table" | "document_index" => {
3064 // TableFormer grids carry first-class cells (#240: text +
3065 // page-point bbox + span rectangle + OTSL header roles) into
3066 // the public model, and the DocLang structure overlay derives
3067 // from them so DCLX emits real span/header tokens. The
3068 // geometric fallback has no per-cell records.
3069 let (mut rows, cells, structure) = match table_rows[i].clone() {
3070 Some(grid) => {
3071 let nrows = grid.rows.len();
3072 let ncols = grid.rows.first().map_or(0, Vec::len);
3073 let structure = structure_from_cells(&grid.cells, nrows, ncols);
3074 (grid.rows, Some(grid.cells), Some(structure))
3075 }
3076 None => {
3077 let rows = reconstruct_table(region, &page.cells);
3078 let rows = if rows.iter().any(|r| r.len() > 1) {
3079 rows
3080 } else {
3081 vec![vec![text.clone()]]
3082 };
3083 (rows, None, None)
3084 }
3085 };
3086 // The paired caption (#265) rides on the table — docling's
3087 // TableItem.captions ref; Markdown prints it above the grid,
3088 // the JSON export emits the $ref, DocLang the <caption>.
3089 let caption = table_caption_for[i]
3090 .map(|ci| md_escape(®ion_texts[ci]))
3091 .filter(|t| !t.is_empty());
3092 // Rich cells (docling#3906): the covering cell's blocks are its
3093 // text followed by the nested picture(s). docling's Markdown
3094 // renders a `RichTableCell` through the serializer — the
3095 // group's children joined by blank lines, newlines flattened
3096 // to spaces — so the flat `rows` text becomes
3097 // `text <!-- image -->`; the first-class `cells` (the JSON
3098 // `table_cells` / `grid`) keep the plain text, as upstream.
3099 let mut cell_blocks: Option<Vec<Vec<Vec<Node>>>> = None;
3100 if let (Some(by_cell), Some(fc)) = (rich_cell_pictures.get(&i), cells.as_ref()) {
3101 let nrows = rows.len();
3102 let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
3103 let mut blocks = vec![vec![Vec::<Node>::new(); ncols]; nrows];
3104 for (cell_idx, pics) in by_cell {
3105 let cell = &fc[*cell_idx];
3106 let (r, c) = (cell.start_row, cell.start_col);
3107 if r >= nrows || c >= ncols {
3108 continue;
3109 }
3110 let mut parts: Vec<String> = Vec::new();
3111 let mut cell_nodes: Vec<Node> = Vec::new();
3112 if !cell.text.trim().is_empty() {
3113 parts.push(cell.text.clone());
3114 cell_nodes.push(Node::Paragraph {
3115 text: cell.text.clone(),
3116 });
3117 }
3118 for &p in pics {
3119 parts.push("<!-- image -->".to_string());
3120 let classification = match &enrichments[p] {
3121 Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
3122 _ => None,
3123 };
3124 #[cfg(feature = "ocr-prep")]
3125 let image = crop_region(page, ®ions[p], picture_scale);
3126 #[cfg(not(feature = "ocr-prep"))]
3127 let image: Option<PictureImage> = None;
3128 cell_nodes.push(located(
3129 norm_loc(®ions[p], page.width, page_h),
3130 Node::Picture {
3131 caption: None,
3132 caption_href: None,
3133 image,
3134 classification,
3135 caption_parent: Default::default(),
3136 },
3137 ));
3138 }
3139 let rendered = parts.join(" ");
3140 for row in rows.iter_mut().skip(r).take(cell.row_span) {
3141 for slot in row.iter_mut().skip(c).take(cell.col_span) {
3142 *slot = rendered.clone();
3143 }
3144 }
3145 blocks[r][c] = cell_nodes;
3146 }
3147 cell_blocks = Some(blocks);
3148 }
3149 nodes.push(located(
3150 loc,
3151 Node::Table(Table {
3152 rows,
3153 location: None,
3154 structure,
3155 cell_blocks,
3156 cells,
3157 caption,
3158 // As for pictures: the caption is the table's child.
3159 caption_parent: CaptionParent::Item,
3160 }),
3161 ));
3162 }
3163 // With formula enrichment the CodeFormula model decodes the region
3164 // to LaTeX; otherwise docling emits a placeholder comment rather
3165 // than the (garbled) raw glyph text.
3166 "formula" => match &enrichments[i] {
3167 Some(Enrichment::Formula { latex }) => nodes.push(Node::Formula {
3168 latex: latex.clone(),
3169 orig: text.clone(),
3170 location: Some(loc),
3171 }),
3172 _ => nodes.push(Node::Paragraph {
3173 text: "<!-- formula-not-decoded -->".into(),
3174 }),
3175 },
3176 // Code blocks: use the space-glyph-only grouping (monospace keeps its
3177 // source spacing) and emit a fenced block, preserving the line breaks
3178 // and indentation of the source (unlike prose, which reflows). pdfium
3179 // still inserts spaces around tight punctuation (`console .log`,
3180 // `add (3 , 5)`); tighten them to match docling-parse's source spacing.
3181 "code" => {
3182 // `code_region_text` preserves line breaks/indentation and tightens
3183 // each line itself; the fallback prose `text` is tightened here.
3184 let code = code_region_text(region, &page.code_cells);
3185 let code = if code.is_empty() {
3186 tighten_code_punct(&text)
3187 } else {
3188 code
3189 };
3190 // With code enrichment the CodeFormula model rewrites the block
3191 // (and names its language); `orig` keeps the raw extraction in
3192 // docling's shape — its parser has no line-preserving code
3193 // path, so its `orig` is the same code with the lines joined
3194 // by single spaces (indentation collapsed).
3195 // docling's parser has no line-preserving code path — its code
3196 // items carry the lines joined by single spaces. That flat
3197 // form is what every byte-conformance surface serializes
3198 // (legacy Markdown, JSON, DocLang); the line-preserving
3199 // extraction rides in `pretty` for strict Markdown only.
3200 let flat = code
3201 .lines()
3202 .map(str::trim)
3203 .filter(|l| !l.is_empty())
3204 .collect::<Vec<_>>()
3205 .join(" ");
3206 let node = match &enrichments[i] {
3207 Some(Enrichment::Code {
3208 language,
3209 text: enriched,
3210 }) => Node::Code {
3211 language: language.clone(),
3212 text: enriched.clone(),
3213 orig: Some(flat),
3214 pretty: None,
3215 },
3216 _ => Node::Code {
3217 language: None,
3218 text: flat,
3219 orig: None,
3220 pretty: Some(code),
3221 },
3222 };
3223 nodes.push(located(loc, node));
3224 // docling emits the `Listing N:` caption after the code block.
3225 if let Some(ci) = code_caption_for[i] {
3226 let cap = md_escape(®ion_texts[ci]);
3227 if !cap.is_empty() {
3228 nodes.push(Node::Paragraph { text: cap });
3229 }
3230 }
3231 }
3232 // text, caption, footnote → paragraph
3233 _ => {
3234 // docling parity (`PageAssembleModel._match_hyperlink`): when
3235 // link annotations cover ≥ half of the region's box, the
3236 // hyperlink attaches to the item and the legacy Markdown
3237 // serializer wraps its full text — 2206.01062's footnote URLs
3238 // render as `[1 https://…](https://…)`. Sparse in-paragraph
3239 // citation links stay below the 0.5 coverage threshold and
3240 // remain plain text, exactly like docling.
3241 //
3242 // Scope: **footnote regions only.** Upstream's page_assemble
3243 // matches every TEXT_ELEM label, but published docling
3244 // observably carries the hyperlink into the document only for
3245 // footnote items — in both committed groundtruth generations
3246 // (docling-JSON and Markdown, independent runs) the fully
3247 // covered plain-text DOI line of 2206.01062 page 1 has
3248 // `hyperlink: None` while the equally covered footnotes carry
3249 // theirs. The corpus is the conformance reference, so match
3250 // the observed behavior; widen the label set if a future
3251 // groundtruth refresh starts linking plain text too.
3252 let escaped = md_escape(&text);
3253 let hyperlink = (region.label == "footnote")
3254 .then(|| region_hyperlink(region, &page.links))
3255 .flatten();
3256 let text = match hyperlink {
3257 Some(uri) => {
3258 // The strict-mode anchor pairs this item covers are
3259 // superseded by the baked whole-item link.
3260 links.retain(|(anchor, href)| {
3261 !(href == &uri && region_texts[i].contains(anchor.as_str()))
3262 });
3263 format!("[{escaped}]({uri})")
3264 }
3265 None => escaped,
3266 };
3267 nodes.push(located(loc, Node::Paragraph { text }))
3268 }
3269 }
3270 }
3271 // A `/Rotate`-normalized scanned page (see `pdfium_backend`) was assembled
3272 // in upright space; rotate the finished geometry back so locations and the
3273 // page size are display-space, like docling and every viewer report them.
3274 if page.rotation != 0 {
3275 rotate_nodes_to_display(&mut nodes, page.rotation);
3276 }
3277 (nodes, links)
3278}
3279
3280/// One child of a picture as docling's `ReadingOrderModel._add_child_elements`
3281/// writes it under the `PictureItem`: a heading for a `section_header` /
3282/// `title` (upstream remaps title to section header), a list item for a
3283/// `list_item`, a furniture-layer text for a page header/footer, a caption,
3284/// otherwise a text item. `None` for a child that claimed no text.
3285fn picture_child_node(region: &Region, text: &str, loc: [u16; 4]) -> Option<Node> {
3286 if text.is_empty() {
3287 return None;
3288 }
3289 Some(match region.label {
3290 "title" | "section_header" => located(
3291 loc,
3292 Node::Heading {
3293 level: 2,
3294 text: md_escape(text),
3295 },
3296 ),
3297 // docling-core's `add_list_item` under a non-list parent opens a
3298 // list group per item, so every child item starts its own list;
3299 // `_add_child_elements` runs the marker processor on it too.
3300 "list_item" => list_item_node(text, loc, true),
3301 "page_header" | "page_footer" => Node::PageFurniture {
3302 footer: region.label == "page_footer",
3303 location: loc,
3304 text: md_escape(text),
3305 },
3306 "caption" => located(
3307 loc,
3308 Node::Caption {
3309 text: md_escape(text),
3310 href: None,
3311 },
3312 ),
3313 _ => located(
3314 loc,
3315 Node::Paragraph {
3316 text: md_escape(text),
3317 },
3318 ),
3319 })
3320}
3321
3322/// Rotate one 0–511 location bbox 90° clockwise on the grid (top-left origin):
3323/// `(x, y) → (511 - y, x)`.
3324fn rot_loc_cw(l: [u16; 4]) -> [u16; 4] {
3325 [511 - l[3], l[0], 511 - l[1], l[2]]
3326}
3327
3328/// Map upright-space geometry back to display space for a page whose `/Rotate`
3329/// was normalized away before inference: every `<location>` rotates `rot`°
3330/// clockwise on the 0–511 grid (the grid is per-axis normalized, so no page
3331/// dims are needed), and the `PageInfo` size returns to the display box. Node
3332/// text and order are untouched — reading order was decided upright, which is
3333/// the whole point.
3334fn rotate_nodes_to_display(nodes: &mut [Node], rot: u16) {
3335 let quarter_turns = (rot / 90) as usize;
3336 let rot_loc = |l: &mut [u16; 4]| {
3337 for _ in 0..quarter_turns {
3338 *l = rot_loc_cw(*l);
3339 }
3340 };
3341 fn walk(node: &mut Node, rot_loc: &impl Fn(&mut [u16; 4]), swap_dims: bool) {
3342 match node {
3343 Node::PageInfo { width, height, .. } => {
3344 if swap_dims {
3345 std::mem::swap(width, height);
3346 }
3347 }
3348 Node::Located { location, inner } => {
3349 rot_loc(location);
3350 walk(inner, rot_loc, swap_dims);
3351 }
3352 Node::Furniture { inner, .. } => walk(inner, rot_loc, swap_dims),
3353 Node::Group { children, .. } | Node::PictureChildren(children) => {
3354 for c in children {
3355 walk(c, rot_loc, swap_dims);
3356 }
3357 }
3358 Node::ListItem { location, .. }
3359 | Node::Formula { location, .. }
3360 | Node::Chart { location, .. } => {
3361 if let Some(l) = location {
3362 rot_loc(l);
3363 }
3364 }
3365 Node::PageFurniture { location, .. } => rot_loc(location),
3366 Node::Table(t) => {
3367 if let Some(l) = &mut t.location {
3368 rot_loc(l);
3369 }
3370 }
3371 _ => {}
3372 }
3373 }
3374 let swap_dims = quarter_turns % 2 == 1;
3375 for node in nodes {
3376 walk(node, &rot_loc, swap_dims);
3377 }
3378}
3379
3380/// Merge paragraph fragments split across a column or page break. docling joins a
3381/// paragraph whose previous fragment ends mid-sentence (a letter, not sentence
3382/// punctuation) with a lowercase continuation: `…definition of` + `lists in…` →
3383/// `…definition of lists in…`. The fragments are consecutive paragraphs, or
3384/// separated only by figure(s) the text wraps around: a column whose body flows
3385/// past a figure resumes below it (`…The wing type that is` ⟶[figure]⟶ `the most
3386/// common…`), and docling emits the whole paragraph before the figure. A heading,
3387/// table, or list between them ends the paragraph (no merge).
3388/// A paragraph that is really a figure/table caption (`Fig. 1. …`, `Table 2 …`).
3389/// Used to skip an unpaired caption when stitching a paragraph that wraps around
3390/// a figure.
3391fn looks_like_caption(text: &str) -> bool {
3392 let head: String = text.trim_start().chars().take(14).collect();
3393 (head.starts_with("Fig") || head.starts_with("Table"))
3394 && head.contains(|c: char| c.is_ascii_digit())
3395}
3396
3397/// A paragraph fragment is "open" — i.e. it might continue into the next
3398/// paragraph — when it ends mid-word (a letter) or with a wrap hyphen/dash.
3399/// docling joins `vocab-` + `ulary` → `vocab- ulary`.
3400fn paragraph_is_open(text: &str) -> bool {
3401 // docling's merge head test (`.+([a-z,\-\u00AD])\s*`): at least two chars,
3402 // ending in an ASCII lowercase letter, a comma, a hyphen, or a soft
3403 // hyphen. The comma matters: 2206's "…In phase four," resumes across the
3404 // page break. Uppercase/non-Latin endings do not merge, exactly as
3405 // upstream (the dash family is already `-` here — clean_text normalized).
3406 let t = text.trim_end();
3407 t.chars().count() >= 2
3408 && t.chars()
3409 .next_back()
3410 .is_some_and(|c| matches!(c, 'a'..='z' | ',' | '-' | '\u{ad}'))
3411}
3412
3413/// The paragraph text inside a node, looking through a [`Node::Located`]
3414/// provenance wrapper (PDF body paragraphs are wrapped since they carry a
3415/// `<location>`). Returns `None` for non-paragraph nodes.
3416fn as_paragraph(n: &Node) -> Option<&str> {
3417 match n {
3418 Node::Paragraph { text } => Some(text),
3419 Node::Located { inner, .. } => match inner.as_ref() {
3420 Node::Paragraph { text } => Some(text),
3421 _ => None,
3422 },
3423 _ => None,
3424 }
3425}
3426
3427/// Whether a node is a picture, looking through a [`Node::Located`] wrapper.
3428fn is_picture_node(n: &Node) -> bool {
3429 match n {
3430 Node::Picture { .. } => true,
3431 Node::Located { inner, .. } => matches!(inner.as_ref(), Node::Picture { .. }),
3432 _ => false,
3433 }
3434}
3435
3436/// A node a forward paragraph merge looks straight past: a figure or *table*
3437/// the text wraps around, or a page header/footer that falls between the two
3438/// fragments of a paragraph continuing across a page break (docling's merge
3439/// skip-labels: page_header, page_footer, table, picture, caption, footnote —
3440/// 2206's "…In phase four," resumes after a full caption+table+figure block).
3441fn is_merge_trailer(n: &Node) -> bool {
3442 is_picture_node(n)
3443 || matches!(
3444 n,
3445 Node::PageFurniture { .. }
3446 | Node::PageInfo { .. }
3447 | Node::Table(_)
3448 | Node::PictureChildren(_)
3449 )
3450 || matches!(n, Node::Located { inner, .. } if matches!(inner.as_ref(), Node::Table(_)))
3451 || as_paragraph(n).is_some_and(looks_like_caption)
3452}
3453
3454/// Rebuild node `i` as a paragraph with `text`, preserving its `<location>`
3455/// wrapper (and thus provenance) if it had one.
3456fn reparagraph(node: &Node, text: String) -> Node {
3457 match node {
3458 Node::Located { location, .. } => located(*location, Node::Paragraph { text }),
3459 _ => Node::Paragraph { text },
3460 }
3461}
3462
3463pub(crate) fn merge_continuations(nodes: &mut Vec<Node>) {
3464 let mut i = 0;
3465 while i + 1 < nodes.len() {
3466 let Some(a) = as_paragraph(&nodes[i]) else {
3467 i += 1;
3468 continue;
3469 };
3470 // A figure/table caption is a self-contained unit; body text resuming
3471 // after a figure is the continuation case, not the caption itself. Never
3472 // stitch *from* a caption — otherwise a caption that ends in a lone glyph
3473 // (`Fig. 5. … PubTabNet. μ`) would swallow a following stray figure label
3474 // (a standalone `μ`) into `… μ μ`.
3475 if looks_like_caption(a) {
3476 i += 1;
3477 continue;
3478 }
3479 if !paragraph_is_open(a) {
3480 i += 1;
3481 continue;
3482 }
3483 // The continuation is the next paragraph, looking past any figures the
3484 // text wraps around — and a figure/table caption that was emitted as its
3485 // own paragraph (an above-the-figure caption that didn't pair), since the
3486 // body text resumes after the whole figure+caption block.
3487 let mut j = i + 1;
3488 while nodes.get(j).is_some_and(is_merge_trailer) {
3489 j += 1;
3490 }
3491 // docling's continuation regex allows either case, but its merge runs
3492 // over the pre-assembly element stream; at node level an uppercase
3493 // start is overwhelmingly a new sentence/heading fragment (allowing it
3494 // swallowed 2305's formula blocks and redp's chapter openers), so the
3495 // continuation stays lowercase-start here.
3496 let cont = nodes.get(j).and_then(as_paragraph).is_some_and(|b| {
3497 b.trim_start()
3498 .chars()
3499 .next()
3500 .is_some_and(char::is_lowercase)
3501 });
3502 if cont {
3503 let a = as_paragraph(&nodes[i]).unwrap().trim_end().to_string();
3504 let b = as_paragraph(&nodes[j]).unwrap().trim_start().to_string();
3505 // A soft hyphen -- or a hard hyphen followed by a lowercase
3506 // continuation (guaranteed lowercase by the `cont` gate above) --
3507 // is a word split across the break: strip it and join without a
3508 // space, docling#3888 ("vocab-" + "ulary" -> "vocabulary");
3509 // docling's older serializer kept the artifact ("vocab- ulary").
3510 // Everything else joins with the space, as before.
3511 let merged = match a.strip_suffix('\u{ad}').or_else(|| a.strip_suffix('-')) {
3512 Some(stem) => format!("{stem}{b}"),
3513 None => format!("{a} {b}"),
3514 };
3515 // Keep node i's provenance wrapper; docling's merged paragraph keeps
3516 // the first fragment's geometry as its primary location.
3517 nodes[i] = reparagraph(&nodes[i], merged);
3518 nodes.remove(j);
3519 // Re-check i: the merged paragraph may continue further.
3520 } else {
3521 i += 1;
3522 }
3523 }
3524}
3525
3526/// How many leading nodes of `nodes` are safe to flush now — i.e. cannot be
3527/// rewritten by a future [`merge_continuations`] once more pages are appended.
3528///
3529/// A forward merge can only start from an "open" paragraph (ends mid-word) and
3530/// only reaches across trailing pictures and figure/table captions. So we scan
3531/// from the end past those skippable trailers: if the first non-skippable node is
3532/// an open paragraph, it (and the trailers after it) must be held; anything else —
3533/// a closed paragraph, a heading, a table, a list — blocks any forward merge, so
3534/// the whole buffer is safe to flush.
3535fn hold_start(nodes: &[Node]) -> usize {
3536 for k in (0..nodes.len()).rev() {
3537 // Skippable trailers (figures, page furniture, captions): a forward merge
3538 // looks straight past them.
3539 if is_merge_trailer(&nodes[k]) {
3540 continue;
3541 }
3542 match as_paragraph(&nodes[k]) {
3543 // An open body paragraph might still pull a continuation off the next
3544 // page — hold from here to the end.
3545 Some(text) if paragraph_is_open(text) => return k,
3546 // A closed paragraph, heading, table, list, etc. ends the paragraph:
3547 // nothing after it can merge backwards across it. Flush everything.
3548 _ => return nodes.len(),
3549 }
3550 }
3551 // Only skippable trailers (or empty) and no open paragraph to anchor a merge.
3552 nodes.len()
3553}
3554
3555/// Streaming counterpart of [`merge_continuations`]: feed per-page node batches in
3556/// document order and get back the prefix that is final (its cross-page merges are
3557/// resolved and no future page can change it), holding back only the small tail
3558/// that might still merge into the next page. Concatenating every flushed batch
3559/// (then [`finish`](Self::finish)) yields exactly the same nodes as running
3560/// [`merge_continuations`] once over the whole document.
3561pub(crate) struct StreamAssembler {
3562 pending: Vec<Node>,
3563}
3564
3565impl StreamAssembler {
3566 pub(crate) fn new() -> Self {
3567 Self {
3568 pending: Vec::new(),
3569 }
3570 }
3571
3572 /// Append one page's nodes, resolve merges within the buffer, and return the
3573 /// now-final prefix to emit (possibly empty).
3574 pub(crate) fn push(&mut self, mut nodes: Vec<Node>) -> Vec<Node> {
3575 self.pending.append(&mut nodes);
3576 merge_continuations(&mut self.pending);
3577 let cut = hold_start(&self.pending);
3578 let tail = self.pending.split_off(cut);
3579 std::mem::replace(&mut self.pending, tail)
3580 }
3581
3582 /// Flush whatever is left after the last page (the held tail is final once no
3583 /// more pages can follow).
3584 pub(crate) fn finish(self) -> Vec<Node> {
3585 self.pending
3586 }
3587}
3588
3589#[cfg(test)]
3590mod tests {
3591 use super::{cells_text, clean_text, merge_overlapping_regulars, reclaim_heading_body_footers};
3592
3593 /// docling drops a picture covering > 90 % of the page (its labels then
3594 /// read out as text); a dominant-but-not-full figure and any other label
3595 /// stay whatever their size.
3596 #[test]
3597 fn full_page_pictures_are_dropped_like_docling() {
3598 use super::drop_full_page_pictures;
3599 use crate::layout::Region;
3600 let region = |label: &'static str, l, t, r, b| Region {
3601 label,
3602 score: 0.99,
3603 l,
3604 t,
3605 r,
3606 b,
3607 };
3608 let mut regions = vec![
3609 region("picture", 0.0, 0.5, 478.9, 241.8),
3610 region("picture", 10.0, 10.0, 400.0, 200.0),
3611 region("table", 0.0, 0.0, 480.0, 243.0),
3612 region("text", 5.0, 5.0, 100.0, 20.0),
3613 ];
3614 drop_full_page_pictures(&mut regions, 480.75, 243.75);
3615 let labels: Vec<_> = regions.iter().map(|r| (r.label, r.l)).collect();
3616 assert_eq!(
3617 labels,
3618 vec![("picture", 10.0), ("table", 0.0), ("text", 5.0)]
3619 );
3620 }
3621 use super::{code_region_text, merge_continuations, resolve_link_anchors, StreamAssembler};
3622 use crate::layout::Region;
3623 use crate::pdfium_backend::{LinkAnnot, PdfPage, TextCell};
3624 use docling_core::Node;
3625
3626 /// The int8-layout guard's coverage metric: cells under detections count,
3627 /// cells outside don't, whitespace cells are ignored, and a cell-less page
3628 /// reads as fully covered (nothing to rescue).
3629 #[test]
3630 fn layout_cell_coverage_counts_claimed_text_cells() {
3631 let cell = |text: &str, l: f32, t: f32| TextCell {
3632 text: text.into(),
3633 l,
3634 t,
3635 r: l + 40.0,
3636 b: t + 10.0,
3637 };
3638 let region = Region {
3639 label: "text",
3640 score: 0.9,
3641 l: 0.0,
3642 t: 0.0,
3643 r: 100.0,
3644 b: 50.0,
3645 };
3646 let cells = vec![
3647 cell("inside", 10.0, 10.0),
3648 cell("also inside", 10.0, 30.0),
3649 cell("outside", 10.0, 200.0),
3650 cell(" ", 10.0, 210.0), // whitespace: not counted at all
3651 ];
3652 let cov = super::layout_cell_coverage(std::slice::from_ref(®ion), &cells);
3653 assert!((cov - 2.0 / 3.0).abs() < 1e-6, "got {cov}");
3654 assert_eq!(super::layout_cell_coverage(&[], &[]), 1.0);
3655 assert_eq!(super::layout_cell_coverage(&[], &cells), 0.0);
3656 }
3657
3658 /// #165: a picture no longer claims cells at 0.2 intersection-over-self.
3659 /// A line straddling the figure border (≤80 % contained) becomes an orphan
3660 /// region and is emitted as page text — before the fix its cells were
3661 /// silently erased. A line fully inside the picture is the picture's child
3662 /// (docling's `_set_cluster_children`): it survives the containment drop,
3663 /// leaves the page's reading order, and is written only under the picture
3664 /// in the JSON — never in the Markdown, like docling's picture serializer.
3665 #[test]
3666 fn border_straddlers_are_page_text_picture_interior_is_a_picture_child() {
3667 let pic = Region {
3668 label: "picture",
3669 score: 0.9,
3670 l: 0.0,
3671 t: 0.0,
3672 r: 100.0,
3673 b: 100.0,
3674 };
3675 // ~35 % of this cell overlaps the picture (l=90..120 of 0..100): above
3676 // the old 0.2 claim (was swallowed), below full containment (survives).
3677 let straddler = TextCell {
3678 text: "axis label".into(),
3679 l: 90.0,
3680 t: 40.0,
3681 r: 120.0,
3682 b: 48.0,
3683 };
3684 let interior = TextCell {
3685 text: "in-figure callout".into(),
3686 l: 10.0,
3687 t: 10.0,
3688 r: 60.0,
3689 b: 18.0,
3690 };
3691 let cells = vec![straddler, interior];
3692 let mut regions = vec![pic];
3693 super::add_orphan_regions(&mut regions, &cells);
3694 super::drop_contained_regulars(&mut regions);
3695 assert_eq!(
3696 regions.iter().filter(|r| r.label == "text").count(),
3697 2,
3698 "both unclaimed lines become orphans, and a picture swallows neither"
3699 );
3700 let parents = super::picture_parents(®ions);
3701 let parent_of = |l: f32| {
3702 regions
3703 .iter()
3704 .zip(&parents)
3705 .find(|(r, _)| r.label == "text" && r.l == l)
3706 .and_then(|(_, p)| *p)
3707 };
3708 assert_eq!(
3709 parent_of(10.0),
3710 Some(0),
3711 "the callout is the picture's child"
3712 );
3713 assert_eq!(parent_of(90.0), None, "the straddler is a page element");
3714
3715 let page = PdfPage::from_cells(200.0, 200.0, 2.0, cells);
3716 let n = regions.len();
3717 let (nodes, _) = super::assemble_page(&page, regions, &vec![None; n], &vec![None; n], None);
3718 let children: Vec<&Node> = nodes
3719 .iter()
3720 .filter_map(|n| match n {
3721 Node::PictureChildren(c) => Some(c),
3722 _ => None,
3723 })
3724 .flatten()
3725 .collect();
3726 assert!(
3727 matches!(children.as_slice(), [Node::Located { inner, .. }]
3728 if matches!(inner.as_ref(), Node::Paragraph { text } if text == "in-figure callout")),
3729 "{children:?}"
3730 );
3731 let mut doc = docling_core::DoclingDocument::new("t");
3732 doc.nodes = nodes;
3733 let md = doc.export_to_markdown();
3734 assert!(md.contains("axis label"), "{md}");
3735 assert!(!md.contains("in-figure callout"), "{md}");
3736 let json = doc.export_to_json_value();
3737 let pic = &json["pictures"][0];
3738 let child = pic["children"][0]["$ref"].as_str().expect("a child ref");
3739 let idx: usize = child.rsplit('/').next().unwrap().parse().unwrap();
3740 assert_eq!(json["texts"][idx]["text"], "in-figure callout");
3741 assert_eq!(json["texts"][idx]["parent"]["$ref"], "#/pictures/0");
3742 assert_eq!(json["texts"][idx]["content_layer"], "body");
3743 }
3744
3745 /// docling#3906's concern, pinned on our side: a picture detected fully
3746 /// inside a table region must survive the containment drop (upstream now
3747 /// attaches it to the table's cell; we keep it as a body sibling — either
3748 /// way it must not vanish). The text region inside the same table is the
3749 /// control: regulars are the ones the drop swallows.
3750 #[test]
3751 fn picture_inside_a_table_region_survives_the_containment_drop() {
3752 let mut regions = vec![
3753 region("table", 0.9, 0.0, 0.0, 200.0, 200.0),
3754 region("picture", 0.9, 20.0, 20.0, 120.0, 120.0),
3755 region("text", 0.9, 20.0, 140.0, 180.0, 180.0),
3756 ];
3757 super::drop_contained_regulars(&mut regions);
3758 let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
3759 assert_eq!(
3760 labels,
3761 ["table", "picture"],
3762 "the in-table picture stays; the in-table regular is the special's child"
3763 );
3764 }
3765
3766 /// Table–caption pairing (#265) is reading-order adjacency, docling's
3767 /// `_find_to_captions`: a caption binds the table directly next to it in
3768 /// the region sequence — above-caption and below-caption both work, and
3769 /// geometry is irrelevant (a same-page caption in the other column of a
3770 /// two-column layout is *not* adjacent, however close its box is). A
3771 /// caption with media on both sides, or separated from the table by a
3772 /// text paragraph, stays unattached.
3773 #[test]
3774 fn table_captions_pair_by_reading_order_adjacency() {
3775 // caption → table (above-caption), then table → caption (below-caption),
3776 // then a caption fenced off by a paragraph, then one between two tables.
3777 let regions = vec![
3778 region("text", 0.9, 0.0, 0.0, 100.0, 10.0), // 0 body text
3779 region("caption", 0.9, 0.0, 12.0, 60.0, 20.0), // 1 above-caption
3780 region("table", 0.9, 20.0, 22.0, 90.0, 60.0), // 2 ← pairs with 1
3781 region("table", 0.9, 0.0, 70.0, 100.0, 110.0), // 3 ← pairs with 4
3782 region("caption", 0.9, 0.0, 112.0, 60.0, 120.0), // 4 below-caption
3783 region("text", 0.9, 0.0, 130.0, 100.0, 140.0), // 5 body text
3784 region("caption", 0.9, 0.0, 142.0, 60.0, 150.0), // 6 fenced by 5/7
3785 region("text", 0.9, 0.0, 152.0, 100.0, 162.0), // 7 body text
3786 region("table", 0.9, 0.0, 170.0, 100.0, 200.0), // 8 unpaired
3787 region("caption", 0.9, 0.0, 202.0, 60.0, 210.0), // 9 ambiguous
3788 region("table", 0.9, 0.0, 212.0, 100.0, 240.0), // 10 unpaired
3789 ];
3790 let mut taken = vec![false; regions.len()];
3791 let pairs = super::pair_table_captions(®ions, &mut taken);
3792 assert_eq!(pairs[2], Some(1), "caption directly above its table pairs");
3793 assert_eq!(pairs[3], Some(4), "caption directly below its table pairs");
3794 assert_eq!(
3795 pairs[8], None,
3796 "a text paragraph between caption and table breaks the bond"
3797 );
3798 assert_eq!(
3799 pairs[10], None,
3800 "a caption between two tables is ambiguous and stays loose"
3801 );
3802 assert!(taken[1] && taken[4] && !taken[6] && !taken[9]);
3803 }
3804
3805 /// A colored terms-and-conditions panel detected as `picture` demotes into
3806 /// per-paragraph `text` regions (the blank line between C.7 and C.8 splits
3807 /// them); a chart whose only text is a few narrow axis labels keeps its
3808 /// crop untouched.
3809 #[test]
3810 fn text_panels_demote_to_paragraphs_but_charts_keep_their_crop() {
3811 let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3812 text: text.to_string(),
3813 l,
3814 t,
3815 r,
3816 b,
3817 };
3818 let panel = Region {
3819 label: "picture",
3820 score: 0.9,
3821 l: 0.0,
3822 t: 0.0,
3823 r: 100.0,
3824 b: 100.0,
3825 };
3826 // Three tight lines, a blank-line gap, two more: two paragraphs.
3827 let cells = vec![
3828 cell(
3829 "C.7. Wenn Sie diesen Vertrag widerrufen,",
3830 5.0,
3831 10.0,
3832 95.0,
3833 18.0,
3834 ),
3835 cell(
3836 "haben wir Ihnen alle Zahlungen, die wir",
3837 5.0,
3838 20.0,
3839 95.0,
3840 28.0,
3841 ),
3842 cell(
3843 "von Ihnen erhalten haben, zurückzuzahlen.",
3844 5.0,
3845 30.0,
3846 90.0,
3847 38.0,
3848 ),
3849 cell(
3850 "C.8. Wir können die Rückzahlung verweigern,",
3851 5.0,
3852 52.0,
3853 95.0,
3854 60.0,
3855 ),
3856 cell(
3857 "bis wir die Waren wieder zurückerhalten haben.",
3858 5.0,
3859 62.0,
3860 92.0,
3861 70.0,
3862 ),
3863 ];
3864 let mut regions = vec![panel.clone()];
3865 super::recover_text_panels(&mut regions, &cells);
3866 assert_eq!(
3867 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3868 ["text", "text"],
3869 "dense panel must demote into one text region per paragraph"
3870 );
3871 assert!(regions[0].b < regions[1].t, "paragraphs split at the gap");
3872 // Sparse narrow labels (a chart): picture survives.
3873 let labels = vec![
3874 cell("0", 5.0, 90.0, 8.0, 95.0),
3875 cell("50", 5.0, 50.0, 10.0, 55.0),
3876 cell("100", 5.0, 10.0, 12.0, 15.0),
3877 cell("t, s", 45.0, 96.0, 55.0, 100.0),
3878 ];
3879 let mut regions = vec![panel];
3880 super::recover_text_panels(&mut regions, &labels);
3881 assert_eq!(
3882 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3883 ["picture"]
3884 );
3885 }
3886
3887 /// An uncaptioned chart on a scanned page whose title, axis labels, and
3888 /// OCR boxes over the plot area are dense and wide enough to pass the
3889 /// coverage/width gates still keeps its crop: its line heights are ragged
3890 /// (title face vs tick labels vs bar-area OCR), failing the uniform-leading
3891 /// gate — a real text panel is set with constant leading (#173).
3892 #[test]
3893 fn dense_titled_chart_keeps_its_crop() {
3894 let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3895 text: text.to_string(),
3896 l,
3897 t,
3898 r,
3899 b,
3900 };
3901 let chart = Region {
3902 label: "picture",
3903 score: 0.9,
3904 l: 0.0,
3905 t: 0.0,
3906 r: 100.0,
3907 b: 100.0,
3908 };
3909 // Five wide lines at wildly different heights: a 12-pt title, 20-pt OCR
3910 // boxes over the bars, 4–5-pt tick/axis labels. Coverage and median
3911 // width both clear the panel thresholds.
3912 let cells = vec![
3913 cell("Underground Water Storage", 10.0, 5.0, 90.0, 17.0),
3914 cell("aquifer recharge zone", 15.0, 30.0, 75.0, 50.0),
3915 cell("confined | unconfined | perched", 12.0, 55.0, 80.0, 59.0),
3916 cell("saturated thickness", 8.0, 70.0, 60.0, 90.0),
3917 cell("distance from well, km", 20.0, 92.0, 85.0, 97.0),
3918 ];
3919 let mut regions = vec![chart];
3920 super::recover_text_panels(&mut regions, &cells);
3921 assert_eq!(
3922 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3923 ["picture"],
3924 "ragged line heights mark a figure, not a text panel"
3925 );
3926 }
3927
3928 /// docling serializes a cluster's cells in docling-parse index order
3929 /// (`_sort_cells`) and joins them with `PageAssembleModel.sanitize_text`:
3930 /// a space after every line except one ending in `-`, which either fuses a
3931 /// wrapped word (alnum on both sides — dash dropped) or glues verbatim (a
3932 /// bare `-` cell: `[0000` `-` `0002` → `[0000 -0002`, the 2305 ORCID line;
3933 /// `-` + `"C" cell -` + `a new table cell` → `-"C" cell a new table cell`,
3934 /// its OTSL list). Verified against the corpus: pure index order beats any
3935 /// geometric re-sort (normal_4pages' heading numerals paint after their
3936 /// text and belong last: `## 들어가며 1`).
3937 #[test]
3938 fn cells_join_in_index_order_with_sanitize_text_rules() {
3939 let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3940 text: text.to_string(),
3941 l,
3942 t,
3943 r,
3944 b,
3945 };
3946 let region = Region {
3947 label: "text",
3948 score: 1.0,
3949 l: 0.0,
3950 t: 95.0,
3951 r: 200.0,
3952 b: 130.0,
3953 };
3954 // ORCID superscript: a bare dash cell is a *detached* dash — kept, and
3955 // since docling#4052 (2.122) it joins with the ordinary space on both
3956 // sides (`[0000 -0002 -6960]` before that fix).
3957 let orcid = vec![
3958 cell("[0000", 10.0, 100.0, 30.0, 110.0),
3959 cell("−", 30.0, 100.0, 34.0, 110.0),
3960 cell("0002", 34.0, 100.0, 50.0, 110.0),
3961 cell("−", 50.0, 100.0, 54.0, 110.0),
3962 cell("6960]", 54.0, 100.0, 70.0, 110.0),
3963 ];
3964 assert_eq!(super::region_text(®ion, &orcid), "[0000 - 0002 - 6960]");
3965 // Wrapped word: dash dropped, lines fused (both boundary words alnum).
3966 let wrapped = vec![
3967 cell("platforms-", 10.0, 100.0, 60.0, 110.0),
3968 cell("reflects the design", 10.0, 112.0, 90.0, 122.0),
3969 ];
3970 assert_eq!(
3971 super::region_text(®ion, &wrapped),
3972 "platformsreflects the design"
3973 );
3974 // Dash-ending lines that are *detached* dashes (a bare bullet cell, a
3975 // `cell -` separator): the dash stays and the lines join with a space
3976 // — docling#4052; before it they glued (`-"C" cell a new table cell`,
3977 // 2305's OTSL list bullets).
3978 let otsl = vec![
3979 cell("–", 10.0, 100.0, 14.0, 110.0),
3980 cell("\"C\" cell -", 16.0, 100.0, 60.0, 110.0),
3981 cell("a new table cell", 10.0, 112.0, 80.0, 122.0),
3982 ];
3983 assert_eq!(
3984 super::region_text(®ion, &otsl),
3985 "- \"C\" cell - a new table cell"
3986 );
3987 // Index order is authoritative — no geometric re-sort.
3988 let numeral = vec![
3989 cell("들어가며", 30.0, 100.0, 80.0, 110.0),
3990 cell("1", 10.0, 98.0, 25.0, 112.0), // big numeral painted last
3991 ];
3992 assert_eq!(super::region_text(®ion, &numeral), "들어가며 1");
3993 }
3994
3995 /// The geometric-reliability gate, on the two shapes it has to tell apart.
3996 #[test]
3997 fn geometric_reliability_rejects_split_column_grids() {
3998 let g = |rows: &[&[&str]]| -> Vec<Vec<String>> {
3999 rows.iter()
4000 .map(|r| r.iter().map(|c| c.to_string()).collect())
4001 .collect()
4002 };
4003 // A genuine grid: dense, every column carrying entries. Nothing for
4004 // TableFormer to improve, so geometry is used as-is.
4005 assert!(super::geometric_table_is_reliable(&g(&[
4006 &["Datum", "Leistung", "Anzahl", "Kosten"],
4007 &["04.07", "Internet", "1", "40.30"],
4008 &["04.07", "Telefon", "2", "8.06"],
4009 ])));
4010 // The left-edge split artefact (the shape a scanned invoice produced):
4011 // one real label column plus values scattered across three sparse ones.
4012 assert!(!super::geometric_table_is_reliable(&g(&[
4013 &["www.magenta.at/faq", "", "", ""],
4014 &["Serviceteam", "", "", ""],
4015 &["Telefon", "0676/2000", "", ""],
4016 &["Kundennummer", "", "", "1.21699482"],
4017 &["Rechnungsnummer", "", "922769430725", ""],
4018 &["Rechnungsdatum", "", "", "04.07.2025"],
4019 ])));
4020 // A column only one row ever uses is a split artefact even when the
4021 // grid is otherwise dense.
4022 assert!(!super::geometric_table_is_reliable(&g(&[
4023 &["a", "b", ""],
4024 &["c", "d", ""],
4025 &["e", "f", "g"],
4026 ])));
4027 // Degenerate shapes are never vouched for — TableFormer may recover
4028 // structure a collapsed reconstruction lost.
4029 assert!(!super::geometric_table_is_reliable(&g(&[&[
4030 "only one column"
4031 ]])));
4032 assert!(!super::geometric_table_is_reliable(&[]));
4033 }
4034
4035 /// A `picture` region is cropped out of the rendered page, whatever built
4036 /// that page. The browser pipeline (#157) has no pdfium but does hand over
4037 /// the rasterized bitmap through `from_cells_with_image`, so it must get
4038 /// the same figure bytes the native path does — that is what makes
4039 /// `images = "embedded"` inline real pixels instead of a placeholder.
4040 #[cfg(feature = "ocr-prep")]
4041 #[test]
4042 fn picture_regions_are_cropped_from_a_host_supplied_page_image() {
4043 let mut img = image::RgbImage::new(200, 200);
4044 // Paint the figure area so the crop is distinguishable from the page.
4045 for y in 100..160 {
4046 for x in 20..120 {
4047 img.put_pixel(x, y, image::Rgb([255, 0, 0]));
4048 }
4049 }
4050 // scale 2.0: the region is in page points, the bitmap in pixels.
4051 let page = PdfPage::from_cells_with_image(100.0, 100.0, 2.0, Vec::new(), img);
4052 let region = Region {
4053 label: "picture",
4054 score: 0.9,
4055 l: 10.0,
4056 t: 50.0,
4057 r: 60.0,
4058 b: 80.0,
4059 };
4060 let (nodes, _) = super::assemble_page(&page, vec![region], &[None], &[None], None);
4061 // Layout-derived nodes carry provenance, so the picture arrives wrapped.
4062 let image = nodes
4063 .iter()
4064 .find_map(|n| match n {
4065 Node::Located { inner, .. } => match &**inner {
4066 Node::Picture { image, .. } => image.as_ref(),
4067 _ => None,
4068 },
4069 Node::Picture { image, .. } => image.as_ref(),
4070 _ => None,
4071 })
4072 .expect("a picture node with cropped pixels");
4073 assert_eq!(image.mimetype, "image/png");
4074 assert_eq!((image.width, image.height), (100, 60), "region × scale");
4075 assert!(!image.data.is_empty(), "PNG bytes were encoded");
4076 }
4077
4078 #[test]
4079 fn link_anchors_split_a_shared_word_cell_between_adjacent_links() {
4080 // A common header layout: one text run holds several pipe-separated
4081 // labels, each carrying its own link annotation. Every link must get
4082 // its own label as the anchor (and the "|" separators must belong to
4083 // none), not the whole run.
4084 let annot = |l: f32, r: f32, uri: &str| LinkAnnot {
4085 l,
4086 t: 100.0,
4087 r,
4088 b: 114.0,
4089 uri: uri.into(),
4090 };
4091 let page = PdfPage {
4092 width: 600.0,
4093 height: 800.0,
4094 scale: 2.0,
4095 cells: Vec::new(),
4096 code_cells: Vec::new(),
4097 // "LinkedIn | GitHub | Credly" = 26 chars over x 100..360.
4098 word_cells: vec![cell(
4099 "LinkedIn | GitHub | Credly",
4100 100.0,
4101 100.0,
4102 360.0,
4103 114.0,
4104 )],
4105 image: image::RgbImage::new(1, 1),
4106 image_layout: None,
4107 links: vec![
4108 annot(100.0, 180.0, "https://l"),
4109 annot(200.0, 260.0, "https://g"),
4110 annot(290.0, 360.0, "https://c"),
4111 ],
4112 rotation: 0,
4113 };
4114 assert_eq!(
4115 resolve_link_anchors(&page),
4116 vec![
4117 ("LinkedIn".to_string(), "https://l".to_string()),
4118 ("GitHub".to_string(), "https://g".to_string()),
4119 ("Credly".to_string(), "https://c".to_string()),
4120 ]
4121 );
4122 }
4123
4124 /// A one-line code cell at `[l, r] × [t, b]` (top-left coords).
4125 fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
4126 TextCell {
4127 text: text.into(),
4128 l,
4129 t,
4130 r,
4131 b,
4132 }
4133 }
4134
4135 /// OCR-path grouping (docling's `_remove_overlapping_clusters("regular")`):
4136 /// the low-score paragraph box RT-DETR draws over its own high-score line
4137 /// boxes collapses to one region — the group's union, with the survivor's
4138 /// label and score — so region-scoped OCR reads each line once. Regions
4139 /// that merely sit near each other, and specials, are untouched.
4140 #[test]
4141 fn merge_overlapping_regulars_collapses_a_block_over_its_lines() {
4142 let mut regions = vec![
4143 region("text", 0.84, 60.0, 186.0, 270.0, 198.0),
4144 region("text", 0.80, 60.0, 160.0, 294.0, 172.0),
4145 region("text", 0.79, 59.0, 107.0, 272.0, 119.0),
4146 // The paragraph box, lower score, containing all three lines.
4147 region("text", 0.52, 59.0, 107.0, 295.0, 200.0),
4148 // Elsewhere on the page: stays as is.
4149 region("section_header", 0.77, 60.0, 71.0, 253.0, 86.0),
4150 // A picture the block overlaps is not a regular — never grouped.
4151 region("picture", 0.9, 50.0, 100.0, 300.0, 210.0),
4152 ];
4153 merge_overlapping_regulars(&mut regions);
4154 assert_eq!(regions.len(), 3, "{regions:?}");
4155 let block = regions
4156 .iter()
4157 .find(|r| r.label == "text")
4158 .expect("one text");
4159 // docling keeps the largest passing candidate unless a rival is both
4160 // comparable in size and > 0.05 more confident; the 16× larger block
4161 // passes, and a smaller line never replaces a larger current best.
4162 // Either way the survivor spans the whole group.
4163 assert_eq!(
4164 (block.l, block.t, block.r, block.b),
4165 (59.0, 107.0, 295.0, 200.0)
4166 );
4167 assert!(regions.iter().any(|r| r.label == "section_header"));
4168 assert!(regions.iter().any(|r| r.label == "picture"));
4169 }
4170
4171 /// The pairwise rules, each in the arrangement where it decides the
4172 /// outcome: docling seeds the survivor with the group's first passing
4173 /// cluster and a later one replaces it only when larger *and* within
4174 /// 0.05 confidence, so a rule that merely lets a cluster pass matters
4175 /// exactly when that cluster comes first — a same-sized list item ahead
4176 /// of a far more confident text box, a code box ahead of the text it
4177 /// contains. Without the rule either would be rejected outright (similar
4178 /// size, rival > 0.05 more confident) and the text box would win.
4179 #[test]
4180 fn merge_overlapping_regulars_follows_the_preference_rules() {
4181 let mut regions = vec![
4182 region("list_item", 0.6, 0.0, 0.0, 102.0, 20.0),
4183 region("text", 0.9, 0.0, 0.0, 100.0, 20.0),
4184 ];
4185 merge_overlapping_regulars(&mut regions);
4186 assert_eq!(regions.len(), 1);
4187 assert_eq!(regions[0].label, "list_item");
4188
4189 let mut regions = vec![
4190 region("code", 0.6, 0.0, 0.0, 100.0, 100.0),
4191 region("text", 0.9, 2.0, 2.0, 98.0, 98.0),
4192 ];
4193 merge_overlapping_regulars(&mut regions);
4194 assert_eq!(regions.len(), 1);
4195 assert_eq!(regions[0].label, "code");
4196
4197 // No rule applies: a near-identical rival that is > 0.05 more
4198 // confident rejects the candidate whatever the order.
4199 for order in [[0.9, 0.6], [0.6, 0.9]] {
4200 let mut regions = vec![
4201 region("text", order[0], 0.0, 0.0, 100.0, 20.0),
4202 region("text", order[1], 0.0, 0.0, 105.0, 21.0),
4203 ];
4204 merge_overlapping_regulars(&mut regions);
4205 assert_eq!(regions.len(), 1);
4206 assert_eq!(regions[0].score, 0.9, "the confident twin wins");
4207 assert_eq!(
4208 (regions[0].r, regions[0].b),
4209 (105.0, 21.0),
4210 "on the union box"
4211 );
4212 }
4213
4214 // Side by side (no containment, IoU 0): nothing to merge.
4215 let mut regions = vec![
4216 region("text", 0.9, 0.0, 0.0, 100.0, 20.0),
4217 region("text", 0.9, 0.0, 22.0, 100.0, 42.0),
4218 ];
4219 merge_overlapping_regulars(&mut regions);
4220 assert_eq!(regions.len(), 2);
4221 }
4222
4223 #[test]
4224 fn footer_under_a_body_less_heading_is_its_text() {
4225 // The reporting CV's last page: `## Languages` at t=773.7..783.4 and
4226 // its one-line body at 798.5..808.7, labelled page_footer (A4, 595 pt).
4227 let mut regions = vec![
4228 region("list_item", 0.9, 66.0, 754.0, 353.0, 765.0),
4229 region("section_header", 0.94, 43.0, 773.7, 89.5, 783.4),
4230 region("page_footer", 0.88, 43.0, 798.5, 527.8, 808.7),
4231 ];
4232 reclaim_heading_body_footers(&mut regions, 595.28);
4233 assert_eq!(regions[2].label, "text");
4234 assert_eq!(regions[1].label, "section_header");
4235
4236 // A heading with its own paragraph and a running footer below: kept.
4237 let mut regions = vec![
4238 region("section_header", 0.9, 43.0, 700.0, 120.0, 710.0),
4239 region("text", 0.9, 43.0, 714.0, 520.0, 780.0),
4240 region("page_footer", 0.9, 43.0, 798.0, 520.0, 808.0),
4241 ];
4242 reclaim_heading_body_footers(&mut regions, 595.28);
4243 assert_eq!(regions[2].label, "page_footer");
4244
4245 // A page number under a trailing heading is too narrow to be a body.
4246 let mut regions = vec![
4247 region("section_header", 0.9, 43.0, 773.0, 120.0, 783.0),
4248 region("page_footer", 0.9, 280.0, 798.0, 300.0, 808.0),
4249 ];
4250 reclaim_heading_body_footers(&mut regions, 595.28);
4251 assert_eq!(regions[1].label, "page_footer");
4252
4253 // Too far below the heading (a real footer after a heading that ends
4254 // the page): kept.
4255 let mut regions = vec![
4256 region("section_header", 0.9, 43.0, 700.0, 120.0, 710.0),
4257 region("page_footer", 0.9, 43.0, 798.0, 520.0, 808.0),
4258 ];
4259 reclaim_heading_body_footers(&mut regions, 595.28);
4260 assert_eq!(regions[1].label, "page_footer");
4261 }
4262
4263 fn region(label: &'static str, score: f32, l: f32, t: f32, r: f32, b: f32) -> Region {
4264 Region {
4265 label,
4266 score,
4267 l,
4268 t,
4269 r,
4270 b,
4271 }
4272 }
4273
4274 #[test]
4275 fn resolve_collapses_nested_code_keeping_the_larger_box() {
4276 // A tight high-score `code` box and a taller lower-score near-duplicate that
4277 // contains it must collapse to one — the *larger* box, so every cell stays
4278 // covered and nothing leaks out as orphan text.
4279 let tight = region("code", 0.95, 78.0, 292.0, 300.0, 330.0);
4280 let wide = region("code", 0.66, 63.0, 260.0, 320.0, 346.0);
4281 let kept = super::resolve(vec![tight, wide]);
4282 assert_eq!(kept.len(), 1, "nested code boxes must collapse to one");
4283 assert!(
4284 kept[0].l == 63.0 && kept[0].b == 346.0,
4285 "the larger containing box is kept"
4286 );
4287 }
4288
4289 #[test]
4290 fn resolve_keeps_distinct_and_differently_typed_regions() {
4291 // A text box fully inside a lower-score *table* must NOT be collapsed (the
4292 // code dedup is code-only), and two separate code blocks stay separate.
4293 let text = region("text", 0.95, 90.0, 210.0, 200.0, 230.0);
4294 let table = region("table", 0.60, 80.0, 200.0, 400.0, 500.0);
4295 assert_eq!(super::resolve(vec![text, table]).len(), 2);
4296
4297 let code_a = region("code", 0.9, 78.0, 100.0, 300.0, 140.0);
4298 let code_b = region("code", 0.9, 78.0, 300.0, 300.0, 360.0); // far below, no overlap
4299 assert_eq!(super::resolve(vec![code_a, code_b]).len(), 2);
4300 }
4301
4302 /// A two-column glossary page came out as three column
4303 /// tables *and* one low-score whole-page table over them. docling's wrapper
4304 /// `_remove_overlapping_clusters` keeps one table per overlapping group
4305 /// (here the whole-page one: > 2× every rival's area and ≤ 0.2 less
4306 /// confident than the running best); `greedy` alone kept all four and
4307 /// emitted every cell twice.
4308 #[test]
4309 fn resolve_keeps_one_table_per_nested_group() {
4310 let kept = super::resolve(vec![
4311 region("table", 0.71, 26.0, 203.0, 183.0, 558.0),
4312 region("table", 0.67, 26.0, 55.0, 183.0, 196.0),
4313 region("table", 0.66, 196.0, 56.0, 354.0, 561.0),
4314 region("table", 0.53, 25.0, 53.0, 354.0, 561.0),
4315 ]);
4316 assert_eq!(kept.len(), 1, "one survivor per overlapping group");
4317 assert_eq!((kept[0].l, kept[0].b), (25.0, 561.0));
4318 // Side-by-side tables that don't overlap stay separate.
4319 let kept = super::resolve(vec![
4320 region("table", 0.9, 26.0, 55.0, 183.0, 558.0),
4321 region("table", 0.9, 196.0, 56.0, 354.0, 561.0),
4322 ]);
4323 assert_eq!(kept.len(), 2);
4324 }
4325
4326 /// A dense data table detected as both picture (0.80) and table (0.62) on one box.
4327 /// The picture is ≥ 0.1 more confident, so `_handle_cross_type_overlaps`
4328 /// keeps both, and the dense table text passed the text-panel gates: the
4329 /// demoted paragraph repeated every cell the table grid renders. A
4330 /// paragraph > 80 % inside a surviving table is the table's child and is
4331 /// not emitted; a panel with no table under it still demotes.
4332 #[test]
4333 fn text_panel_over_a_table_does_not_repeat_its_cells() {
4334 let lines = |t0: f32| -> Vec<TextCell> {
4335 (0..4)
4336 .map(|i| {
4337 let t = t0 + 10.0 * i as f32;
4338 cell("1 2 3 4 5 6 7 8 9 10 11 12", 5.0, t, 95.0, t + 8.0)
4339 })
4340 .collect()
4341 };
4342 let mut cells = lines(0.0);
4343 cells.extend(lines(200.0));
4344 let mut regions = vec![
4345 region("picture", 0.80, 0.0, 0.0, 100.0, 45.0),
4346 region("table", 0.62, 0.0, 0.0, 100.0, 45.0),
4347 region("picture", 0.80, 0.0, 200.0, 100.0, 245.0),
4348 ];
4349 super::recover_text_panels(&mut regions, &cells);
4350 assert_eq!(
4351 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
4352 ["table", "text"]
4353 );
4354 assert_eq!(regions[1].t, 200.0, "the table-free panel still demotes");
4355 }
4356
4357 /// docling's `ListItemMarkerProcessor` + docling-core's Markdown list rules
4358 /// (see `list_item_node`): a bullet glyph or `-` is split off and the
4359 /// serializer's own `-` printed (2305's OTSL list reads `- "C" cell`, not
4360 /// `- - "C" cell`); `N.` is an ordered item; any other marker with a letter
4361 /// or digit stays in the text behind the bullet; no marker → plain bullet.
4362 #[test]
4363 fn list_item_markers_split_like_docling() {
4364 let text_of = |n: &Node| match n {
4365 Node::ListItem {
4366 ordered,
4367 number,
4368 text,
4369 marker,
4370 ..
4371 } => (*ordered, *number, text.clone(), marker.clone()),
4372 other => panic!("{other:?}"),
4373 };
4374 let loc = [0, 0, 100, 10];
4375 assert_eq!(
4376 text_of(&super::list_item_node(
4377 "- \"C\" cell - a new table cell",
4378 loc,
4379 false
4380 )),
4381 (
4382 false,
4383 0,
4384 "\"C\" cell - a new table cell".into(),
4385 Some("-".into())
4386 )
4387 );
4388 assert_eq!(
4389 text_of(&super::list_item_node("• Bullet text", loc, false)),
4390 (false, 0, "Bullet text".into(), Some("•".into()))
4391 );
4392 assert_eq!(
4393 text_of(&super::list_item_node("3. Third step", loc, false)),
4394 (true, 3, "Third step".into(), Some("3.".into()))
4395 );
4396 assert_eq!(
4397 text_of(&super::list_item_node("a) Option", loc, false)),
4398 (false, 0, "a) Option".into(), Some("a)".into()))
4399 );
4400 assert_eq!(
4401 text_of(&super::list_item_node("1.2 Nested outline", loc, false)),
4402 (false, 0, "1.2 Nested outline".into(), Some("1.2".into()))
4403 );
4404 // 2203's `3.a. If all IOU scores…`: a compound marker is not an item
4405 // number — docling prints `- 3.a. If all…`.
4406 assert_eq!(
4407 text_of(&super::list_item_node("3.a. If all IOU scores", loc, false)),
4408 (
4409 false,
4410 0,
4411 "3.a. If all IOU scores".into(),
4412 Some("3.a.".into())
4413 )
4414 );
4415 // A glued symbol-font bullet is stripped, a spaced one is the marker.
4416 assert_eq!(
4417 text_of(&super::list_item_node("•Glued", loc, false)),
4418 (false, 0, "Glued".into(), Some("·".into()))
4419 );
4420 // No whitespace after the glyph → not a marker (docling's `\s` is required).
4421 assert_eq!(
4422 text_of(&super::list_item_node("-5 degrees", loc, false)),
4423 (false, 0, "-5 degrees".into(), Some("·".into()))
4424 );
4425 // The remaining numbered shapes, first-wins like docling's list.
4426 for (input, marker, body) in [
4427 ("1.2.3. Deep", "1.2.3.", "Deep"),
4428 ("9a) Nine-a", "9a)", "Nine-a"),
4429 ("(3.a) Paren", "(3.a)", "Paren"),
4430 ("12) Twelve", "12)", "Twelve"),
4431 ("(4) Four", "(4)", "Four"),
4432 ("[7] Seven", "[7]", "Seven"),
4433 ("iv. Roman", "iv.", "Roman"),
4434 ("IX. Roman", "IX.", "Roman"),
4435 ("b. Letter", "b.", "Letter"),
4436 ("B) Letter", "B)", "Letter"),
4437 ] {
4438 assert_eq!(
4439 super::split_list_marker(input),
4440 Some((marker, body, true)),
4441 "{input}"
4442 );
4443 }
4444 // A `1.2.` whose optional dot would eat the separator backtracks like
4445 // Python's regex; a marker with nothing after the whitespace is none.
4446 assert_eq!(
4447 super::split_list_marker("1.2.\tx"),
4448 Some(("1.2.", "x", true))
4449 );
4450 assert_eq!(super::split_list_marker("1. "), None);
4451 assert_eq!(super::split_list_marker("• "), None);
4452 assert_eq!(
4453 text_of(&super::list_item_node("Plain item", loc, false)),
4454 (false, 0, "Plain item".into(), Some("·".into()))
4455 );
4456 }
4457
4458 #[test]
4459 fn code_language_label_above_code_is_detected() {
4460 // A bare "XML" token directly above a code box is a language label; a real
4461 // heading above the same code is not; a language word with no code below is
4462 // left alone.
4463 let label = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
4464 let code = region("code", 0.7, 77.0, 552.0, 290.0, 640.0);
4465 let heading = region("section_header", 0.9, 76.0, 500.0, 260.0, 512.0);
4466 let cells = vec![
4467 cell("XML", 78.0, 541.0, 94.0, 548.0), // inside `label`
4468 cell("Overview", 78.0, 501.0, 250.0, 511.0), // inside `heading`
4469 ];
4470 let drop = super::code_language_labels(&[label, code, heading], &cells);
4471 assert_eq!(drop, vec![true, false, false], "only the label is consumed");
4472
4473 // Same label with no code region present → not consumed.
4474 let label2 = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
4475 let only = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
4476 assert_eq!(super::code_language_labels(&[label2], &only), vec![false]);
4477
4478 // A label swallowed into the top of a wider code box (negative gap) is still
4479 // recognized.
4480 let inside_lbl = region("text", 0.9, 76.0, 540.0, 96.0, 549.0);
4481 let wide_code = region("code", 0.7, 63.0, 531.0, 320.0, 654.0);
4482 let cells2 = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
4483 assert_eq!(
4484 super::code_language_labels(&[inside_lbl, wide_code], &cells2),
4485 vec![true, false]
4486 );
4487
4488 assert!(super::is_code_language("XML") && super::is_code_language("c#"));
4489 assert!(!super::is_code_language("Configure") && !super::is_code_language("XML schema"));
4490 }
4491
4492 #[test]
4493 fn code_region_text_keeps_lines_and_indentation() {
4494 // Three source lines; each glyph is 6 units wide (width / chars = 6), so the
4495 // `int X;` line indented to x=22 is (22-10)/6 = 2 spaces in.
4496 let region = Region {
4497 label: "code",
4498 score: 1.0,
4499 l: 0.0,
4500 t: -5.0,
4501 r: 100.0,
4502 b: 40.0,
4503 };
4504 let cells = vec![
4505 cell("struct P {", 10.0, 0.0, 70.0, 10.0),
4506 cell("int X;", 22.0, 12.0, 58.0, 22.0),
4507 cell("}", 10.0, 24.0, 16.0, 34.0),
4508 ];
4509 assert_eq!(code_region_text(®ion, &cells), "struct P {\n int X;\n}");
4510 }
4511
4512 #[test]
4513 fn code_region_text_tightens_punctuation_without_eating_indentation() {
4514 // A fluent `.Foo()` line at x=22 (2 chars in). Per-line tightening must not
4515 // consume the leading indent space by matching " ." across it.
4516 let region = Region {
4517 label: "code",
4518 score: 1.0,
4519 l: 0.0,
4520 t: -5.0,
4521 r: 100.0,
4522 b: 40.0,
4523 };
4524 let cells = vec![
4525 cell("builder", 10.0, 0.0, 52.0, 10.0),
4526 // pdfium spaced the call: ".Foo (x)" tightens to ".Foo(x)", still 2-indented.
4527 cell(".Foo (x)", 22.0, 12.0, 70.0, 22.0),
4528 ];
4529 assert_eq!(code_region_text(®ion, &cells), "builder\n .Foo(x)");
4530 }
4531
4532 #[test]
4533 fn code_region_text_orders_out_of_order_cells_and_ignores_blank_lines() {
4534 let region = Region {
4535 label: "code",
4536 score: 1.0,
4537 l: 0.0,
4538 t: -5.0,
4539 r: 100.0,
4540 b: 60.0,
4541 };
4542 // Fed bottom-up and with a whitespace-only cell; output is top-down, no blank.
4543 let cells = vec![
4544 cell("b();", 10.0, 24.0, 34.0, 34.0),
4545 cell(" ", 10.0, 12.0, 20.0, 22.0),
4546 cell("a();", 10.0, 0.0, 34.0, 10.0),
4547 ];
4548 assert_eq!(code_region_text(®ion, &cells), "a();\nb();");
4549 // No code cells → empty, so the caller falls back to the prose text.
4550 assert_eq!(code_region_text(®ion, &[]), "");
4551 }
4552
4553 fn para(text: &str) -> Node {
4554 Node::Paragraph { text: text.into() }
4555 }
4556
4557 /// Run a node sequence through [`StreamAssembler`] with the given page splits
4558 /// and assert the flushed result equals one-shot [`merge_continuations`].
4559 fn assert_stream_eq(nodes: &[Node], splits: &[usize]) {
4560 let mut want = nodes.to_vec();
4561 merge_continuations(&mut want);
4562
4563 let mut asm = StreamAssembler::new();
4564 let mut got = Vec::new();
4565 let mut start = 0;
4566 for &end in splits {
4567 got.extend(asm.push(nodes[start..end].to_vec()));
4568 start = end;
4569 }
4570 got.extend(asm.push(nodes[start..].to_vec()));
4571 got.extend(asm.finish());
4572 assert_eq!(got, want, "stream assembly diverged (splits={splits:?})");
4573 }
4574
4575 #[test]
4576 fn stream_assembler_matches_merge_continuations() {
4577 // Open fragment + lowercase continuation split across a page boundary.
4578 let cross = [para("the definition of"), para("lists in scope")];
4579 assert_stream_eq(&cross, &[1]);
4580 assert_stream_eq(&cross, &[]);
4581
4582 // Continuation that wraps around a figure (+ its caption) on the boundary.
4583 let wrap = [
4584 para("the wing type that is"),
4585 Node::Picture {
4586 caption: None,
4587 caption_href: None,
4588 image: None,
4589 classification: None,
4590 caption_parent: Default::default(),
4591 },
4592 para("Fig. 1. a diagram"),
4593 para("the most common kind"),
4594 ];
4595 for splits in [&[][..], &[1][..], &[2][..], &[3][..], &[1, 3][..]] {
4596 assert_stream_eq(&wrap, splits);
4597 }
4598
4599 // A heading between fragments blocks the merge (must still flush correctly).
4600 let blocked = [
4601 para("ends mid word and"),
4602 Node::Heading {
4603 level: 2,
4604 text: "New Section".into(),
4605 },
4606 para("more body here"),
4607 ];
4608 for splits in [&[][..], &[1][..], &[2][..]] {
4609 assert_stream_eq(&blocked, splits);
4610 }
4611
4612 // A chain across three pages: each page is one open lowercase fragment.
4613 let chain = [
4614 para("alpha beta"),
4615 para("gamma delta"),
4616 para("epsilon zeta"),
4617 ];
4618 assert_stream_eq(&chain, &[1, 2]);
4619 }
4620
4621 #[test]
4622 fn clean_text_dehyphenates_and_normalizes_typography() {
4623 // U+0002 line-wrap hyphen + the join space → merged word (like docling).
4624 assert_eq!(clean_text("com\u{2} pact"), "compact");
4625 assert_eq!(clean_text("end-to\u{2} end deep"), "end-toend deep");
4626 // A stray wrap hyphen (no following join) is dropped.
4627 assert_eq!(clean_text("word\u{2}"), "word");
4628 // Typographic punctuation → ASCII: every curly quote becomes `'`
4629 // (docling-parse's sanitizer table), a literal `"` stays.
4630 assert_eq!(
4631 clean_text("Graph\u{2019}s \u{201c}x\u{201d} \"y\""),
4632 "Graph's 'x' \"y\""
4633 );
4634 assert_eq!(clean_text("a\u{2026}"), "a...");
4635 // The docling-parse sanitizer's internal spacing is preserved as
4636 // placed; line breaks/tabs normalize to a space, ends trim.
4637 assert_eq!(clean_text("a b\nc"), "a b c");
4638 }
4639
4640 /// docling#4064: a form's children are emitted together where the form
4641 /// sits in the top-level order, not interleaved with surrounding text.
4642 #[test]
4643 fn form_children_stay_together_in_reading_order() {
4644 let reg = |label: &'static str, l: f32, t: f32, r: f32, b: f32| Region {
4645 label,
4646 score: 0.9,
4647 l,
4648 t,
4649 r,
4650 b,
4651 };
4652 // Page: intro text, then a form spanning the left column with two
4653 // fields and a table inside, while a right-column paragraph sits
4654 // level with the form's first field (it would otherwise be read
4655 // between the form's children).
4656 let mut items = vec![
4657 reg("text", 50.0, 50.0, 550.0, 70.0), // 0 intro
4658 reg("form", 50.0, 100.0, 300.0, 400.0), // 1 container
4659 reg("text", 60.0, 110.0, 290.0, 130.0), // 2 field A (child)
4660 reg("text", 320.0, 110.0, 550.0, 130.0), // 3 right column paragraph
4661 reg("table", 60.0, 150.0, 290.0, 300.0), // 4 table (child)
4662 reg("text", 60.0, 320.0, 290.0, 340.0), // 5 field B (child)
4663 reg("text", 50.0, 450.0, 550.0, 470.0), // 6 outro
4664 ];
4665 let cids = super::cluster_cids(&items, &[]);
4666 super::order_with_containers(&mut items, &cids, 600.0, 800.0, |r| r);
4667 let order: Vec<(&str, f32)> = items.iter().map(|r| (r.label, r.t)).collect();
4668 // The form block (container, then its children top-down) is one unit.
4669 let form_pos = order.iter().position(|(l, _)| *l == "form").unwrap();
4670 assert_eq!(
4671 &order[form_pos..form_pos + 4],
4672 &[
4673 ("form", 100.0),
4674 ("text", 110.0),
4675 ("table", 150.0),
4676 ("text", 320.0)
4677 ]
4678 );
4679 assert_eq!(order[0], ("text", 50.0));
4680 assert_eq!(order[order.len() - 1], ("text", 450.0));
4681 // Without a container the plain order interleaves by geometry.
4682 let mut flat: Vec<Region> = items
4683 .iter()
4684 .filter(|r| r.label != "form")
4685 .cloned()
4686 .collect();
4687 let cids = super::cluster_cids(&flat, &[]);
4688 super::order_regions(&mut flat, &cids, 600.0, 800.0, |r| r);
4689 assert_ne!(
4690 flat.iter().map(|r| r.t).collect::<Vec<_>>(),
4691 order
4692 .iter()
4693 .filter(|(l, _)| *l != "form")
4694 .map(|(_, t)| *t)
4695 .collect::<Vec<_>>()
4696 );
4697 }
4698
4699 /// docling#3906: a picture inside a table lands in the covering cell,
4700 /// chosen by the picture's inferred grid position when cell boxes overlap.
4701 #[test]
4702 fn picture_matches_the_cell_at_its_grid_position() {
4703 let cell = |r: usize, c: usize, bbox: [f32; 4]| docling_core::TableCell {
4704 text: format!("r{r}c{c}"),
4705 bbox: Some(bbox),
4706 start_row: r,
4707 start_col: c,
4708 row_span: 1,
4709 col_span: 1,
4710 column_header: false,
4711 row_header: false,
4712 row_section: false,
4713 };
4714 // 2×2 grid; the (1,0) cell box is generous and also covers the picture.
4715 let cells = vec![
4716 cell(0, 0, [0.0, 0.0, 100.0, 50.0]),
4717 cell(0, 1, [100.0, 0.0, 200.0, 50.0]),
4718 cell(1, 0, [0.0, 50.0, 100.0, 100.0]),
4719 cell(1, 1, [100.0, 50.0, 200.0, 100.0]),
4720 ];
4721 let pic = Region {
4722 label: "picture",
4723 score: 0.9,
4724 l: 110.0,
4725 t: 60.0,
4726 r: 190.0,
4727 b: 95.0,
4728 };
4729 assert_eq!(super::match_picture_to_cell(&pic, &cells), Some((1.0, 3)));
4730 // A picture only half inside any cell is not nested.
4731 let straddling = Region {
4732 label: "picture",
4733 score: 0.9,
4734 l: 60.0,
4735 t: 60.0,
4736 r: 160.0,
4737 b: 95.0,
4738 };
4739 assert_eq!(super::match_picture_to_cell(&straddling, &cells), None);
4740 }
4741
4742 /// docling#4052 (2.122): a line-final dash fuses the wrapped word only
4743 /// when attached to it; a detached dash is a literal and the lines join
4744 /// with a space.
4745 #[test]
4746 fn line_final_hyphen_fuses_only_when_attached_to_a_word() {
4747 let line = |text: &str, t: f32| TextCell {
4748 text: text.to_string(),
4749 l: 0.0,
4750 t,
4751 r: 100.0,
4752 b: t + 10.0,
4753 };
4754 // `algo-` / `rithms`: attached hyphen, alnum on both sides → fused.
4755 assert_eq!(
4756 cells_text(vec![&line("algo-", 0.0), &line("rithms", 12.0)]),
4757 "algorithms"
4758 );
4759 // `pp. 545-` / `561`: attached, digits count as alnum → `545561` (upstream).
4760 assert_eq!(
4761 cells_text(vec![&line("pp. 545-", 0.0), &line("561", 12.0)]),
4762 "pp. 545561"
4763 );
4764 // A dash after whitespace — a separator or a lone `-` cell — is kept and
4765 // the lines take the ordinary joining space.
4766 assert_eq!(
4767 cells_text(vec![&line("range -", 0.0), &line("wide", 12.0)]),
4768 "range - wide"
4769 );
4770 assert_eq!(
4771 cells_text(vec![&line("-", 0.0), &line("item", 12.0)]),
4772 "- item"
4773 );
4774 // Attached but the next line opens with no word (`x-` / `...`): dash
4775 // kept and, as before, no separating space.
4776 assert_eq!(
4777 cells_text(vec![&line("x-", 0.0), &line("...", 12.0)]),
4778 "x-..."
4779 );
4780 }
4781
4782 #[test]
4783 fn lam_alef_only_swaps_a_genuinely_reversed_ligature() {
4784 // A mid-word `alef-variant + lam` is pdfium's reversed lam-alef ligature and
4785 // is swapped back to logical `lam + alef-variant` (`ب أ ل` → `ب ل أ`).
4786 assert_eq!(
4787 clean_text("\u{0628}\u{0623}\u{0644}"),
4788 "\u{0628}\u{0644}\u{0623}"
4789 );
4790 // But when the alef-variant is *already* preceded by a lam it is the logical
4791 // ligature `لآ`; the following lam is the next syllable's letter and must not
4792 // move. `التعلم الآلي` must stay `الآلي`, not become `اللآي`.
4793 assert_eq!(
4794 clean_text("\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"),
4795 "\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"
4796 );
4797 }
4798
4799 /// The #419 page, in points: three layout boxes over one paragraph, two of
4800 /// them ending partway through a line. The sliced lines miss the 0.2 claim
4801 /// and become orphans; the third model box starts above the second orphan,
4802 /// so unfitted the reading order emits that box first and strands the line.
4803 fn sliced_paragraph() -> (Vec<Region>, Vec<TextCell>) {
4804 let line = |text: &str, t: f32, r: f32| cell(text, 60.0, t, r, t + 11.0);
4805 let cells = vec![
4806 line("The mission of this series is to improve", 135.0, 458.0),
4807 line("The books in this series are technical,", 147.0, 458.0),
4808 line("substantial. The authors are", 159.0, 458.0),
4809 line("highly experienced craftsmen and", 171.5, 458.0), // sliced: 1.5/11 under box A
4810 line("actually works in practice, as opposed", 185.0, 458.0),
4811 line("about what the author has done, not", 197.0, 458.0),
4812 line("about programming, there will be lots", 210.5, 458.0), // sliced: 1.5/11 under box B
4813 line("will be lots of case studies from real", 223.0, 206.0), // C's line
4814 ];
4815 let regions = vec![
4816 region("text", 0.9, 60.0, 132.0, 458.0, 173.0), // A: three lines + a sliver of the 4th
4817 region("text", 0.9, 60.0, 184.0, 458.0, 212.0), // B: two lines + a sliver of the 7th
4818 region("text", 0.9, 60.0, 216.0, 206.0, 227.0), // C: last line, box opening 5.5pt too early
4819 ];
4820 (regions, cells)
4821 }
4822
4823 fn ordered_texts(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
4824 let mut items: Vec<Region> = regions.to_vec();
4825 let cids = super::cluster_cids(&items, cells);
4826 super::order_regions(&mut items, &cids, 500.0, 700.0, |r| r);
4827 super::region_texts_exclusive(&items, cells)
4828 .into_iter()
4829 .map(|t| t.chars().take(9).collect())
4830 .collect()
4831 }
4832
4833 /// #419: fitted to its cells, a model box that cut a line in half no longer
4834 /// overlaps the orphan that line became, so the orphan orders where it
4835 /// reads; unfitted, the same page strands the line after the paragraph.
4836 #[test]
4837 fn fitting_boxes_to_cells_puts_a_sliced_line_back_in_order() {
4838 let (mut regions, cells) = sliced_paragraph();
4839 super::add_orphan_regions(&mut regions, &cells);
4840 assert_eq!(regions.len(), 5, "two orphan lines");
4841 // The defect, for the record: C (top 216) is not strictly below the
4842 // orphan at 210.5–221.5, so the graph orders C first.
4843 assert_eq!(
4844 ordered_texts(®ions, &cells).last().map(String::as_str),
4845 Some("about pro")
4846 );
4847
4848 super::fit_regions_to_cells(&mut regions, &cells);
4849 assert_eq!(regions.len(), 5);
4850 // A ends on its last claimed line, C starts on its only one.
4851 assert_eq!((regions[0].t, regions[0].b), (135.0, 170.0));
4852 assert_eq!((regions[2].t, regions[2].b), (223.0, 234.0));
4853 assert_eq!(
4854 ordered_texts(®ions, &cells),
4855 [
4856 "The missi",
4857 "highly ex",
4858 "actually ",
4859 "about pro",
4860 "will be l"
4861 ]
4862 );
4863 }
4864
4865 /// An orphan the fitted paragraph box surrounds (a short middle line the
4866 /// narrow model box missed while claiming the lines around it) is folded
4867 /// into the paragraph; an empty regular box goes away, a formula stays, a
4868 /// picture is never refitted, and a page with no cells is left untouched.
4869 #[test]
4870 fn fitting_folds_surrounded_orphans_and_drops_empty_regulars() {
4871 let wide = |text: &str, t: f32| cell(text, 60.0, t, 400.0, t + 11.0);
4872 let cells = vec![
4873 wide("first line of the paragraph", 100.0),
4874 cell("stray", 250.0, 112.0, 400.0, 123.0), // clear of the narrow box
4875 wide("third line of the paragraph", 124.0),
4876 ];
4877 let mut regions = vec![
4878 // Narrow box: claims the wide lines at 0.41, misses the short one.
4879 region("text", 0.9, 60.0, 98.0, 200.0, 136.0),
4880 region("section_header", 0.8, 60.0, 300.0, 200.0, 320.0), // no cells
4881 region("formula", 0.8, 60.0, 340.0, 200.0, 360.0), // no cells, kept
4882 region("picture", 0.8, 0.0, 400.0, 500.0, 600.0),
4883 ];
4884 super::add_orphan_regions(&mut regions, &cells);
4885 assert_eq!(regions.len(), 5, "the short line became an orphan");
4886 super::fit_regions_to_cells(&mut regions, &cells);
4887 let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
4888 assert_eq!(labels, ["text", "formula", "picture"]);
4889 let para = ®ions[0];
4890 assert_eq!(
4891 (para.l, para.t, para.r, para.b),
4892 (60.0, 100.0, 400.0, 135.0)
4893 );
4894 assert_eq!(
4895 super::region_texts_exclusive(®ions, &cells)[0],
4896 "first line of the paragraph stray third line of the paragraph"
4897 );
4898 assert_eq!(
4899 (regions[2].t, regions[2].b),
4900 (400.0, 600.0),
4901 "picture untouched"
4902 );
4903
4904 let mut untouched = vec![region("text", 0.9, 0.0, 0.0, 10.0, 10.0)];
4905 super::fit_regions_to_cells(&mut untouched, &[]);
4906 assert_eq!(untouched.len(), 1, "no cells yet: nothing dropped");
4907 }
4908}