docling_pdf/assemble.rs
1//! Layout-driven assembly: map detected [`Region`]s + text cells to a
2//! [`DoclingDocument`], mirroring docling's page-assembly + reading-order.
3//!
4//! Overlapping detections are resolved greedily by score, each text cell is
5//! assigned to its best-containing region, regions are ordered in reading order
6//! (two-column aware), and each becomes a typed node by its layout label.
7
8use docling_core::{CaptionParent, Node, PictureClass, PictureImage, Table};
9#[cfg(any(feature = "ml", feature = "ocr-prep"))]
10use image::RgbImage;
11
12use crate::layout::Region;
13use crate::pdfium_backend::{PdfPage, TextCell};
14
15fn area(l: f32, t: f32, r: f32, b: f32) -> f32 {
16 ((r - l).max(0.0)) * ((b - t).max(0.0))
17}
18
19/// Intersection area of two boxes.
20fn inter(a: &Region, l: f32, t: f32, r: f32, b: f32) -> f32 {
21 let il = a.l.max(l);
22 let it = a.t.max(t);
23 let ir = a.r.min(r);
24 let ib = a.b.min(b);
25 area(il, it, ir, ib)
26}
27
28/// Wrapper (structured-region) labels, ported from docling
29/// `LayoutPostprocessor.WRAPPER_TYPES`: a region that *contains* other regions
30/// and renders as a structured block (a table / table-of-contents index), not as
31/// its own flat text.
32fn is_wrapper(label: &str) -> bool {
33 matches!(
34 label,
35 "table" | "document_index" | "form" | "key_value_region"
36 )
37}
38
39/// Labels docling's table-structure (TableFormer) model runs on and that render
40/// as a Markdown table: a plain `table` and a `document_index` (a table of
41/// contents), which docling assembles as a `TableItem` too.
42pub fn is_table_like(label: &str) -> bool {
43 matches!(label, "table" | "document_index")
44}
45
46/// Greedily keep regions by descending score, dropping a region that is mostly
47/// covered by an already-kept one (RT-DETR emits overlapping duplicates).
48fn greedy(mut regions: Vec<Region>) -> Vec<Region> {
49 regions.sort_by(|a, b| b.score.total_cmp(&a.score));
50 let mut kept: Vec<Region> = Vec::new();
51 for r in regions {
52 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
53 let covered = kept.iter().any(|k| {
54 let i = inter(&r, k.l, k.t, k.r, k.b);
55 let ka = area(k.l, k.t, k.r, k.b).max(1.0);
56 // drop if most of r is inside k, or they strongly mutually overlap
57 i / ra > 0.7 || i / (ra + ka - i) > 0.5
58 });
59 if !covered {
60 kept.push(r);
61 }
62 }
63 kept
64}
65
66/// Resolve overlapping RT-DETR detections, ported from the bucket structure of
67/// docling's `LayoutPostprocessor`: regular, picture and wrapper clusters live in
68/// **separate** spatial indexes and are de-overlapped independently, so a
69/// high-score picture never suppresses a lower-score table or table-of-contents
70/// index (the redp5110 TOC that was otherwise replaced by a picture box). A
71/// cross-type pass first drops a picture that nearly coincides with a table
72/// (`_handle_cross_type_overlaps`), keeping the structured table.
73/// docling's `_remove_overlapping_clusters("picture")`: same-label picture
74/// detections whose boxes heavily overlap (IoU > 0.8, or either box > 80 %
75/// contained in the other) form one group, and a single survivor is kept per
76/// group. Survivor selection ports `_should_prefer_cluster` /
77/// `_select_best_cluster_from_group` with the picture params
78/// (`area_threshold` 2.0, `conf_threshold` 0.3): a candidate is rejected only
79/// when a rival is both comparable in size (candidate ≤ 2× its area) and
80/// clearly more confident (> 0.3); among the survivors the *larger* box wins
81/// unless it is > 0.3 less confident. Net effect on the corpus: a figure the
82/// detector proposes both whole and as its sub-panels (2206's four-thumbnail
83/// Figure 1) collapses to the whole-figure box, exactly like docling.
84pub(crate) fn dedup_pictures(regions: &mut Vec<Region>) {
85 remove_overlapping_specials(regions, |l| l == "picture", 2.0, 0.3);
86}
87
88/// docling's `_remove_overlapping_clusters` for one special bucket (the
89/// regions whose label satisfies `in_bucket`), with that bucket's
90/// `OVERLAP_PARAMS`: picture (2.0, 0.3) — see [`dedup_pictures`] — or wrapper
91/// (2.0, 0.2) for the table bucket in [`resolve`]. Regions outside the bucket
92/// are untouched; a bucket member overlapping nothing is always kept.
93fn remove_overlapping_specials(
94 regions: &mut Vec<Region>,
95 in_bucket: impl Fn(&str) -> bool,
96 area_threshold: f32,
97 conf_threshold: f32,
98) {
99 let idx: Vec<usize> = (0..regions.len())
100 .filter(|&i| in_bucket(regions[i].label))
101 .collect();
102 if idx.len() < 2 {
103 return;
104 }
105 // Union-find over the bucket.
106 let mut parent: Vec<usize> = (0..idx.len()).collect();
107 fn find(parent: &mut [usize], i: usize) -> usize {
108 let mut root = i;
109 while parent[root] != root {
110 root = parent[root];
111 }
112 let mut cur = i;
113 while parent[cur] != root {
114 let next = parent[cur];
115 parent[cur] = root;
116 cur = next;
117 }
118 root
119 }
120 let boxed = |r: &Region| (r.l, r.t, r.r, r.b);
121 for a in 0..idx.len() {
122 for b in (a + 1)..idx.len() {
123 let (ra, rb) = (®ions[idx[a]], ®ions[idx[b]]);
124 let (al, at, ar, ab_) = boxed(ra);
125 let (bl, bt, br, bb) = boxed(rb);
126 let ix = (ar.min(br) - al.max(bl)).max(0.0);
127 let iy = (ab_.min(bb) - at.max(bt)).max(0.0);
128 let inter = ix * iy;
129 let aa = area(al, at, ar, ab_).max(f32::EPSILON);
130 let ba = area(bl, bt, br, bb).max(f32::EPSILON);
131 let iou = inter / (aa + ba - inter).max(f32::EPSILON);
132 if iou > 0.8 || inter / aa > 0.8 || inter / ba > 0.8 {
133 let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
134 if pa != pb {
135 parent[pa] = pb;
136 }
137 }
138 }
139 }
140 // Per group, run docling's pairwise preference + larger-wins selection.
141 let mut groups: std::collections::HashMap<usize, Vec<usize>> = std::collections::HashMap::new();
142 for i in 0..idx.len() {
143 let root = find(&mut parent, i);
144 groups.entry(root).or_default().push(i);
145 }
146 let mut drop = vec![false; regions.len()];
147 for group in groups.values() {
148 if group.len() < 2 {
149 continue;
150 }
151 let area_of = |i: usize| {
152 let r = ®ions[idx[i]];
153 area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
154 };
155 let mut best: Option<usize> = None;
156 for &cand in group {
157 let passes = group.iter().all(|&other| {
158 if other == cand {
159 return true;
160 }
161 let area_ratio = area_of(cand) / area_of(other);
162 let conf_diff = regions[idx[other]].score - regions[idx[cand]].score;
163 !(area_ratio <= area_threshold && conf_diff > conf_threshold)
164 });
165 if passes {
166 best = Some(match best {
167 None => cand,
168 Some(cur) => {
169 if area_of(cand) > area_of(cur)
170 && regions[idx[cur]].score - regions[idx[cand]].score <= conf_threshold
171 {
172 cand
173 } else {
174 cur
175 }
176 }
177 });
178 }
179 }
180 // Every candidate rejected can't happen with docling's rule (rejection
181 // needs a strictly better rival); guard with highest score anyway.
182 let keep = best.unwrap_or_else(|| {
183 *group
184 .iter()
185 .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(®ions[idx[b]].score))
186 .expect("non-empty group")
187 });
188 for &i in group {
189 if i != keep {
190 drop[idx[i]] = true;
191 }
192 }
193 }
194 let mut keep_iter = drop.into_iter();
195 regions.retain(|_| !keep_iter.next().expect("aligned"));
196}
197
198/// docling's `_remove_overlapping_clusters("regular")` for what [`greedy`]
199/// leaves standing, run on OCR'd pages before the region-scoped OCR (found
200/// with #471's synthetic scan: every line of the page came out twice).
201///
202/// `greedy` keeps regions by descending score and drops a candidate mostly
203/// inside an already-kept one — so a *lower*-score block that contains
204/// several higher-score line boxes (RT-DETR's favourite reading of a sparse
205/// scanned page: every line, plus the paragraph) survives alongside them.
206/// On a digital page that is harmless: [`fit_regions_to_cells`] hands each
207/// text cell to one owner and drops the regions left empty. On a scanned
208/// page the cells don't exist yet — they come from OCR of *each* region's
209/// crop — so the block and its lines were each recognized, and the block,
210/// which then owned both cell sets, read every line twice.
211///
212/// Upstream never has this problem because its OCR runs over the bitmap
213/// before layout postprocessing and each cell is assigned once; the
214/// postprocessor's regular pass then groups clusters that overlap (IoU > 0.8,
215/// or either > 80 % contained in the other) with a union-find, keeps one
216/// survivor per group via `_should_prefer_cluster` /
217/// `_select_best_cluster_from_group` (`area_threshold` 1.3, `conf_threshold`
218/// 0.05; a LIST_ITEM beats a same-sized TEXT, a CODE box beats what it
219/// contains) and merges the losers' cells into it, whose box is then fitted
220/// to those cells. The equivalent for region-scoped OCR: one survivor per
221/// group, keeping its label and score, with the group's **union** box so the
222/// single crop still covers every merged line. Pictures and wrappers are not
223/// regulars and are left alone ([`dedup_pictures`], [`resolve`]).
224pub(crate) fn merge_overlapping_regulars(regions: &mut Vec<Region>) {
225 let idx: Vec<usize> = (0..regions.len())
226 .filter(|&i| regions[i].label != "picture" && !is_wrapper(regions[i].label))
227 .collect();
228 if idx.len() < 2 {
229 return;
230 }
231 let mut parent: Vec<usize> = (0..idx.len()).collect();
232 fn find(parent: &mut [usize], i: usize) -> usize {
233 let mut root = i;
234 while parent[root] != root {
235 root = parent[root];
236 }
237 let mut cur = i;
238 while parent[cur] != root {
239 let next = parent[cur];
240 parent[cur] = root;
241 cur = next;
242 }
243 root
244 }
245 for a in 0..idx.len() {
246 for b in (a + 1)..idx.len() {
247 let (ra, rb) = (®ions[idx[a]], ®ions[idx[b]]);
248 let i = inter(ra, rb.l, rb.t, rb.r, rb.b);
249 let aa = area(ra.l, ra.t, ra.r, ra.b).max(f32::EPSILON);
250 let ba = area(rb.l, rb.t, rb.r, rb.b).max(f32::EPSILON);
251 if i / (aa + ba - i).max(f32::EPSILON) > 0.8 || i / aa > 0.8 || i / ba > 0.8 {
252 let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
253 if pa != pb {
254 parent[pa] = pb;
255 }
256 }
257 }
258 }
259 let mut groups: std::collections::BTreeMap<usize, Vec<usize>> =
260 std::collections::BTreeMap::new();
261 for i in 0..idx.len() {
262 let root = find(&mut parent, i);
263 groups.entry(root).or_default().push(i);
264 }
265 const AREA_THRESHOLD: f32 = 1.3;
266 const CONF_THRESHOLD: f32 = 0.05;
267 let area_of = |i: usize| {
268 let r = ®ions[idx[i]];
269 area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
270 };
271 // `_should_prefer_cluster(candidate, other)` with the regular params.
272 let prefer = |cand: usize, other: usize| -> bool {
273 let (c, o) = (®ions[idx[cand]], ®ions[idx[other]]);
274 let area_ratio = area_of(cand) / area_of(other);
275 if c.label == "list_item" && o.label == "text" && (1.0 - area_ratio).abs() < 0.2 {
276 return true;
277 }
278 if c.label == "code" && inter(o, c.l, c.t, c.r, c.b) / area_of(other) > 0.8 {
279 return true;
280 }
281 !(area_ratio <= AREA_THRESHOLD && o.score - c.score > CONF_THRESHOLD)
282 };
283 let mut drop = vec![false; regions.len()];
284 let mut unions: Vec<(usize, (f32, f32, f32, f32))> = Vec::new();
285 for group in groups.values() {
286 if group.len() < 2 {
287 continue;
288 }
289 let mut best: Option<usize> = None;
290 for &cand in group {
291 if group
292 .iter()
293 .all(|&other| other == cand || prefer(cand, other))
294 {
295 best = Some(match best {
296 None => cand,
297 Some(cur)
298 if area_of(cand) > area_of(cur)
299 && regions[idx[cur]].score - regions[idx[cand]].score
300 <= CONF_THRESHOLD =>
301 {
302 cand
303 }
304 Some(cur) => cur,
305 });
306 }
307 }
308 // docling falls back to the group's first cluster; the highest score
309 // is the deterministic equivalent for a set with no insertion order.
310 let keep = best.unwrap_or_else(|| {
311 *group
312 .iter()
313 .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(®ions[idx[b]].score))
314 .expect("non-empty group")
315 });
316 let mut u = (
317 f32::INFINITY,
318 f32::INFINITY,
319 f32::NEG_INFINITY,
320 f32::NEG_INFINITY,
321 );
322 for &i in group {
323 let r = ®ions[idx[i]];
324 u = (u.0.min(r.l), u.1.min(r.t), u.2.max(r.r), u.3.max(r.b));
325 if i != keep {
326 drop[idx[i]] = true;
327 }
328 }
329 unions.push((idx[keep], u));
330 }
331 for (i, (l, t, r, b)) in unions {
332 let k = &mut regions[i];
333 (k.l, k.t, k.r, k.b) = (l, t, r, b);
334 }
335 let mut keep_iter = drop.into_iter();
336 regions.retain(|_| !keep_iter.next().expect("aligned"));
337}
338
339/// `intersection_over_union` of two regions.
340fn iou(a: &Region, b: &Region) -> f32 {
341 let i = inter(a, b.l, b.t, b.r, b.b);
342 let u = area(a.l, a.t, a.r, a.b) + area(b.l, b.t, b.r, b.b) - i;
343 if u > 0.0 {
344 i / u
345 } else {
346 0.0
347 }
348}
349
350/// docling's `_resolve_coincident_pairs` (#4059, 2.122): for every (loser,
351/// winner) pair at a near-identical box (IoU > 0.8) whose confidences are
352/// within 0.1 (`loser.score - winner.score < 0.1`), the loser label is dropped
353/// so the label with the richer downstream semantic survives. Nothing else —
354/// containment, area — is considered; a clearly more confident loser stays.
355fn coincident_losers(regions: &[Region], losers: &[usize], winners: &[usize]) -> Vec<usize> {
356 let mut out = Vec::new();
357 for &li in losers {
358 for &wi in winners {
359 if iou(®ions[li], ®ions[wi]) > 0.8 && regions[li].score - regions[wi].score < 0.1
360 {
361 out.push(li);
362 break;
363 }
364 }
365 }
366 out
367}
368
369/// docling's `_handle_cross_type_overlaps` (2.122/2.123 shape): the layout
370/// model can emit one grounded region under several labels, and the picture /
371/// table / container buckets are de-overlapped independently, so such a region
372/// survives twice. Elect a winner for the near-identical pairs:
373///
374/// | pair | loser | winner |
375/// |---------------------------------------|-----------|---------------------|
376/// | TABLE vs DOCUMENT_INDEX | table | document_index |
377/// | PICTURE vs TABLE / DOCUMENT_INDEX | picture | the table-like |
378/// | FORM / KEY_VALUE_REGION vs *surviving* TABLE / DOCUMENT_INDEX / PICTURE | container | structured element |
379///
380/// IoU (not containment) so a genuine small figure inside a large table region
381/// is not removed; the confidence tolerance keeps a clearly more confident
382/// loser (an earlier port dropped every coincident picture regardless).
383fn handle_cross_type_overlaps(regions: Vec<Region>) -> Vec<Region> {
384 let by = |pred: &dyn Fn(&str) -> bool| -> Vec<usize> {
385 (0..regions.len())
386 .filter(|&i| pred(regions[i].label))
387 .collect()
388 };
389 let tables = by(&|l| l == "table");
390 let doc_indices = by(&|l| l == "document_index");
391 let pictures = by(&|l| l == "picture");
392 let containers = by(&|l| matches!(l, "form" | "key_value_region"));
393 let mut drop = vec![false; regions.len()];
394 for i in coincident_losers(®ions, &tables, &doc_indices) {
395 drop[i] = true;
396 }
397 let table_like: Vec<usize> = tables.iter().chain(&doc_indices).copied().collect();
398 for i in coincident_losers(®ions, &pictures, &table_like) {
399 drop[i] = true;
400 }
401 let structured: Vec<usize> = table_like
402 .iter()
403 .chain(&pictures)
404 .copied()
405 .filter(|&i| !drop[i])
406 .collect();
407 for i in coincident_losers(®ions, &containers, &structured) {
408 drop[i] = true;
409 }
410 let mut drop = drop.into_iter();
411 let mut regions = regions;
412 regions.retain(|_| !drop.next().expect("aligned"));
413 regions
414}
415
416pub fn resolve(regions: Vec<Region>) -> Vec<Region> {
417 let regions = handle_cross_type_overlaps(regions);
418 // De-overlap each bucket on its own.
419 let pictures = greedy(
420 regions
421 .iter()
422 .filter(|r| r.label == "picture")
423 .cloned()
424 .collect(),
425 );
426 // Tables and containers are separate buckets since docling 2.123
427 // (`TABLE_TYPES` vs `CONTAINER_TYPES`, docling#4064): a form drawn around a
428 // table no longer competes with it for survival — the table nests inside
429 // the container instead (`order_with_containers`).
430 let mut tables = greedy(
431 regions
432 .iter()
433 .filter(|r| is_table_like(r.label))
434 .cloned()
435 .collect(),
436 );
437 // `greedy` only drops a table mostly inside a *more* confident one, so a
438 // low-score whole-page table proposed over the column tables it contains
439 // (a two-column glossary page: 0.53 over 0.71/0.67/0.66) survived next to them and every
440 // cell was emitted twice. docling's `_remove_overlapping_clusters(tables,
441 // "wrapper")` groups tables whose boxes overlap (IoU > 0.8, or either one
442 // > 80 % inside the other) and keeps one per group: run it on what
443 // `greedy` leaves, like `merge_overlapping_regulars` does for regulars.
444 remove_overlapping_specials(&mut tables, |_| true, 2.0, 0.2);
445 let containers = greedy(
446 regions
447 .iter()
448 .filter(|r| matches!(r.label, "form" | "key_value_region"))
449 .cloned()
450 .collect(),
451 );
452 let mut kept = greedy(
453 regions
454 .iter()
455 .filter(|r| r.label != "picture" && !is_wrapper(r.label))
456 .cloned()
457 .collect(),
458 );
459 dedup_nested_code(&mut kept);
460 kept.extend(pictures);
461 kept.extend(tables);
462 kept.extend(containers);
463 kept
464}
465
466/// Drop a regular region that is >80% contained in a surviving special region we
467/// render **as a single unit** — a table/table-of-contents index — ported from
468/// docling's "Remove regular clusters that are included in wrappers" step: the
469/// special absorbs it as a child (a table cell), so it must not also be emitted
470/// as its own paragraph/list-item. This stops the survey list-items from
471/// appearing both inside the detected table and again as bullets
472/// (`table_mislabeled_as_picture`).
473///
474/// `picture` regions are **not** in the swallow set: docling keeps a picture's
475/// contained clusters as the `PictureItem`'s *children* in the document JSON
476/// (`_set_cluster_children`, `ReadingOrderModel._add_child_elements`), while
477/// its `MarkdownPictureSerializer` prints only the caption and the image — the
478/// children never reach the Markdown (verified against the corpus groundtruth:
479/// `amt_handbook`'s in-figure callout labels are absent). So the regulars
480/// inside a picture stay in the region list, claim their cells and are fitted
481/// like any regular, and [`assemble_page`] lifts them out of the page's
482/// reading order into a [`Node::PictureChildren`] after their picture (see
483/// [`picture_parents`]). A line only partially under a figure box (straddling
484/// its border, ≤80 % contained) is no picture's child and is emitted, as
485/// docling emits it (#165).
486///
487/// `form` / `key_value_region` wrappers are deliberately **excluded**: this
488/// pipeline does not render them as a structured block (they are skipped), so
489/// their textual content comes precisely from the contained regular regions —
490/// A `page_footer` that is really the body of the last heading on the page
491/// becomes `text`.
492///
493/// The layout model labels the bottom margin by position as much as by
494/// content: a one-line paragraph that happens to sit where a running footer
495/// would — a CV's `Languages` line under its `## Languages` heading, the last
496/// entry of a section that runs to the page edge — comes back as
497/// `page_footer` (0.88 against 0.49 for `text` on the reporting file, with
498/// the fp32 model and either renderer alike), and Markdown drops furniture,
499/// so the document loses real content while the JSON keeps it under the
500/// wrong label. docling fails the same way. Deliberate deviation, gated so a
501/// real footer keeps its label: the nearest `section_header` above the footer
502/// must overlap it horizontally, end within 2.5 footer heights of its top,
503/// and be the **last body element** of its column — no other non-furniture
504/// region starts at or below the heading's bottom edge over the same span —
505/// and the footer must be a line of text, not a page number: at least 40 %
506/// of the page width. A running footer never follows a heading that has no
507/// body of its own, so the combination is the misread heading body.
508pub fn reclaim_heading_body_footers(regions: &mut [Region], page_w: f32) {
509 let n = regions.len();
510 let overlap_x = |a: &Region, b: &Region| a.r.min(b.r) - a.l.max(b.l) > 0.0;
511 for fi in 0..n {
512 let f = regions[fi].clone();
513 if f.label != "page_footer" || f.r - f.l < 0.4 * page_w {
514 continue;
515 }
516 let fh = (f.b - f.t).max(1.0);
517 // The nearest heading above the footer, over the footer's span.
518 let heading = (0..n)
519 .filter(|&j| {
520 let h = ®ions[j];
521 j != fi && h.label == "section_header" && h.b <= f.t + 0.5 * fh && overlap_x(h, &f)
522 })
523 .min_by(|&a, &b| regions[b].b.total_cmp(®ions[a].b));
524 let Some(hi) = heading else {
525 continue;
526 };
527 let h = regions[hi].clone();
528 if f.t - h.b > 2.5 * fh {
529 continue;
530 }
531 // The heading must have no body of its own: nothing but the footer
532 // starts at or below its bottom edge over the heading's or footer's
533 // span (a heading whose paragraph follows is not this case, and a
534 // heading with the footer far below it was filtered above).
535 let has_body = (0..n).any(|j| {
536 let r = ®ions[j];
537 j != fi
538 && j != hi
539 && !matches!(r.label, "page_footer" | "page_header")
540 && r.t >= h.b - 0.5 * fh
541 && (overlap_x(r, &h) || overlap_x(r, &f))
542 });
543 if has_body {
544 continue;
545 }
546 regions[fi].label = "text";
547 }
548}
549
550/// dropping those would erase the page (e.g. `right_to_left_03`'s form-heavy
551/// pages). Runs *after* [`drop_false_pictures`] so a phantom picture can't
552/// swallow real text on its way out.
553pub fn drop_contained_regulars(regions: &mut Vec<Region>) {
554 let specials: Vec<(f32, f32, f32, f32)> = regions
555 .iter()
556 .filter(|r| is_table_like(r.label))
557 .map(|r| (r.l, r.t, r.r, r.b))
558 .collect();
559 if specials.is_empty() {
560 return;
561 }
562 regions.retain(|r| {
563 if r.label == "picture" || is_wrapper(r.label) {
564 return true;
565 }
566 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
567 !specials
568 .iter()
569 .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.8)
570 });
571}
572
573/// docling's `_set_cluster_children` for pictures: a regular region (one that
574/// claims cells) whose box is > 80 % inside a picture's box
575/// (`intersection_over_self`) is that picture's child — not a page element
576/// (it leaves the reading order and the layout score), but an item under the
577/// `PictureItem` in the JSON. Returns, per region, the index of its parent
578/// picture: the smallest containing one when pictures nest (upstream would
579/// attach it to each).
580pub fn picture_parents(regions: &[Region]) -> Vec<Option<usize>> {
581 regions
582 .iter()
583 .map(|r| {
584 if !claims_cells(r) {
585 return None;
586 }
587 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
588 regions
589 .iter()
590 .enumerate()
591 .filter(|(_, p)| p.label == "picture" && inter(r, p.l, p.t, p.r, p.b) / ra > 0.8)
592 .min_by(|(_, a), (_, b)| {
593 area(a.l, a.t, a.r, a.b).total_cmp(&area(b.l, b.t, b.r, b.b))
594 })
595 .map(|(i, _)| i)
596 })
597 .collect()
598}
599
600/// True for a bare, single-token source-code language label (`XML`, `C#`, `JSON`,
601/// `bash`, …) — the little header the docs render above a code block. Matched
602/// case-insensitively; anything with whitespace or longer than a token is out.
603fn is_code_language(t: &str) -> bool {
604 let t = t.trim();
605 if t.is_empty() || t.chars().any(char::is_whitespace) || t.chars().count() > 12 {
606 return false;
607 }
608 const LANGS: &[&str] = &[
609 "xml",
610 "html",
611 "xhtml",
612 "json",
613 "jsonc",
614 "yaml",
615 "yml",
616 "toml",
617 "ini",
618 "c#",
619 "csharp",
620 "f#",
621 "fsharp",
622 "vb",
623 "c",
624 "c++",
625 "cpp",
626 "java",
627 "kotlin",
628 "scala",
629 "go",
630 "golang",
631 "rust",
632 "swift",
633 "javascript",
634 "js",
635 "typescript",
636 "ts",
637 "jsx",
638 "tsx",
639 "python",
640 "py",
641 "ruby",
642 "rb",
643 "php",
644 "perl",
645 "lua",
646 "r",
647 "dart",
648 "bash",
649 "sh",
650 "shell",
651 "powershell",
652 "zsh",
653 "batch",
654 "cmd",
655 "sql",
656 "tsql",
657 "plsql",
658 "graphql",
659 "dockerfile",
660 "makefile",
661 "css",
662 "scss",
663 "sass",
664 "less",
665 "markdown",
666 "md",
667 "tex",
668 "latex",
669 "diff",
670 "proto",
671 "razor",
672 "cshtml",
673 "xaml",
674 "aspx",
675 "http",
676 ];
677 let lower = t.to_ascii_lowercase();
678 LANGS.contains(&lower.as_str())
679}
680
681/// Mark the region indices that are a code block's **language label** — a bare
682/// `XML`/`C#`/… token sitting directly above a `code` region — so they are consumed
683/// rather than emitted as their own stray paragraph/heading. The label may also be
684/// captured inside a wider code box (rendered as the fence's first line); dropping
685/// the standalone copy just removes the duplicate.
686fn code_language_labels(regions: &[Region], cells: &[TextCell]) -> Vec<bool> {
687 let mut drop = vec![false; regions.len()];
688 for (i, r) in regions.iter().enumerate() {
689 if matches!(r.label, "code" | "picture" | "table") {
690 continue;
691 }
692 if !is_code_language(®ion_text(r, cells)) {
693 continue;
694 }
695 // The label sits just above the code (a blank line's gap) or is swallowed
696 // into the top of a wider code box; either way it is that block's label.
697 // The window is generous because the label's own font is small, so a
698 // one-line gap is several times its height.
699 let line_h = (r.b - r.t).abs().max(1.0);
700 let window = (line_h * 4.0).max(28.0);
701 let labels_code = regions.iter().enumerate().any(|(j, c)| {
702 if j == i || c.label != "code" {
703 return false;
704 }
705 let gap = c.t - r.b; // >0 when the code is below the label
706 let h_overlap = (r.r.min(c.r) - r.l.max(c.l)).max(0.0);
707 gap > -line_h * 3.0 && gap < window && h_overlap > 0.0
708 });
709 if labels_code {
710 drop[i] = true;
711 }
712 }
713 drop
714}
715
716/// Collapse `code` regions where one is nested inside another, keeping the larger.
717///
718/// RT-DETR sometimes emits a tight code box *and* a wider near-duplicate that also
719/// captures the block's language label (`XML`, `C#`, …). When the tight box scores
720/// higher it is kept first, and the wider container — not "mostly inside" the tight
721/// box — survives [`resolve`]'s greedy pass, so the block is emitted twice. Keeping
722/// the **larger** box (rather than dropping it) collapses the pair without leaking
723/// the container's extra cells back out as orphan text, since the larger box still
724/// covers every cell. Restricted to `code` so genuinely distinct nested regions of
725/// other kinds are untouched.
726fn dedup_nested_code(kept: &mut Vec<Region>) {
727 let mut drop = vec![false; kept.len()];
728 for i in 0..kept.len() {
729 if kept[i].label != "code" {
730 continue;
731 }
732 let ai = area(kept[i].l, kept[i].t, kept[i].r, kept[i].b).max(1.0);
733 for j in 0..kept.len() {
734 if i == j || drop[j] || kept[j].label != "code" {
735 continue;
736 }
737 let aj = area(kept[j].l, kept[j].t, kept[j].r, kept[j].b).max(1.0);
738 // Drop i when it is mostly inside a strictly larger code box j.
739 let overlap = inter(&kept[i], kept[j].l, kept[j].t, kept[j].r, kept[j].b);
740 if aj > ai && overlap / ai > 0.7 {
741 drop[i] = true;
742 break;
743 }
744 }
745 }
746 let mut keep = drop.iter();
747 kept.retain(|_| !*keep.next().unwrap());
748}
749
750/// Fraction of the page's non-empty text cells that some detected region
751/// claims (>0.2 intersection-over-self, docling's assignment rule). 1.0 for a
752/// page without text cells.
753///
754/// The int8-layout guard keys off this: a dense digital page whose detections
755/// cover almost none of its text is the signature of quantized confidences
756/// flipping under the 0.5 label thresholds on this CPU's kernels — not of a
757/// genuinely empty layout — and is worth re-running on the fp32 graph.
758pub fn layout_cell_coverage(regions: &[Region], cells: &[TextCell]) -> f32 {
759 let mut total = 0usize;
760 let mut covered = 0usize;
761 for c in cells {
762 if c.text.trim().is_empty() {
763 continue;
764 }
765 total += 1;
766 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
767 if regions
768 .iter()
769 .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
770 {
771 covered += 1;
772 }
773 }
774 if total == 0 {
775 1.0
776 } else {
777 covered as f32 / total as f32
778 }
779}
780
781/// Append `text` regions for cells the layout left uncovered ("orphan cells"),
782/// the way docling's `LayoutPostprocessor` does (`create_orphan_clusters`): any
783/// non-empty cell that no kept region covers (>50% of the cell's area) becomes a
784/// text region of its own, so text the detector missed (a stray `.`, a small
785/// label) is still emitted instead of silently dropped. Adjacent orphan cells on a
786/// line are merged so a missed paragraph doesn't shatter into one block per line.
787pub fn add_orphan_regions(regions: &mut Vec<Region>, cells: &[TextCell]) {
788 // docling assigns each cell to its single best-overlapping cluster at
789 // intersection-over-self > 0.2 and serializes exactly the assigned cells —
790 // and since [`region_texts_exclusive`] now emits under that very rule, the
791 // claim test here matches it: any cell over 0.2 will actually render in
792 // its best region, everything else becomes an orphan. Completeness by
793 // construction, with no (0.2, 0.5] hole (the old > 0.5 serializer needed
794 // the claim test raised to > 0.5 to keep right_to_left_03's `20300` from
795 // vanishing; the exclusive port closes that structurally).
796 //
797 // Only *regular* clusters claim cells: docling's `_find_unassigned_cells`
798 // walks `regular_clusters` alone, so a cell under a `picture` or a wrapper
799 // (`table`/`document_index`/`form`/`key_value_region`) that no regular
800 // cluster covers still becomes an orphan text cluster (#165). The orphans
801 // that end up *fully* inside the special are re-dropped by
802 // [`drop_contained_regulars`] (docling's Markdown drops them the same way
803 // — a picture's children never reach its `MarkdownPictureSerializer`
804 // output, a table's text renders through the reconstructed grid). The
805 // observable fix is the border-straddlers: a line only partially under a
806 // figure box used to lose its cells to the picture's 0.2 claim and vanish
807 // — now it forms an orphan region and is emitted, as docling does.
808 let assigned = |c: &TextCell| {
809 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
810 regions
811 .iter()
812 .filter(|r| r.label != "picture" && !is_wrapper(r.label))
813 .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
814 };
815 // Collect orphan cells (non-empty, unassigned), in page order.
816 let mut orphans: Vec<&TextCell> = cells
817 .iter()
818 .filter(|c| !c.text.trim().is_empty() && !assigned(c))
819 .collect();
820 if orphans.is_empty() {
821 return;
822 }
823 orphans.sort_by(|a, b| a.t.total_cmp(&b.t).then(a.l.total_cmp(&b.l)));
824 // Merge cells that sit on the same line and nearly touch into one region, so a
825 // dropped multi-word line stays one block (docling's refinement merges these).
826 let mut merged: Vec<Region> = Vec::new();
827 for c in orphans {
828 let h = (c.b - c.t).abs().max(1.0);
829 if let Some(last) = merged.last_mut() {
830 let same_line = (last.t - c.t).abs() < h * 0.5;
831 let touching = c.l <= last.r + h && c.l >= last.l - h;
832 // Both tolerances scale with the cell's own height, so a run set
833 // vertically (the arXiv stamp up the margin, #528 — a cell a few
834 // points wide and hundreds tall) would read as "on the line" of
835 // whatever precedes it and glue a whole column into one region.
836 // Lines of one row differ by a drop cap's few multiples at most.
837 let lh = (last.b - last.t).abs().max(1.0);
838 let comparable = h <= 4.0 * lh && lh <= 4.0 * h;
839 if same_line && touching && comparable {
840 last.l = last.l.min(c.l);
841 last.r = last.r.max(c.r);
842 last.t = last.t.min(c.t);
843 last.b = last.b.max(c.b);
844 continue;
845 }
846 }
847 merged.push(Region {
848 label: "text",
849 score: 0.0,
850 l: c.l,
851 t: c.t,
852 r: c.r,
853 b: c.b,
854 });
855 }
856 regions.extend(merged);
857}
858
859/// Demote a `picture` region that is really a **text panel** — a paragraph block
860/// the layout model boxed as a figure because it is typeset on a colored
861/// background (terms-and-conditions callouts, quote boxes) — into ordinary
862/// `text` regions, one per paragraph, so its words are read instead of shipped
863/// as pixels. docling loses this text the same way (cells assigned to a picture
864/// cluster are never serialized); this is a deliberate improvement, not parity.
865///
866/// The gate is conservative so a genuine figure keeps its crop: the region must
867/// contain at least three text lines whose median width spans most of the panel
868/// (axis labels and chat bubbles are narrow and varied) and whose cells cover a
869/// substantial fraction of its area (a photo or chart with sparse labels does
870/// not). Paragraph boundaries are re-derived from the line pitch: a vertical gap
871/// clearly larger than the panel's own leading starts a new `text` region, so
872/// the panel doesn't collapse into one giant paragraph.
873///
874/// Works on any cell source — the digital text layer or OCR lines recognized
875/// from the picture crop — so the native and browser paths, with or without
876/// force-OCR, demote identically.
877pub fn recover_text_panels(regions: &mut Vec<Region>, cells: &[TextCell]) {
878 // A *captioned* picture is a genuine figure whatever it contains — the
879 // corpus is full of document screenshots ("Figure 3: …" above a page
880 // image) that are exactly as dense and wide as a text panel. Only an
881 // uncaptioned picture is a demotion candidate.
882 let captioned: Vec<bool> = regions
883 .iter()
884 .map(|r| {
885 r.label == "picture"
886 && regions.iter().any(|c| {
887 c.label == "caption" && c.r.min(r.r) - c.l.max(r.l) > 0.0 && {
888 let gap = if c.t >= r.b {
889 c.t - r.b
890 } else if r.t >= c.b {
891 r.t - c.b
892 } else {
893 f32::MAX // vertically overlapping: not a caption
894 };
895 gap <= 25.0
896 }
897 })
898 })
899 .collect();
900 let mut out: Vec<Region> = Vec::with_capacity(regions.len());
901 // Synthesized paragraphs and the demoted panels' boxes are kept separate
902 // from `out` until the end: the dedup filter below must not confuse a
903 // paragraph we just built with a pre-existing region inside the panel.
904 let mut demoted_paras: Vec<Region> = Vec::new();
905 let mut demoted_boxes: Vec<(f32, f32, f32, f32)> = Vec::new();
906 for (i, r) in regions.drain(..).enumerate() {
907 if r.label != "picture" || captioned[i] {
908 out.push(r);
909 continue;
910 }
911 let inside: Vec<&TextCell> = cells
912 .iter()
913 .filter(|c| {
914 !c.text.trim().is_empty() && {
915 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
916 inter(&r, c.l, c.t, c.r, c.b) / ca > 0.5
917 }
918 })
919 .collect();
920 // Group the contained cells into lines by vertical overlap (the same
921 // rule region_text orders by), tracking each line's union box.
922 let mut lines: Vec<(f32, f32, f32, f32)> = Vec::new(); // (t, b, l, r)
923 for c in &inside {
924 let (ct, cb) = (c.t.min(c.b), c.t.max(c.b));
925 match lines.iter_mut().find(|(lt, lb, _, _)| {
926 let ov = cb.min(*lb) - ct.max(*lt);
927 ov > 0.5 * (cb - ct).min(*lb - *lt).max(1.0)
928 }) {
929 Some((lt, lb, ll, lr)) => {
930 *lt = lt.min(ct);
931 *lb = lb.max(cb);
932 *ll = ll.min(c.l);
933 *lr = lr.max(c.r);
934 }
935 None => lines.push((ct, cb, c.l, c.r)),
936 }
937 }
938 if lines.len() < 3 {
939 out.push(r);
940 continue;
941 }
942 let panel_w = (r.r - r.l).max(1.0);
943 let coverage = inside.iter().map(|c| area(c.l, c.t, c.r, c.b)).sum::<f32>()
944 / area(r.l, r.t, r.r, r.b).max(1.0);
945 let mut widths: Vec<f32> = lines.iter().map(|(_, _, l, rr)| rr - l).collect();
946 widths.sort_by(f32::total_cmp);
947 // A figure's text is ragged: a title line, small axis/tick labels, and
948 // OCR boxes over the plot area come out at wildly different heights,
949 // whereas a real text panel is set in one face with constant leading.
950 // Require near-uniform line heights (median absolute deviation ≤ 35%
951 // of the median) so an uncaptioned chart keeps its crop even when its
952 // labels are dense enough to pass the coverage gate (#173) — garbled
953 // OCR of its bars is not content.
954 let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
955 heights.sort_by(f32::total_cmp);
956 let h_med = heights[heights.len() / 2].max(1.0);
957 let mut devs: Vec<f32> = heights.iter().map(|h| (h - h_med).abs()).collect();
958 devs.sort_by(f32::total_cmp);
959 let uniform = devs[devs.len() / 2] <= 0.35 * h_med;
960 let text_panel = coverage >= 0.2 && widths[widths.len() / 2] >= 0.45 * panel_w && uniform;
961 if !text_panel {
962 out.push(r);
963 continue;
964 }
965 lines.sort_by(|a, b| a.0.total_cmp(&b.0));
966 let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
967 heights.sort_by(f32::total_cmp);
968 let h = heights[heights.len() / 2].max(1.0);
969 let mut gaps: Vec<f32> = lines
970 .windows(2)
971 .map(|w| (w[1].0 - w[0].1).max(0.0))
972 .collect();
973 gaps.sort_by(f32::total_cmp);
974 let leading = if gaps.is_empty() {
975 0.0
976 } else {
977 gaps[gaps.len() / 2]
978 };
979 let brk = (1.8 * leading).max(0.75 * h);
980 let mut para: Option<(f32, f32, f32, f32)> = None; // (l, t, r, b) union
981 for (t, b, l, rr) in &lines {
982 match &mut para {
983 Some((pl, _, pr, pb)) if *t - *pb <= brk => {
984 *pl = pl.min(*l);
985 *pr = pr.max(*rr);
986 *pb = pb.max(*b);
987 }
988 _ => {
989 if let Some((pl, pt, pr, pb)) = para.take() {
990 demoted_paras.push(Region {
991 label: "text",
992 score: r.score,
993 l: pl,
994 t: pt,
995 r: pr,
996 b: pb,
997 });
998 }
999 para = Some((*l, *t, *rr, *b));
1000 }
1001 }
1002 }
1003 if let Some((pl, pt, pr, pb)) = para {
1004 demoted_paras.push(Region {
1005 label: "text",
1006 score: r.score,
1007 l: pl,
1008 t: pt,
1009 r: pr,
1010 b: pb,
1011 });
1012 }
1013 demoted_boxes.push((r.l, r.t, r.r, r.b));
1014 }
1015 // The paragraphs are rebuilt from *all* of the panel's cells, so any
1016 // surviving text region inside a demoted panel (an orphan cluster or a
1017 // layout-detected fragment — pictures no longer swallow them, #165) would
1018 // say the same words twice. Consume those; wrappers and pictures stay.
1019 if !demoted_boxes.is_empty() {
1020 out.retain(|r| {
1021 r.label == "picture" || is_wrapper(r.label) || {
1022 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
1023 !demoted_boxes
1024 .iter()
1025 .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.5)
1026 }
1027 });
1028 }
1029 // docling's "Remove regular clusters that are included in wrappers" (a
1030 // regular > 80 % inside a table is absorbed by it) already ran as
1031 // [`drop_contained_regulars`], but before this demotion created new
1032 // regulars. Apply it to them too: a panel that coincides with a table (a
1033 // dense data table detected as picture 0.80 and table 0.62 on one box;
1034 // `_handle_cross_type_overlaps` keeps both once the picture is ≥ 0.1 more
1035 // confident) rebuilds the table's words as a paragraph the grid already
1036 // renders. A panel inside another picture is left as it was.
1037 demoted_paras.retain(|p| {
1038 let pa = area(p.l, p.t, p.r, p.b).max(1.0);
1039 !out.iter()
1040 .any(|s| is_table_like(s.label) && inter(p, s.l, s.t, s.r, s.b) / pa > 0.8)
1041 });
1042 out.extend(demoted_paras);
1043 *regions = out;
1044}
1045
1046/// Drop a `picture` detection covering more than 90 % of the page — docling's
1047/// `LayoutPostprocessor._process_special_clusters` "Filter out full-page
1048/// pictures" (upstream since 2.15), applied to the thresholded detections
1049/// before overlap resolution. A box that big is the page itself, not a figure
1050/// on it: the layout model emits one for a whole-page diagram (a LaTeX figure
1051/// PDF cropped to its drawing), a plate, or a scan, and keeping it would swallow
1052/// every text cell on the page as picture children — the diagram's labels and
1053/// caption vanish behind a lone `<!-- image -->`, where docling reads them out as
1054/// text. `page_w`/`page_h` is the display-frame page box.
1055pub fn drop_full_page_pictures(regions: &mut Vec<Region>, page_w: f32, page_h: f32) {
1056 let page_area = (page_w * page_h).max(1.0);
1057 regions.retain(|r| r.label != "picture" || area(r.l, r.t, r.r, r.b) / page_area <= 0.90);
1058}
1059
1060/// Drop a `picture` detection that is a small, empty, low-confidence margin box on
1061/// a **text page** — a false positive the RT-DETR layout sometimes emits (e.g.
1062/// `right_to_left_02`'s phantom right-column picture, score 0.40); docling does not
1063/// emit it. The gate is deliberately narrow so a genuine figure is never dropped:
1064/// (1) only on pages with a digital text layer — image/scanned/figure pages have
1065/// no `cells` yet at this point (OCR runs later), so their pictures, which *are*
1066/// the content, are kept; (2) only a box covering < 25 % of the page (a margin
1067/// artifact, not a dominant figure); (3) only when it contains no text and scores
1068/// below 0.5 (real empty figures in the corpus all score ≥ 0.86).
1069pub fn drop_false_pictures(
1070 regions: &mut Vec<Region>,
1071 cells: &[TextCell],
1072 page_w: f32,
1073 page_h: f32,
1074) {
1075 if cells.iter().all(|c| c.text.trim().is_empty()) {
1076 return; // no digital text layer (image/scanned page) — keep all pictures
1077 }
1078 // A text-document page carries several text-bearing non-picture regions (so a
1079 // spurious margin picture is clearly extra). A slide / figure page has at most
1080 // one — there the picture is the content, so never drop it.
1081 let content_regions = regions
1082 .iter()
1083 .filter(|r| r.label != "picture" && !region_text(r, cells).trim().is_empty())
1084 .count();
1085 if content_regions < 2 {
1086 return;
1087 }
1088 let page_area = (page_w * page_h).max(1.0);
1089 regions.retain(|r| {
1090 if r.label != "picture" || r.score >= 0.5 {
1091 return true;
1092 }
1093 if area(r.l, r.t, r.r, r.b) / page_area >= 0.25 {
1094 return true; // a dominant figure, not a margin artifact
1095 }
1096 // Keep it if any text cell falls mostly inside (a real captioned/labelled
1097 // figure); drop only the genuinely empty low-confidence boxes.
1098 cells.iter().any(|c| {
1099 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1100 !c.text.trim().is_empty() && inter(r, c.l, c.t, c.r, c.b) / ca > 0.5
1101 })
1102 });
1103}
1104
1105/// A small digit-only region in the top/bottom margin: a page number. docling
1106/// emits `right_to_left_02`'s bottom `11` as the page's *first* text item (its
1107/// reading-order model floats the page number to the front), whereas our
1108/// position-based ordering would place a bottom region last.
1109fn is_page_number(region: &Region, cells: &[TextCell], page_h: f32) -> bool {
1110 let t = region_text(region, cells);
1111 let t = t.trim();
1112 !t.is_empty()
1113 && t.chars().all(|c| c.is_ascii_digit())
1114 && (region.b - region.t).abs() < 30.0
1115 && (region.t < page_h * 0.12 || region.b > page_h * 0.88)
1116}
1117
1118/// docling's `form` / `key_value_region` *containers* (2.123, docling#4064):
1119/// every region sitting > 0.8 inside one — text, list items, and since #4064
1120/// tables and pictures too — is that container's child. Children are
1121/// reading-ordered among themselves and emitted as one block where the
1122/// container falls in the page's top-level order (a `form_area` /
1123/// `key_value_area` group upstream), instead of interleaving with the text
1124/// around the form. A child inside several containers belongs to the smallest
1125/// (then most confident, then first); a container with children shrinks to
1126/// their union for the top-level ordering, like upstream's bbox adjustment.
1127///
1128/// The containers themselves are still not emitted (`is_skipped`), so the
1129/// Markdown is exactly upstream's — a group prints only its children.
1130///
1131/// `cids` are the items' positions in docling's assembly order
1132/// ([`cluster_cids`]) — the reading-order predictor's same-row rule (#424)
1133/// pairs consecutive ones, within the top level and within each container.
1134fn order_with_containers<T: Clone>(
1135 items: &mut Vec<T>,
1136 cids: &[usize],
1137 page_w: f32,
1138 page_h: f32,
1139 reg: impl Fn(&T) -> &Region,
1140) {
1141 let is_container = |r: &Region| matches!(r.label, "form" | "key_value_region");
1142 let containers: Vec<usize> = (0..items.len())
1143 .filter(|&i| is_container(reg(&items[i])))
1144 .collect();
1145 if containers.is_empty() {
1146 order_regions(items, cids, page_w, page_h, reg);
1147 return;
1148 }
1149 // Parent container per item (containers never nest in each other here —
1150 // upstream assigns regulars and tables/pictures only).
1151 let mut parent: Vec<Option<usize>> = vec![None; items.len()];
1152 for i in 0..items.len() {
1153 let r = reg(&items[i]);
1154 if is_container(r) {
1155 continue;
1156 }
1157 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
1158 let mut best: Option<(usize, f32, f32)> = None; // (idx, area, -score)
1159 for &c in &containers {
1160 let cr = reg(&items[c]);
1161 if inter(r, cr.l, cr.t, cr.r, cr.b) / ra > 0.8 {
1162 let key = (area(cr.l, cr.t, cr.r, cr.b), -cr.score);
1163 if best.is_none_or(|(_, a, s)| key.0 < a || (key.0 == a && key.1 < s)) {
1164 best = Some((c, key.0, key.1));
1165 }
1166 }
1167 }
1168 parent[i] = best.map(|(c, _, _)| c);
1169 }
1170 // Top-level pass: non-children plus the containers, the latter shrunk to
1171 // their children's union.
1172 let mut top: Vec<(usize, Region)> = Vec::new();
1173 for i in 0..items.len() {
1174 if parent[i].is_some() {
1175 continue;
1176 }
1177 let mut r = reg(&items[i]).clone();
1178 if is_container(&r) {
1179 let kids: Vec<&Region> = (0..items.len())
1180 .filter(|&k| parent[k] == Some(i))
1181 .map(|k| reg(&items[k]))
1182 .collect();
1183 if !kids.is_empty() {
1184 r.l = kids.iter().map(|k| k.l).fold(f32::INFINITY, f32::min);
1185 r.t = kids.iter().map(|k| k.t).fold(f32::INFINITY, f32::min);
1186 r.r = kids.iter().map(|k| k.r).fold(f32::NEG_INFINITY, f32::max);
1187 r.b = kids.iter().map(|k| k.b).fold(f32::NEG_INFINITY, f32::max);
1188 }
1189 }
1190 top.push((i, r));
1191 }
1192 let top_cids: Vec<usize> = top.iter().map(|(i, _)| cids[*i]).collect();
1193 order_regions(&mut top, &top_cids, page_w, page_h, |it| &it.1);
1194 let mut out: Vec<T> = Vec::with_capacity(items.len());
1195 for (i, _) in top {
1196 if is_container(reg(&items[i])) {
1197 let kid_idx: Vec<usize> = (0..items.len()).filter(|&k| parent[k] == Some(i)).collect();
1198 let mut kids: Vec<T> = kid_idx.iter().map(|&k| items[k].clone()).collect();
1199 let kid_cids: Vec<usize> = kid_idx.iter().map(|&k| cids[k]).collect();
1200 order_regions(&mut kids, &kid_cids, page_w, page_h, ®);
1201 out.push(items[i].clone());
1202 out.extend(kids);
1203 } else {
1204 out.push(items[i].clone());
1205 }
1206 }
1207 *items = out;
1208}
1209
1210/// Furniture / not-yet-emitted labels.
1211fn is_skipped(label: &str) -> bool {
1212 matches!(
1213 label,
1214 "page_header" | "page_footer" | "form" | "key_value_region"
1215 )
1216}
1217
1218/// Reading-order sort of a page's regions, via the ported rule-based
1219/// [`reading_order`](crate::reading_order) predictor (docling's
1220/// `ReadingOrderPredictor`): an up/down geometry graph with same-row links
1221/// between `cids`-consecutive elements (#424), horizontal dilation and a
1222/// depth-first traversal, with `page_header`/`page_footer` ordered as their own
1223/// groups (first/last) as docling does.
1224fn order_regions<T: Clone>(
1225 items: &mut Vec<T>,
1226 cids: &[usize],
1227 page_w: f32,
1228 page_h: f32,
1229 reg: impl Fn(&T) -> &Region,
1230) {
1231 let boxes: Vec<(f32, f32, f32, f32)> = items
1232 .iter()
1233 .map(|it| {
1234 let r = reg(it);
1235 (r.l, r.t, r.r, r.b)
1236 })
1237 .collect();
1238 let is_header: Vec<bool> = items
1239 .iter()
1240 .map(|it| reg(it).label == "page_header")
1241 .collect();
1242 let is_footer: Vec<bool> = items
1243 .iter()
1244 .map(|it| reg(it).label == "page_footer")
1245 .collect();
1246 let order =
1247 crate::reading_order::order_page(&boxes, cids, &is_header, &is_footer, page_w, page_h);
1248 *items = order.iter().map(|&i| items[i].clone()).collect();
1249}
1250
1251/// docling's assembly order of a page's clusters (`LayoutPostprocessor`'s
1252/// final `_sort_clusters(mode="id")`, #424): each region's rank when sorted by
1253/// its first source cell, then by top edge, then left edge; a region with no
1254/// cells sorts after every one that has some. docling numbers its page
1255/// elements (`cid`) in this order, and the reading-order predictor's same-row
1256/// rule pairs elements with consecutive numbers, so the ranks are what
1257/// [`order_with_containers`] hands the predictor.
1258///
1259/// A regular region's first cell is the smallest index among the cells it
1260/// claims. A table, picture or container has no cells of its own upstream
1261/// either — its cells are its *children's*: the regular clusters > 0.8 inside
1262/// it, and upstream every cell no regular cluster claimed is an orphan cluster
1263/// of its own, so a table's interior text (which no regular cluster claims)
1264/// reaches the table through those orphans. Here that is the cells > 0.8
1265/// inside the region plus the claimed cells of the regular regions > 0.8
1266/// inside it. Without the interior cells every table would sort last, and two
1267/// side-by-side tables would then be consecutive and row-linked — reading the
1268/// right table's caption ahead of the left column's headings (2206 page 8).
1269pub fn cluster_cids(regions: &[Region], cells: &[TextCell]) -> Vec<usize> {
1270 let owned = assign_cells(regions, cells);
1271 let first_cell: Vec<usize> = regions
1272 .iter()
1273 .enumerate()
1274 .map(|(i, r)| {
1275 if claims_cells(r) {
1276 return owned[i].iter().copied().min().unwrap_or(usize::MAX);
1277 }
1278 let interior = cells
1279 .iter()
1280 .enumerate()
1281 .filter(|(_, c)| {
1282 !c.text.trim().is_empty()
1283 && inter(r, c.l, c.t, c.r, c.b) / area(c.l, c.t, c.r, c.b).max(1.0) > 0.8
1284 })
1285 .map(|(ci, _)| ci)
1286 .min();
1287 let children = regions
1288 .iter()
1289 .enumerate()
1290 .filter(|(j, child)| {
1291 *j != i && claims_cells(child) && {
1292 let ca = area(child.l, child.t, child.r, child.b).max(1.0);
1293 inter(r, child.l, child.t, child.r, child.b) / ca > 0.8
1294 }
1295 })
1296 .filter_map(|(j, _)| owned[j].iter().copied().min())
1297 .min();
1298 interior
1299 .into_iter()
1300 .chain(children)
1301 .min()
1302 .unwrap_or(usize::MAX)
1303 })
1304 .collect();
1305 let mut by_source: Vec<usize> = (0..regions.len()).collect();
1306 // Stable, like Python's `sorted`: full ties keep the layout order.
1307 by_source.sort_by(|&a, &b| {
1308 first_cell[a]
1309 .cmp(&first_cell[b])
1310 .then(regions[a].t.total_cmp(®ions[b].t))
1311 .then(regions[a].l.total_cmp(®ions[b].l))
1312 });
1313 let mut cids = vec![0; regions.len()];
1314 for (rank, &i) in by_source.iter().enumerate() {
1315 cids[i] = rank;
1316 }
1317 cids
1318}
1319
1320/// Clean a region's assembled text: undo soft-hyphen line wraps, map curly
1321/// quotes and the ellipsis to ASCII (matching docling), and collapse runs of
1322/// whitespace. pdfium emits the line-wrap hyphen as U+0002 in this corpus
1323/// (U+00AD elsewhere), so `word\u{2} continuation` is one hyphenated word —
1324/// drop the hyphen + the joining space and merge (`com\u{2} pact` → `compact`,
1325/// `end-to\u{2} end` → `end-toend`), exactly as docling does.
1326///
1327/// Token spacing is otherwise left as the geometric join produced it. We do not
1328/// tighten punctuation spacing: docling preserves the PDF's own spaces (it keeps
1329/// `{ ahn }`, `Name 1 .`, `[ 9 ]`), and a geometric gap heuristic diverges from
1330/// it more than a plain single-space join does.
1331/// An ordered-list enumeration marker at the start of a list item: leading ASCII
1332/// digits followed by `.`, e.g. `1. Undo/Redo` → `(1, "Undo/Redo")`. Returns
1333/// `None` when the text doesn't start with `digits.`.
1334/// docling's `ListItemMarkerProcessor` bullet patterns
1335/// (`docling/models/postprocessing/list_marker_processor.py`), one glyph each.
1336const LIST_BULLET_MARKERS: &str = "\u{2022}\u{2023}\u{25E6}\u{2043}\u{204C}\u{204D}\u{2219}\u{25AA}\u{25AB}\u{25CF}\u{25CB}-*+•·‣⁃►▶▸➤➢✓✔✗✘";
1337
1338/// docling's numbered-marker patterns as byte-length scanners over the start
1339/// of the text, in its first-wins order (the compound ones first, as they are
1340/// the more specific). Each returns the marker's candidate lengths, longest
1341/// (greedy) first — the alternatives Python's regex would backtrack through
1342/// before the `\s(.+)` tail. Hand-rolled rather than `regex` because this file
1343/// is part of the `pdf-text` (wasm) build, where the `regex` crate is not.
1344/// `\d` is Unicode-aware in Python, hence `char::is_numeric`; the letters are
1345/// ASCII classes in both.
1346const LIST_NUMBERED_MARKERS: &[fn(&str) -> Vec<usize>] = &[
1347 // `\d+(?:\.\d+)+\.?` — 1.1 1.2.3 1.1.
1348 |s| {
1349 let mut i = digits(s, 0);
1350 if i == 0 {
1351 return Vec::new();
1352 }
1353 let mut groups = 0;
1354 while s[i..].starts_with('.') && digits(s, i + 1) > i + 1 {
1355 i = digits(s, i + 1);
1356 groups += 1;
1357 }
1358 if groups == 0 {
1359 return Vec::new();
1360 }
1361 if s[i..].starts_with('.') {
1362 vec![i + 1, i]
1363 } else {
1364 vec![i]
1365 }
1366 },
1367 // `\d+\.?[a-zA-Z]\.` — 9a. 3.a.
1368 |s| digits_dot_letter(s, 0, '.').into_iter().collect(),
1369 // `\d+\.?[a-zA-Z]\)` — 9a) 3.a)
1370 |s| digits_dot_letter(s, 0, ')').into_iter().collect(),
1371 // `\(\d+\.?[a-zA-Z]\)` — (9a) (3.a)
1372 |s| {
1373 if !s.starts_with('(') {
1374 return Vec::new();
1375 }
1376 digits_dot_letter(s, 1, ')').into_iter().collect()
1377 },
1378 // `\d+\.` — 1. 2. 3.
1379 |s| digits_then(s, 0, '.').into_iter().collect(),
1380 // `\d+\)` — 1) 2) 3)
1381 |s| digits_then(s, 0, ')').into_iter().collect(),
1382 // `\(\d+\)` — (1) (2) (3)
1383 |s| {
1384 if !s.starts_with('(') {
1385 return Vec::new();
1386 }
1387 digits_then(s, 1, ')').into_iter().collect()
1388 },
1389 // `\[\d+\]` — [1] [2] [3]
1390 |s| {
1391 if !s.starts_with('[') {
1392 return Vec::new();
1393 }
1394 digits_then(s, 1, ']').into_iter().collect()
1395 },
1396 // `[ivxlcdm]+\.` — i. ii. iii.
1397 |s| class_run_then(s, "ivxlcdm", '.').into_iter().collect(),
1398 // `[IVXLCDM]+\.` — I. II. III.
1399 |s| class_run_then(s, "IVXLCDM", '.').into_iter().collect(),
1400 // `[a-z]\.` / `[A-Z]\.` / `[a-z]\)` / `[A-Z]\)`
1401 |s| {
1402 letter_then(s, char::is_ascii_lowercase, '.')
1403 .into_iter()
1404 .collect()
1405 },
1406 |s| {
1407 letter_then(s, char::is_ascii_uppercase, '.')
1408 .into_iter()
1409 .collect()
1410 },
1411 |s| {
1412 letter_then(s, char::is_ascii_lowercase, ')')
1413 .into_iter()
1414 .collect()
1415 },
1416 |s| {
1417 letter_then(s, char::is_ascii_uppercase, ')')
1418 .into_iter()
1419 .collect()
1420 },
1421];
1422
1423/// Byte offset just past the run of `\d` characters starting at `from`
1424/// (`from` itself when there is none).
1425fn digits(s: &str, from: usize) -> usize {
1426 s[from..]
1427 .char_indices()
1428 .find(|(_, c)| !c.is_numeric())
1429 .map_or(s.len(), |(i, _)| from + i)
1430}
1431
1432/// `\d+<close>` from `from`: the length through `close`, if it matches.
1433fn digits_then(s: &str, from: usize, close: char) -> Option<usize> {
1434 let end = digits(s, from);
1435 (end > from && s[end..].starts_with(close)).then(|| end + close.len_utf8())
1436}
1437
1438/// `\d+\.?[a-zA-Z]<close>` from `from`.
1439fn digits_dot_letter(s: &str, from: usize, close: char) -> Option<usize> {
1440 let mut i = digits(s, from);
1441 if i == from {
1442 return None;
1443 }
1444 if s[i..].starts_with('.') {
1445 i += 1;
1446 }
1447 let letter = s[i..].chars().next().filter(char::is_ascii_alphabetic)?;
1448 i += letter.len_utf8();
1449 s[i..].starts_with(close).then(|| i + close.len_utf8())
1450}
1451
1452/// `[<class>]+<close>` at the start.
1453fn class_run_then(s: &str, class: &str, close: char) -> Option<usize> {
1454 let end = s
1455 .char_indices()
1456 .find(|(_, c)| !class.contains(*c))
1457 .map_or(s.len(), |(i, _)| i);
1458 (end > 0 && s[end..].starts_with(close)).then(|| end + close.len_utf8())
1459}
1460
1461/// `[<letter class>]<close>` at the start.
1462fn letter_then(s: &str, class: fn(&char) -> bool, close: char) -> Option<usize> {
1463 let letter = s.chars().next().filter(class)?;
1464 let i = letter.len_utf8();
1465 s[i..].starts_with(close).then(|| i + close.len_utf8())
1466}
1467
1468/// docling's `ListItemMarkerProcessor.process_list_item`: the item's original
1469/// text is matched against `^(marker)\s(.+)` (DOTALL) for the bullet patterns,
1470/// then the numbered ones in order; a hit splits it into
1471/// `(marker, text, enumerated)`. `\s` is one Unicode whitespace character and
1472/// `.+` everything after it, which must be non-empty.
1473fn split_list_marker(text: &str) -> Option<(&str, &str, bool)> {
1474 let tail_after = |len: usize| -> Option<&str> {
1475 let ws = text[len..].chars().next()?;
1476 if !ws.is_whitespace() {
1477 return None;
1478 }
1479 let rest = &text[len + ws.len_utf8()..];
1480 (!rest.is_empty()).then_some(rest)
1481 };
1482 let first = text.chars().next()?;
1483 if LIST_BULLET_MARKERS.contains(first) {
1484 if let Some(rest) = tail_after(first.len_utf8()) {
1485 return Some((&text[..first.len_utf8()], rest, false));
1486 }
1487 }
1488 for matcher in LIST_NUMBERED_MARKERS {
1489 for len in matcher(text) {
1490 if let Some(rest) = tail_after(len) {
1491 return Some((&text[..len], rest, true));
1492 }
1493 }
1494 }
1495 None
1496}
1497
1498/// A PDF `list_item` region as docling emits it: `ListItemMarkerProcessor`
1499/// splits the marker off (see [`split_list_marker`]), and docling-core's
1500/// Markdown list serializer (default `orig_list_item_marker_mode = AUTO`,
1501/// `ensure_valid_list_item_marker`) prints it as `N. text` for an `N.` marker
1502/// (`case_already_valid`: the marker verbatim); as `- text` for a bullet glyph
1503/// (no letter or digit in the marker: only the `-` the serializer adds); and
1504/// as `- a) text` / `- 1.2 text` / `- [3] text` for any other marker
1505/// (`case_auto`: the serializer's `-`, then the original marker) — spelled
1506/// here as a bullet item whose text carries the marker, the way the DOCX and
1507/// DOC backends already spell theirs. An item without a recognizable marker is
1508/// a plain bullet. The symbol-font bullets docling-parse filters out of its
1509/// cells (`•◦▪·*` glued to the text) are stripped before the match, as before.
1510fn list_item_node(text: &str, loc: [u16; 4], first_in_list: bool) -> Node {
1511 // docling's match runs on the text as docling-parse hands it over; the
1512 // glued symbol-font bullets it never sees are stripped only when the raw
1513 // text matches no marker (`•Text` → `Text`, but `• Text` → marker `•`).
1514 let stripped = text
1515 .trim_start_matches(['•', '◦', '▪', '·', '*'])
1516 .trim_start();
1517 let split = split_list_marker(text).or_else(|| split_list_marker(stripped));
1518 let bullet = |text: String, marker: &str| Node::ListItem {
1519 ordered: false,
1520 number: 0,
1521 first_in_list,
1522 text: md_escape(&text),
1523 level: 0,
1524 // docling keeps the marker as the DocLang list marker
1525 // (`<ldiv><marker>·</marker></ldiv>`); Markdown prints its own `-`.
1526 marker: Some(marker.to_string()),
1527 location: Some(loc),
1528 dclx: None,
1529 href: None,
1530 layer: None,
1531 };
1532 // docling-core's `case_already_valid` is a *full* `\d+\.` match: `3.a.`
1533 // and `1.2` are `case_auto` markers (`- 3.a. text`), not item numbers.
1534 let is_number_dot = |m: &str| {
1535 m.strip_suffix('.')
1536 .is_some_and(|d| !d.is_empty() && d.chars().all(char::is_numeric))
1537 };
1538 match split {
1539 Some((marker, body, true)) if is_number_dot(marker) => {
1540 let number = parse_ordered_marker(marker).map_or(0, |(n, _)| n);
1541 Node::ListItem {
1542 ordered: true,
1543 number,
1544 first_in_list,
1545 text: md_escape(body),
1546 level: 0,
1547 marker: Some(marker.to_string()),
1548 location: Some(loc),
1549 dclx: None,
1550 href: None,
1551 layer: None,
1552 }
1553 }
1554 // `case_auto`: a marker holding a letter or digit rides in the text.
1555 Some((marker, body, true)) => bullet(format!("{marker} {body}"), marker),
1556 Some((marker, body, false)) => bullet(body.to_string(), marker),
1557 None => bullet(stripped.to_string(), "·"),
1558 }
1559}
1560
1561fn parse_ordered_marker(s: &str) -> Option<(u64, String)> {
1562 let digits: String = s.chars().take_while(|c| c.is_ascii_digit()).collect();
1563 if digits.is_empty() {
1564 return None;
1565 }
1566 let rest = s[digits.len()..].strip_prefix('.')?;
1567 let number = digits.parse().ok()?;
1568 Some((number, rest.trim_start().to_string()))
1569}
1570
1571/// Escape markdown special characters the way docling-core's markdown serializer
1572/// does (`markdown.py` post_process): `_` → `\_`, then HTML-escape `&`, `<`, `>`
1573/// (quote=False, so quotes are left). Applied to prose (headings, list items,
1574/// paragraphs); code blocks, the formula placeholder, and table cells are left raw.
1575fn md_escape(text: &str) -> String {
1576 text.replace('_', "\\_")
1577 .replace('&', "&")
1578 .replace('<', "<")
1579 .replace('>', ">")
1580}
1581
1582fn clean_text(text: &str) -> String {
1583 // Typographic-quote normalization follows docling-parse's sanitizer table
1584 // (`pdf_sanitators/constants.h`): every curly quote — single *and double* —
1585 // becomes the ASCII apostrophe `'`, and `‚` a comma. A `"` in docling's
1586 // output only ever comes from a literal `quotedbl` glyph, never from `“ ”`
1587 // (2206's `'text in the wild"` pairs a curly open with a literal-quote
1588 // close). This replaces an earlier Hangul-only special case that patched
1589 // one symptom of mapping `“ ”` to `"`.
1590 let replaced = text
1591 .replace("\u{2} ", "")
1592 .replace("\u{ad} ", "")
1593 .replace(['\u{2}', '\u{ad}'], "") // any stray wrap hyphens not at a join
1594 .replace(
1595 [
1596 '\u{2018}', '\u{2019}', '\u{201b}', '\u{201c}', '\u{201d}', '\u{201e}', '\u{201f}',
1597 ],
1598 "'",
1599 ) // ‘ ’ ‛ “ ” „ ‟ → '
1600 .replace('\u{201a}', ",") // ‚ → ,
1601 .replace(
1602 [
1603 '\u{2010}', '\u{2011}', '\u{2012}', '\u{2013}', '\u{2014}', '\u{2015}', '\u{2212}',
1604 ],
1605 "-",
1606 ) // hyphen/dash family → -
1607 .replace('\u{2044}', "/") // ⁄ fraction slash → /
1608 .replace('\u{2022}', "\u{b7}") // • → · (docling never emits •; inline CCS-concept separators)
1609 .replace('\u{2026}', "..."); // … → ...
1610 // The docling-parse sanitizer already placed the correct spacing (e.g.
1611 // justified double spaces); preserve internal runs of spaces, only
1612 // normalizing line breaks/tabs and trimming the ends.
1613 let out = replaced.replace(['\n', '\r', '\t'], " ").trim().to_string();
1614 fix_arabic_lam_alef(&out)
1615}
1616
1617/// pdfium decomposes the Arabic lam-alef ligature (لا / لإ / لأ / لآ) into its
1618/// glyph constituents in *visual* order — `alef-variant, lam` — but docling keeps
1619/// logical order, `lam, alef-variant`. Swap a mid-word `alef-variant + lam` back
1620/// to `lam + alef-variant`. "Mid-word" (the previous char is an Arabic letter)
1621/// distinguishes the ligature from the definite article `ال` (word-initial
1622/// `alef + lam`), which must stay. No-op for non-Arabic text.
1623fn fix_arabic_lam_alef(s: &str) -> String {
1624 let is_arabic_letter = |c: char| ('\u{0620}'..='\u{064A}').contains(&c);
1625 let chars: Vec<char> = s.chars().collect();
1626 if !chars.iter().any(|&c| is_arabic_letter(c)) {
1627 return s.to_string(); // no-op for non-Arabic text
1628 }
1629 // Pass 1: swap mid-word `alef-variant + lam` → `lam + alef-variant`. Only the
1630 // hamza/madda alef variants (إ أ آ) are safe: the definite article is always
1631 // plain `ا + ل`, so plain `alef + lam` is ambiguous (a legitimate `فعالة` vs a
1632 // reversed `لا` ligature look identical) — leaving plain alef alone avoids
1633 // corrupting legitimate words.
1634 let mut a: Vec<char> = Vec::with_capacity(chars.len());
1635 let mut i = 0;
1636 while i < chars.len() {
1637 let c = chars[i];
1638 if matches!(c, '\u{0622}' | '\u{0623}' | '\u{0625}')
1639 && chars.get(i + 1) == Some(&'\u{0644}')
1640 && i > 0
1641 && is_arabic_letter(chars[i - 1])
1642 // A preceding lam means this alef-variant is *already* the logical
1643 // `lam + alef` ligature; the following lam is the next syllable's
1644 // letter, not a reversed ligature — swapping it corrupts `لآل` → `للآ`
1645 // (e.g. التعلم الآلي → الآلي, not اللآي).
1646 && chars[i - 1] != '\u{0644}'
1647 {
1648 a.push('\u{0644}');
1649 a.push(c);
1650 i += 2;
1651 continue;
1652 }
1653 a.push(c);
1654 i += 1;
1655 }
1656 // Pass 2: insert a space at Arabic↔Latin boundaries (bidi script switch) that
1657 // pdfium runs together — docling separates the embedded Latin run (`وPython`
1658 // → `و Python`).
1659 let mut out: Vec<char> = Vec::with_capacity(a.len());
1660 for (j, &c) in a.iter().enumerate() {
1661 if j > 0 {
1662 let p = a[j - 1];
1663 if (is_arabic_letter(p) && c.is_ascii_alphabetic())
1664 || (p.is_ascii_alphabetic() && is_arabic_letter(c))
1665 {
1666 out.push(' ');
1667 }
1668 }
1669 out.push(c);
1670 }
1671 out.into_iter().collect()
1672}
1673
1674/// docling's `PageAssembleModel._match_hyperlink`: the URI whose link
1675/// annotations cover at least half of the region's box, or `None`. Coverage is
1676/// intersection-over-region-area, **accumulated per URI** — a URL that wraps
1677/// across lines carries several annotation rects that sum toward the same
1678/// target. Ties resolve to the first-seen URI (Python's `max` over dict
1679/// insertion order); the winner still needs `>= 0.5`
1680/// (`_HYPERLINK_COVERAGE_THRESHOLD`).
1681pub(crate) fn region_hyperlink(
1682 region: &Region,
1683 links: &[crate::pdfium_backend::LinkAnnot],
1684) -> Option<String> {
1685 if links.is_empty() {
1686 return None;
1687 }
1688 let area = (region.r - region.l).max(0.0) * (region.b - region.t).max(0.0);
1689 if area <= 0.0 {
1690 return None;
1691 }
1692 let mut coverage: Vec<(&str, f32)> = Vec::new();
1693 for link in links {
1694 let ix = (region.r.min(link.r) - region.l.max(link.l)).max(0.0);
1695 let iy = (region.b.min(link.b) - region.t.max(link.t)).max(0.0);
1696 let c = ix * iy / area;
1697 match coverage.iter_mut().find(|(uri, _)| *uri == link.uri) {
1698 Some((_, acc)) => *acc += c,
1699 None => coverage.push((&link.uri, c)),
1700 }
1701 }
1702 let mut best: Option<(&str, f32)> = None;
1703 for (uri, c) in coverage {
1704 // Strictly greater keeps the first-seen URI on ties, like Python's max.
1705 if best.is_none_or(|(_, bc)| c > bc) {
1706 best = Some((uri, c));
1707 }
1708 }
1709 let (uri, c) = best?;
1710 (c >= 0.5).then(|| normalize_uri(uri))
1711}
1712
1713/// The pydantic-`AnyUrl` normalization docling's hyperlink value passes
1714/// through on its way to the serializer: a URL with an authority but no path
1715/// gains a trailing `/` (`https://arxiv.org` → `https://arxiv.org/`). Other
1716/// AnyUrl canonicalizations (scheme/host lowercasing, percent-encoding) don't
1717/// occur in PDF link annotations in practice, so they are not reproduced.
1718fn normalize_uri(uri: &str) -> String {
1719 if let Some((_, rest)) = uri.split_once("://") {
1720 if !rest.is_empty() && !rest.contains(['/', '?', '#']) {
1721 return format!("{uri}/");
1722 }
1723 }
1724 uri.to_string()
1725}
1726
1727/// Resolve each page hyperlink to the visible text it covers, as `(anchor, uri)`
1728/// in reading order. The anchor is the cells whose centre falls in the link rect,
1729/// joined left-to-right and cleaned the same way prose is (so it matches the
1730/// serialized text), deduped against the immediately-preceding link so pdfium's
1731/// occasional duplicate annotation doesn't double-list. Empty anchors are dropped.
1732pub(crate) fn resolve_link_anchors(page: &PdfPage) -> Vec<(String, String)> {
1733 let mut out: Vec<(String, String)> = Vec::new();
1734 // Use per-word cells, not the line-merged `cells`: a link rect covers a few
1735 // words on a line, and a whole merged line cell would over-capture (its centre
1736 // lands in one link's rect, grabbing the entire line as that link's anchor).
1737 let words = if page.word_cells.is_empty() {
1738 &page.cells
1739 } else {
1740 &page.word_cells
1741 };
1742 for link in &page.links {
1743 // A cell participates when its centre row is inside the rect and it
1744 // overlaps the rect horizontally. A cell can be *wider* than the rect:
1745 // PDFs often draw a whole header line as one text run ("LinkedIn |
1746 // GitHub | Credly"), which docling-parse's word grouping keeps as one
1747 // cell even though each label carries its own link annotation —
1748 // centre-in-rect alone would hand the entire line to every link.
1749 // [`cell_text_in_rect`] clips such a cell to the tokens under the rect.
1750 let mut inside: Vec<(&TextCell, String)> = words
1751 .iter()
1752 .filter(|c| {
1753 let cy = (c.t + c.b) / 2.0;
1754 cy >= link.t && cy <= link.b && c.r.min(link.r) > c.l.max(link.l)
1755 })
1756 .filter_map(|c| {
1757 let text = cell_text_in_rect(c, link.l, link.r);
1758 (!text.is_empty()).then_some((c, text))
1759 })
1760 .collect();
1761 // Reading order: top band then left-to-right (link anchors are LTR).
1762 let band = inside
1763 .iter()
1764 .map(|(c, _)| (c.b - c.t).abs())
1765 .fold(0.0f32, f32::max)
1766 .max(1.0);
1767 inside.sort_by_key(|(c, _)| ((c.t / band).round() as i64, (c.l * 10.0) as i64));
1768 let anchor = clean_text(
1769 &inside
1770 .iter()
1771 .map(|(_, t)| t.trim())
1772 .filter(|t| !t.is_empty())
1773 .collect::<Vec<_>>()
1774 .join(" "),
1775 );
1776 if anchor.is_empty() {
1777 continue;
1778 }
1779 if out
1780 .last()
1781 .is_some_and(|(a, u)| a == &anchor && u == &link.uri)
1782 {
1783 continue;
1784 }
1785 out.push((anchor, link.uri.clone()));
1786 }
1787 out
1788}
1789
1790/// The part of a cell's text that lies under a link rect's x-range. A cell
1791/// fully inside the rect (by centre) returns its whole text. A wider cell is
1792/// split into whitespace tokens whose x-spans are estimated proportionally to
1793/// their character positions (kerning makes this approximate, so selection
1794/// snaps to whole tokens, never characters); tokens whose estimated centre
1795/// falls inside the rect are kept. Returns "" when nothing falls inside.
1796fn cell_text_in_rect(c: &TextCell, l: f32, r: f32) -> String {
1797 let cx = (c.l + c.r) / 2.0;
1798 if cx >= l && cx <= r && c.l >= l - (c.r - c.l) * 0.25 && c.r <= r + (c.r - c.l) * 0.25 {
1799 return c.text.trim().to_string();
1800 }
1801 let chars: Vec<char> = c.text.chars().collect();
1802 let n = chars.len();
1803 if n == 0 || c.r <= c.l {
1804 return String::new();
1805 }
1806 let per = (c.r - c.l) / n as f32;
1807 let mut out: Vec<String> = Vec::new();
1808 let mut token = String::new();
1809 let mut start = 0usize;
1810 // A trailing sentinel space flushes the last token.
1811 for (i, &ch) in chars.iter().enumerate().chain(std::iter::once((n, &' '))) {
1812 if ch.is_whitespace() {
1813 if !token.is_empty() {
1814 let mid = c.l + (start as f32 + (i - start) as f32 / 2.0) * per;
1815 if mid >= l && mid <= r {
1816 out.push(std::mem::take(&mut token));
1817 } else {
1818 token.clear();
1819 }
1820 }
1821 } else {
1822 if token.is_empty() {
1823 start = i;
1824 }
1825 token.push(ch);
1826 }
1827 }
1828 out.join(" ")
1829}
1830
1831/// Cells assigned to a region (best container), in reading order, joined.
1832fn region_text(region: &Region, cells: &[TextCell]) -> String {
1833 let inside: Vec<&TextCell> = cells
1834 .iter()
1835 .filter(|c| {
1836 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1837 inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1838 })
1839 .collect();
1840 cells_text(inside)
1841}
1842
1843/// docling's exclusive cell assignment (`_assign_cells_to_clusters`): every
1844/// non-empty cell goes to the single best-overlapping *regular* region at
1845/// intersection-over-self > 0.2, and each region serializes exactly its
1846/// assigned cells. A cell under two overlapping boxes is emitted once (by the
1847/// better-covering one), and a cell only partially under its region — e.g.
1848/// normal_4pages' big section numeral, ~30 % inside the heading box — still
1849/// joins it (`## 들어가며 1`) instead of leaking as an orphan. Pictures and
1850/// wrappers never claim (docling walks regular clusters only); ties go to the
1851/// first region, like docling's strict `>` best-overlap scan.
1852pub fn region_texts_exclusive(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
1853 let owned = assign_cells(regions, cells);
1854 // Non-claimers (tables/wrappers/pictures) keep the inclusive > 0.5 text:
1855 // docling fills a special cluster's cells from its contained children, and
1856 // downstream table assembly gates on that text being non-empty.
1857 regions
1858 .iter()
1859 .zip(owned)
1860 .map(|(r, cs)| {
1861 if claims_cells(r) {
1862 cells_text(cs.iter().map(|&i| &cells[i]).collect())
1863 } else {
1864 region_text(r, cells)
1865 }
1866 })
1867 .collect()
1868}
1869
1870/// A *regular* region in docling's sense — one that claims cells. Pictures and
1871/// the wrappers (`table`, `document_index`, `form`, `key_value_region`) fill
1872/// their cells from contained children instead.
1873fn claims_cells(r: &Region) -> bool {
1874 r.label != "picture" && !is_wrapper(r.label)
1875}
1876
1877/// docling's `_assign_cells_to_clusters`: each non-empty cell's index goes to
1878/// the single best-overlapping regular region at intersection-over-self > 0.2
1879/// (ties to the first region, like docling's strict `>` scan). One entry per
1880/// region, in region order.
1881fn assign_cells(regions: &[Region], cells: &[TextCell]) -> Vec<Vec<usize>> {
1882 let mut owned: Vec<Vec<usize>> = vec![Vec::new(); regions.len()];
1883 for (ci, c) in cells.iter().enumerate() {
1884 if c.text.trim().is_empty() {
1885 continue;
1886 }
1887 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1888 let mut best: Option<(usize, f32)> = None;
1889 for (i, r) in regions.iter().enumerate() {
1890 if !claims_cells(r) {
1891 continue;
1892 }
1893 let ov = inter(r, c.l, c.t, c.r, c.b) / ca;
1894 if ov > 0.2 && best.is_none_or(|(_, b)| ov > b) {
1895 best = Some((i, ov));
1896 }
1897 }
1898 if let Some((i, _)) = best {
1899 owned[i].push(ci);
1900 }
1901 }
1902 owned
1903}
1904
1905/// docling's regular-cluster refinement after cell assignment
1906/// (`LayoutPostprocessor._process_regular_clusters`, #419), run once the page's
1907/// cells are final and before reading order:
1908///
1909/// 1. every regular region's box becomes the union of the cells it claimed
1910/// (`_adjust_cluster_bboxes` — a regular cluster's bbox *is* its cells'
1911/// bbox; a table's is the union with the model box, and pictures keep
1912/// theirs, so neither is touched here);
1913/// 2. a regular region that claimed no cell is dropped (`keep_empty_clusters`
1914/// is off; a `formula` is kept, as upstream keeps it);
1915/// 3. an orphan text region (`score == 0.0`, from [`add_orphan_regions`]) that
1916/// now sits > 0.8 inside another regular region's fitted box is folded into
1917/// it (`_remove_overlapping_clusters` at containment 0.8, the larger box
1918/// winning the group) — up to three rounds, like upstream's loop.
1919///
1920/// Why it matters: the layout model's box can end partway through a line. That
1921/// line fails the 0.2 claim and becomes an orphan — recoverable — but the
1922/// *model* box still overlaps the orphan's line by a few points, so the
1923/// reading-order graph, which links only strictly-above pairs, gets no edge
1924/// between them and may emit the next paragraph first, stranding the line
1925/// after the paragraph it belongs in (1540 of 6050 text blocks on the #419
1926/// book began mid-sentence). Fitted to its cells, the box ends on a line
1927/// boundary and the orphan slots in between; an orphan the fitted box
1928/// swallows joins the paragraph outright. Cell assignment is untouched: a
1929/// region's fitted box contains every cell it claimed, so
1930/// [`region_texts_exclusive`] hands it the same cells afterwards.
1931///
1932/// A page with no cells yet (a scan before OCR) is left alone: dropping every
1933/// text region for want of cells would be wrong, and the OCR paths call this
1934/// again once the cells exist.
1935pub fn fit_regions_to_cells(regions: &mut Vec<Region>, cells: &[TextCell]) {
1936 if !cells.iter().any(|c| !c.text.trim().is_empty()) {
1937 return;
1938 }
1939 for _ in 0..3 {
1940 let owned = assign_cells(regions, cells);
1941 let mut fitted: Vec<Region> = Vec::with_capacity(regions.len());
1942 for (r, own) in regions.iter().zip(&owned) {
1943 if !claims_cells(r) {
1944 fitted.push(r.clone());
1945 continue;
1946 }
1947 if own.is_empty() {
1948 if r.label == "formula" {
1949 fitted.push(r.clone());
1950 }
1951 continue;
1952 }
1953 let mut f = r.clone();
1954 f.l = own
1955 .iter()
1956 .map(|&i| cells[i].l)
1957 .fold(f32::INFINITY, f32::min);
1958 f.t = own
1959 .iter()
1960 .map(|&i| cells[i].t)
1961 .fold(f32::INFINITY, f32::min);
1962 f.r = own
1963 .iter()
1964 .map(|&i| cells[i].r)
1965 .fold(f32::NEG_INFINITY, f32::max);
1966 f.b = own
1967 .iter()
1968 .map(|&i| cells[i].b)
1969 .fold(f32::NEG_INFINITY, f32::max);
1970 fitted.push(f);
1971 }
1972 let mut changed = fitted.len() != regions.len();
1973 // Fold orphans into the regular region whose fitted box holds them.
1974 let mut drop = vec![false; fitted.len()];
1975 for i in 0..fitted.len() {
1976 let o = &fitted[i];
1977 if !(o.score == 0.0 && o.label == "text") {
1978 continue;
1979 }
1980 let oa = area(o.l, o.t, o.r, o.b).max(1.0);
1981 let mut best: Option<(usize, f32)> = None;
1982 for (j, r) in fitted.iter().enumerate() {
1983 if j == i || drop[j] || r.score == 0.0 || !claims_cells(r) {
1984 continue;
1985 }
1986 let ov = inter(r, o.l, o.t, o.r, o.b) / oa;
1987 if ov > 0.8 && best.is_none_or(|(_, b)| ov > b) {
1988 best = Some((j, ov));
1989 }
1990 }
1991 if let Some((j, _)) = best {
1992 let (l, t, r, b) = (o.l, o.t, o.r, o.b);
1993 let host = &mut fitted[j];
1994 host.l = host.l.min(l);
1995 host.t = host.t.min(t);
1996 host.r = host.r.max(r);
1997 host.b = host.b.max(b);
1998 drop[i] = true;
1999 changed = true;
2000 }
2001 }
2002 let mut drop = drop.into_iter();
2003 fitted.retain(|_| !drop.next().expect("aligned"));
2004 *regions = fitted;
2005 if !changed {
2006 break;
2007 }
2008 }
2009}
2010
2011/// Join a prefiltered cell list into the region's text (docling's
2012/// `sanitize_text` over the sanitizer's cell order).
2013fn cells_text(inside: Vec<&TextCell>) -> String {
2014 // docling orders a cluster's cells by their docling-parse cell index
2015 // alone (`LayoutPostprocessor._sort_cells`: `sorted(cells, key=c.index)`)
2016 // — the sanitizer's output order, which our `cells` slice already is.
2017 // No geometric re-sort: normal_4pages' big section numerals paint
2018 // *after* their heading text, and docling's `## 들어가며 1` (numeral
2019 // last) only falls out of pure index order — a band sort dragged the
2020 // numeral to the front. The overlap-grouped line restore this replaced
2021 // measured strictly worse on the corpus (it fixed nothing the index
2022 // order broke, and broke the numerals).
2023 let joined = {
2024 // docling's `PageAssembleModel.sanitize_text`, ported verbatim over the
2025 // parse-index-ordered lines: append a separating space to a line —
2026 // unless it ends with `-`. A dash-ending line whose last word and the
2027 // next line's first word are both alphanumeric is a wrapped word: the
2028 // dash is dropped and the lines fuse (`platforms-` + `reflects` →
2029 // `platformsreflects`, `pp. 545-` + `561` → `545561`). Any other
2030 // dash-ending line — e.g. the *bare* `-` cell a superscript ORCID or an
2031 // inline `–` bullet splits off (its word list is empty, so the fuse
2032 // test fails) — keeps its dash and still takes no trailing space:
2033 // `[0000` `-` `0002` joins as docling's `[0000 -0002`, and the OTSL
2034 // list's `-` + `"C" cell -` + `a new table cell` collapses to
2035 // `-"C" cell a new table cell`. Our cells still carry the raw dash
2036 // family (docling-parse normalizes to `-` before this; clean_text does
2037 // it after), so the endswith test matches them all.
2038 let texts: Vec<&str> = inside
2039 .iter()
2040 .map(|c| c.text.trim())
2041 // Skip whitespace-only cells (a justified line's trailing space
2042 // glyph): an empty line would double the separator.
2043 .filter(|t| !t.is_empty())
2044 .collect();
2045 let last_word_alnum = |s: &str| {
2046 s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
2047 .rfind(|w| !w.is_empty())
2048 .is_some_and(|w| w.chars().all(char::is_alphanumeric))
2049 };
2050 let first_word_alnum = |s: &str| {
2051 s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
2052 .find(|w| !w.is_empty())
2053 .is_some_and(|w| w.chars().all(char::is_alphanumeric))
2054 };
2055 let mut out = String::new();
2056 for (i, t) in texts.iter().enumerate() {
2057 if i > 0 {
2058 let prev = texts[i - 1];
2059 let dashish = matches!(
2060 prev.chars().last(),
2061 Some(
2062 '-' | '\u{2010}'
2063 | '\u{2011}'
2064 | '\u{2012}'
2065 | '\u{2013}'
2066 | '\u{2014}'
2067 | '\u{2015}'
2068 | '\u{2212}'
2069 )
2070 );
2071 // docling#4052 (2.122): a dash only splits a word when it is
2072 // *attached* to one — the character before it is alphanumeric.
2073 // A dash that follows whitespace (a separator dash, a bullet
2074 // marker, a wrapped `-prefixed` token, the bare `-` cell an
2075 // ORCID splits off) is a literal character: it is kept and the
2076 // lines join with the ordinary space.
2077 let attached = prev.chars().rev().nth(1).is_some_and(char::is_alphanumeric);
2078 if dashish && attached {
2079 if last_word_alnum(prev) && first_word_alnum(t) {
2080 out.pop(); // wrapped word: fuse without the dash
2081 }
2082 // an attached dash never takes a separating space
2083 } else {
2084 out.push(' ');
2085 }
2086 }
2087 out.push_str(t);
2088 }
2089 out
2090 };
2091 clean_text(&joined)
2092}
2093
2094/// Tighten the spaces pdfium leaves around tight punctuation in a code line
2095/// (`console .log` → `console.log`, `add (3 , 5)` → `add(3, 5)`), matching
2096/// docling-parse's source spacing.
2097fn tighten_code_punct(s: &str) -> String {
2098 s.replace(" .", ".")
2099 .replace(" ,", ",")
2100 .replace(" ;", ";")
2101 .replace(" )", ")")
2102 .replace(" (", "(")
2103}
2104
2105/// Assemble a **code** region's text with its line structure preserved.
2106///
2107/// Unlike [`region_text`] — which joins every cell with a single space, the right
2108/// thing for prose reflow — a code block's line breaks and indentation are
2109/// significant. The `code_cells` are already one physical source line each
2110/// (grouped space-glyph-only, so monospace runs keep their spacing), so this:
2111///
2112/// 1. groups the cells into vertical line bands and orders them top→bottom,
2113/// left→right;
2114/// 2. joins the lines with `\n` (rather than spaces), keeping the carriage
2115/// returns; and
2116/// 3. reconstructs each line's leading indentation from its left offset, in units
2117/// of the block's estimated monospace character width, so nesting survives.
2118///
2119/// Typography is normalized per line via [`clean_text`] (smart quotes, dashes,
2120/// ellipsis), which never merges lines. Returns an empty string if the region has
2121/// no code cells (the caller falls back to the prose text).
2122fn code_region_text(region: &Region, cells: &[TextCell]) -> String {
2123 let mut inside: Vec<&TextCell> = cells
2124 .iter()
2125 .filter(|c| {
2126 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2127 inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
2128 })
2129 .filter(|c| !c.text.trim().is_empty())
2130 .collect();
2131 if inside.is_empty() {
2132 return String::new();
2133 }
2134
2135 // Quantize the top edge into ~line bands (like `region_text`), then order the
2136 // cells by band (top→bottom) and, within a band, by left edge.
2137 let band = inside
2138 .iter()
2139 .map(|c| (c.b - c.t).abs())
2140 .fold(0.0f32, f32::max)
2141 .max(1.0);
2142 let line_of = |c: &TextCell| (c.t / band).round() as i64;
2143 inside.sort_by_key(|c| (line_of(c), (c.l * 10.0) as i64));
2144
2145 // Estimate one monospace character's width (total ink width / total glyphs) to
2146 // convert a line's left offset into a count of leading spaces. Measured over
2147 // all lines so a single short line can't skew it.
2148 let (mut total_w, mut total_chars) = (0.0f32, 0usize);
2149 for c in &inside {
2150 let n = c.text.trim().chars().count();
2151 if n > 0 {
2152 total_w += (c.r - c.l).max(0.0);
2153 total_chars += n;
2154 }
2155 }
2156 let char_w = if total_chars > 0 {
2157 (total_w / total_chars as f32).max(1.0)
2158 } else {
2159 1.0
2160 };
2161 // The block's own left margin is the zero-indent baseline.
2162 let base_l = inside.iter().map(|c| c.l).fold(f32::INFINITY, f32::min);
2163
2164 let mut lines: Vec<String> = Vec::new();
2165 let mut cur: Option<i64> = None;
2166 for c in &inside {
2167 // Tighten pdfium's spaced punctuation per line (on the trimmed content, so
2168 // the reconstructed leading indentation is never nibbled).
2169 let text = tighten_code_punct(&clean_text(c.text.trim()));
2170 if Some(line_of(c)) == cur {
2171 // A second cell sharing this band (rare — e.g. split columns): keep it
2172 // on the same source line, separated by a space.
2173 if let Some(last) = lines.last_mut() {
2174 last.push(' ');
2175 last.push_str(&text);
2176 }
2177 continue;
2178 }
2179 let indent = ((c.l - base_l) / char_w).round().max(0.0) as usize;
2180 lines.push(format!("{}{}", " ".repeat(indent), text));
2181 cur = Some(line_of(c));
2182 }
2183 lines.join("\n")
2184}
2185
2186/// Reconstruct a table's grid geometrically from the text cells inside its
2187/// region: cluster cells into rows (by vertical centre) and columns (by clustered
2188/// left edges), then place each cell. A model-free stand-in for TableFormer that
2189/// recovers grid-aligned tables from the precise PDF text layer (it does not
2190/// resolve row/column spans).
2191pub fn reconstruct_table(region: &Region, cells: &[TextCell]) -> Vec<Vec<String>> {
2192 let mut inside: Vec<&TextCell> = cells
2193 .iter()
2194 .filter(|c| {
2195 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2196 inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
2197 })
2198 .collect();
2199 if inside.is_empty() {
2200 return Vec::new();
2201 }
2202 inside.sort_by(|a, b| a.t.total_cmp(&b.t));
2203
2204 // Rows: consecutive cells whose vertical centre is within ~0.7 line height.
2205 let mut rows: Vec<(f32, Vec<&TextCell>)> = Vec::new();
2206 for c in &inside {
2207 let cyc = (c.t + c.b) / 2.0;
2208 let lh = (c.b - c.t).abs().max(1.0);
2209 if let Some((ryc, row)) = rows.last_mut() {
2210 if (cyc - *ryc).abs() < lh * 0.7 {
2211 row.push(c);
2212 continue;
2213 }
2214 }
2215 rows.push((cyc, vec![c]));
2216 }
2217
2218 // Columns: cluster left edges (merge those within a tolerance).
2219 let tol = {
2220 let mut hs: Vec<f32> = inside.iter().map(|c| (c.b - c.t).abs()).collect();
2221 hs.sort_by(f32::total_cmp);
2222 hs[hs.len() / 2].max(4.0) * 1.5
2223 };
2224 let mut lefts: Vec<f32> = inside.iter().map(|c| c.l).collect();
2225 lefts.sort_by(f32::total_cmp);
2226 let mut col_starts: Vec<f32> = Vec::new();
2227 for l in lefts {
2228 if col_starts.last().is_none_or(|&last| l - last > tol) {
2229 col_starts.push(l);
2230 }
2231 }
2232 let ncols = col_starts.len().max(1);
2233 let col_of = |l: f32| -> usize {
2234 col_starts
2235 .iter()
2236 .rposition(|&s| l + tol * 0.5 >= s)
2237 .unwrap_or(0)
2238 .min(ncols - 1)
2239 };
2240
2241 let mut grid = Vec::with_capacity(rows.len());
2242 for (_, mut row) in rows {
2243 row.sort_by(|a, b| a.l.total_cmp(&b.l));
2244 let mut cols = vec![String::new(); ncols];
2245 for c in row {
2246 let ci = col_of(c.l);
2247 // Strip the wrap-hyphen control char so it never lands in a cell.
2248 let t = c.text.trim().replace(['\u{2}', '\u{ad}'], "");
2249 if cols[ci].is_empty() {
2250 cols[ci] = t;
2251 } else {
2252 cols[ci].push(' ');
2253 cols[ci].push_str(&t);
2254 }
2255 }
2256 grid.push(cols);
2257 }
2258 grid
2259}
2260
2261/// Does the geometric reconstruction of a table look trustworthy enough to use
2262/// as-is, instead of paying for TableFormer?
2263///
2264/// [`reconstruct_table`] derives columns by clustering cell **left edges**. On a
2265/// clean grid that is exact, but when a column's entries are not left-aligned
2266/// (or the OCR boxes wobble) the clustering splits one real column into several,
2267/// and the result is a wide, mostly-empty grid — the "spurious empty columns"
2268/// failure TableFormer exists to fix.
2269///
2270/// Two symptoms separate the two cases, and both are properties of the grid
2271/// alone (no model needed):
2272/// * **density** — a real table is mostly full; a split-up one is mostly holes;
2273/// * **thin columns** — a column carrying at most one entry across several rows
2274/// is almost always a split artefact rather than a real column.
2275///
2276/// Deliberately conservative: it answers `true` only for grids that are plainly
2277/// well-formed, so the expensive path stays the default whenever there is doubt.
2278/// A caller that skips TableFormer on `true` trades no quality for the time.
2279pub fn geometric_table_is_reliable(rows: &[Vec<String>]) -> bool {
2280 let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
2281 // Fewer than two columns is not a grid this heuristic can vouch for: it is
2282 // exactly the shape a collapsed table takes, and TableFormer may recover
2283 // real structure from it.
2284 if rows.len() < 2 || ncols < 2 {
2285 return false;
2286 }
2287 let filled = |c: &String| !c.trim().is_empty();
2288 let total = rows.len() * ncols;
2289 let full = rows.iter().flatten().filter(|c| filled(c)).count();
2290 if (full as f32) < MIN_TABLE_FILL * total as f32 {
2291 return false;
2292 }
2293 // A column used by at most one row, when there are rows enough to tell.
2294 if rows.len() >= 3 {
2295 for ci in 0..ncols {
2296 let used = rows
2297 .iter()
2298 .filter(|r| r.get(ci).is_some_and(filled))
2299 .count();
2300 if used <= 1 {
2301 return false;
2302 }
2303 }
2304 }
2305 true
2306}
2307
2308/// Share of a geometric grid's cells that must carry text for it to be trusted
2309/// without TableFormer. Chosen well above the density a left-edge split
2310/// produces (those land nearer a third) and below what a genuine table with a
2311/// few blank cells reaches.
2312const MIN_TABLE_FILL: f32 = 0.6;
2313
2314/// The union bbox of the text cells assigned to a region (same >50%-overlap
2315/// rule as [`region_text`]), or `None` when no cell lands in it. docling's
2316/// LayoutPostprocessor shrinks a regular cluster's bbox to its cells, and the
2317/// enrichment crops are taken from that cell-tight box — cropping the raw
2318/// detector box instead hands the VLM surrounding chrome (e.g. the `Listing N:`
2319/// caption under a code block) that changes its output.
2320pub fn region_cell_bbox(region: &Region, cells: &[TextCell]) -> Option<[f32; 4]> {
2321 let mut bbox: Option<[f32; 4]> = None;
2322 for c in cells {
2323 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2324 if inter(region, c.l, c.t, c.r, c.b) / ca <= 0.5 {
2325 continue;
2326 }
2327 bbox = Some(match bbox {
2328 None => [c.l, c.t, c.r, c.b],
2329 Some([l, t, r, b]) => [l.min(c.l), t.min(c.t), r.max(c.r), b.max(c.b)],
2330 });
2331 }
2332 bbox
2333}
2334
2335/// One region's enrichment-model result, produced by the pipeline's opt-in
2336/// passes (issue #76) and applied during assembly.
2337#[derive(Debug, Clone)]
2338pub enum Enrichment {
2339 /// DocumentPictureClassifier predictions, descending confidence.
2340 PictureClasses(Vec<PictureClass>),
2341 /// CodeFormulaV2 output for a `code` region: the rewritten source text and
2342 /// the `<_language_>` prefix (when the model emitted one).
2343 Code {
2344 language: Option<String>,
2345 text: String,
2346 },
2347 /// CodeFormulaV2 output for a `formula` region: the decoded LaTeX.
2348 Formula { latex: String },
2349}
2350
2351/// Crop a region (page points, already expanded by the caller if needed) from
2352/// the rendered page image and resize it to `target_scale` pixels per point —
2353/// the enrichment-model equivalent of docling's
2354/// `page.get_image(scale=…, cropbox=…)`, sourced from the existing
2355/// [`crate::pdfium_backend::RENDER_SCALE`] render instead of a fresh pdfium
2356/// pass (the page bitmap is already the exact docling render at scale 2).
2357#[cfg(feature = "ml")]
2358pub fn crop_region_scaled(page: &PdfPage, bbox: [f32; 4], target_scale: f32) -> Option<RgbImage> {
2359 let s = page.scale;
2360 let [l, t, r, b] = bbox;
2361 let (iw, ih) = (page.image.width(), page.image.height());
2362 let x = (l * s).max(0.0) as u32;
2363 let y = (t * s).max(0.0) as u32;
2364 if x >= iw || y >= ih {
2365 return None;
2366 }
2367 let w = (((r - l.max(0.0)) * s) as u32).min(iw - x);
2368 let h = (((b - t.max(0.0)) * s) as u32).min(ih - y);
2369 if w == 0 || h == 0 {
2370 return None;
2371 }
2372 let crop = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
2373 // docling renders the crop at `target_scale` directly; from the scale-2
2374 // page render that is a resize to the same pixel geometry
2375 // (`round(width_points * scale)`, PIL's BICUBIC ≙ CatmullRom).
2376 let tw = ((w as f32 / s) * target_scale).round().max(1.0) as u32;
2377 let th = ((h as f32 / s) * target_scale).round().max(1.0) as u32;
2378 if (tw, th) == (w, h) {
2379 return Some(crop);
2380 }
2381 Some(image::imageops::resize(
2382 &crop,
2383 tw,
2384 th,
2385 image::imageops::FilterType::CatmullRom,
2386 ))
2387}
2388
2389/// Resample `img` (rendered at `from` px/pt, covering `w_pt`×`h_pt` points)
2390/// to `to` px/pt — docling's `round(points * scale)` pixel geometry, PIL's
2391/// BICUBIC ≙ CatmullRom. Unchanged when the geometry already matches.
2392#[cfg(feature = "ocr-prep")]
2393fn rescale(img: RgbImage, w_pt: f32, h_pt: f32, to: f32) -> RgbImage {
2394 let tw = (w_pt * to).round().max(1.0) as u32;
2395 let th = (h_pt * to).round().max(1.0) as u32;
2396 if (tw, th) == img.dimensions() {
2397 return img;
2398 }
2399 image::imageops::resize(&img, tw, th, image::imageops::FilterType::CatmullRom)
2400}
2401
2402/// Encode `img` as a PNG [`PictureImage`] rendered at `scale` px/pt.
2403#[cfg(feature = "ocr-prep")]
2404fn png_image(img: &RgbImage, scale: f32) -> Option<PictureImage> {
2405 let mut buf = std::io::Cursor::new(Vec::new());
2406 img.write_to(&mut buf, image::ImageFormat::Png).ok()?;
2407 Some(PictureImage {
2408 mimetype: "image/png".into(),
2409 width: img.width(),
2410 height: img.height(),
2411 data: buf.into_inner(),
2412 dpi: PictureImage::dpi_for_scale(scale),
2413 })
2414}
2415
2416/// The whole page render as docling's `PageItem.image` (#520): at `scale`
2417/// px/pt (`None` = the render's own), `None` when the page has no bitmap.
2418#[cfg(feature = "ocr-prep")]
2419pub fn page_image(page: &PdfPage, scale: Option<f32>) -> Option<PictureImage> {
2420 if page.image.width() == 0 || page.image.height() == 0 || page.scale <= 0.0 {
2421 return None;
2422 }
2423 let scale = scale.unwrap_or(page.scale);
2424 let img = rescale(page.image.clone(), page.width, page.height, scale);
2425 png_image(&img, scale)
2426}
2427
2428/// Crop a layout region from the rendered page image and encode it as PNG (the
2429/// figure bytes docling stores on a `PictureItem`). Region coordinates are page
2430/// points; the image is rendered at `page.scale` and resampled to `scale`
2431/// px/pt when one is given (docling's `images_scale`, #520). The image's `dpi`
2432/// is 72·scale (#519).
2433#[cfg(feature = "ocr-prep")]
2434fn crop_region(page: &PdfPage, region: &Region, scale: Option<f32>) -> Option<PictureImage> {
2435 let s = page.scale;
2436 let (iw, ih) = (page.image.width(), page.image.height());
2437 let x = (region.l * s).max(0.0) as u32;
2438 let y = (region.t * s).max(0.0) as u32;
2439 if x >= iw || y >= ih {
2440 return None;
2441 }
2442 let w = (((region.r - region.l) * s) as u32).min(iw - x);
2443 let h = (((region.b - region.t) * s) as u32).min(ih - y);
2444 if w == 0 || h == 0 {
2445 return None;
2446 }
2447 let sub = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
2448 match scale {
2449 Some(to) if (to - s).abs() > f32::EPSILON => {
2450 // The crop's own point extent (the pixel box, back in points), so
2451 // the resampled geometry is `round(points * scale)`.
2452 let img = rescale(sub, w as f32 / s, h as f32 / s, to);
2453 png_image(&img, to)
2454 }
2455 _ => png_image(&sub, s),
2456 }
2457}
2458
2459/// For each `picture` region, find the `caption` region closest below it (and
2460/// horizontally overlapping); docling pairs them and emits the caption first.
2461/// Each caption is claimed by at most one picture.
2462fn pair_captions(regions: &[Region]) -> Vec<Option<usize>> {
2463 let mut pairs = vec![None; regions.len()];
2464 let mut taken = vec![false; regions.len()];
2465 for (pi, p) in regions.iter().enumerate() {
2466 if p.label != "picture" {
2467 continue;
2468 }
2469 let mut best: Option<(usize, f32)> = None;
2470 for (ci, c) in regions.iter().enumerate() {
2471 if c.label != "caption" || taken[ci] {
2472 continue;
2473 }
2474 let line_h = (c.b - c.t).abs().max(1.0);
2475 let gap = c.t - p.b; // caption sits below the picture
2476 let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
2477 if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
2478 let dist = gap.abs();
2479 if best.is_none_or(|(_, bd)| dist < bd) {
2480 best = Some((ci, dist));
2481 }
2482 }
2483 }
2484 if let Some((ci, _)) = best {
2485 pairs[pi] = Some(ci);
2486 taken[ci] = true;
2487 }
2488 }
2489 pairs
2490}
2491
2492/// Pair each `code` region with the `caption` region just **above** it (a
2493/// `Listing N:` label). docling renders the code block first, then its caption,
2494/// so the caption is consumed from its own (earlier) reading-order slot and
2495/// re-emitted after the code.
2496fn pair_code_captions(regions: &[Region]) -> Vec<Option<usize>> {
2497 let mut pairs = vec![None; regions.len()];
2498 let mut taken = vec![false; regions.len()];
2499 for (pi, p) in regions.iter().enumerate() {
2500 if p.label != "code" {
2501 continue;
2502 }
2503 let mut best: Option<(usize, f32)> = None;
2504 for (ci, c) in regions.iter().enumerate() {
2505 if c.label != "caption" || taken[ci] {
2506 continue;
2507 }
2508 let line_h = (c.b - c.t).abs().max(1.0);
2509 let gap = p.t - c.b; // caption sits above the code
2510 let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
2511 if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
2512 let dist = gap.abs();
2513 if best.is_none_or(|(_, bd)| dist < bd) {
2514 best = Some((ci, dist));
2515 }
2516 }
2517 }
2518 if let Some((ci, _)) = best {
2519 pairs[pi] = Some(ci);
2520 taken[ci] = true;
2521 }
2522 }
2523 pairs
2524}
2525
2526/// Pair each `table`/`document_index` region with its `caption` (#265) the way
2527/// docling's `ReadingOrderPredictor._find_to_captions` does: by **reading-order
2528/// adjacency**, not geometry. A caption claims the media element
2529/// (table/picture/code) immediately next to it in the ordered region sequence,
2530/// and only when exactly one side holds one — a caption sandwiched between two
2531/// media elements stays unattached, and a text paragraph between caption and
2532/// table breaks the bond. This is what lets a flush-left "Table 3: …" label
2533/// bind a centered grid it doesn't horizontally overlap, while a caption in
2534/// the neighbouring column of a two-column page — geometrically close — never
2535/// pairs across the gutter. Runs after the picture and code pairings (the
2536/// picture/code arms of the same upstream matcher), so a caption they claimed
2537/// stays claimed. docling attaches these as `TableItem.captions` refs; the
2538/// paired caption is consumed from its own reading-order slot and rides on the
2539/// table node instead.
2540fn pair_table_captions(regions: &[Region], taken: &mut [bool]) -> Vec<Option<usize>> {
2541 let is_media = |label: &str| is_table_like(label) || matches!(label, "picture" | "code");
2542 let mut pairs: Vec<Option<usize>> = vec![None; regions.len()];
2543 for ci in 0..regions.len() {
2544 if regions[ci].label != "caption" || taken[ci] {
2545 continue;
2546 }
2547 // Furniture (headers/footers, form chrome) is not part of docling's
2548 // body-element sequence, so it neither bonds nor blocks.
2549 let prev = regions[..ci].iter().rposition(|r| !is_skipped(r.label));
2550 let next = regions[ci + 1..]
2551 .iter()
2552 .position(|r| !is_skipped(r.label))
2553 .map(|off| ci + 1 + off);
2554 let prev_media = prev.is_some_and(|j| is_media(regions[j].label));
2555 let next_media = next.is_some_and(|j| is_media(regions[j].label));
2556 let target = match (prev_media, next_media) {
2557 (true, false) => prev,
2558 (false, true) => next,
2559 // Ambiguous (media on both sides) or no media at all: leave the
2560 // caption in its own reading-order slot, as docling does.
2561 _ => None,
2562 };
2563 if let Some(ti) = target {
2564 // A first claim wins (a table with captions above *and* below
2565 // keeps the earlier one — docling's nearest-first tiebreak).
2566 if is_table_like(regions[ti].label) && pairs[ti].is_none() {
2567 pairs[ti] = Some(ci);
2568 taken[ci] = true;
2569 }
2570 }
2571 }
2572 pairs
2573}
2574
2575/// Assemble one page from its (already overlap-resolved) layout regions and
2576/// text cells.
2577/// Normalize a layout region (page points, top-left origin) to DocLang's 0–511
2578/// location grid: `clamp(round(512 · coord / page_dim), 0, 511)`, per axis,
2579/// order `[x0, y0, x1, y1]`. Mirrors docling_core's
2580/// `_doclang_utils._create_location_tokens_for_bbox` (resolution 512) so the
2581/// emitted `<location>` tokens line up with the Python groundtruth. Our heron
2582/// cluster boxes match docling's to within ~1 grid unit; the residual (mainly
2583/// the aspect-ratio-stretch vs letterbox preprocessing difference) is absorbed
2584/// by the conformance harness's geometry tolerance.
2585fn norm_loc(region: &Region, page_w: f32, page_h: f32) -> [u16; 4] {
2586 let q = |v: f32, dim: f32| -> u16 {
2587 if dim <= 0.0 {
2588 return 0;
2589 }
2590 let g = (512.0 * (v as f64) / (dim as f64)).round() as i64;
2591 g.clamp(0, 511) as u16
2592 };
2593 [
2594 q(region.l, page_w),
2595 q(region.t, page_h),
2596 q(region.r, page_w),
2597 q(region.b, page_h),
2598 ]
2599}
2600
2601/// Wrap a node in its layout provenance so the DocLang serializer emits the four
2602/// `<location>` tokens as the element's head (Markdown/JSON render `inner`
2603/// unchanged).
2604fn located(loc: [u16; 4], inner: Node) -> Node {
2605 Node::Located {
2606 location: loc,
2607 inner: Box::new(inner),
2608 }
2609}
2610
2611/// Stamp the real 1-based page number onto a page's leading marker (see
2612/// [`assemble_page`], which emits it with `page_no: 0` because only the
2613/// document-level collector knows the true index — `--pages` windows shift it).
2614pub fn stamp_page_no(nodes: &mut [Node], page_no: usize) {
2615 if let Some(Node::PageInfo { page_no: p, .. }) = nodes.first_mut() {
2616 *p = page_no;
2617 }
2618}
2619
2620/// A dense table grid plus its first-class cells (#240): `rows` is the text
2621/// grid every serializer renders (spans replicate their anchor's text);
2622/// `cells` are the docling-parity per-cell records (text, page-point bbox,
2623/// span rectangle, OTSL header roles). Produced by the TableFormer paths
2624/// (`tf_core`); lives in this always-compiled module so the pure-text (wasm
2625/// `pdf-text`) build sees the type.
2626#[derive(Clone, Debug)]
2627pub struct TableGrid {
2628 pub rows: Vec<Vec<String>>,
2629 pub cells: Vec<docling_core::TableCell>,
2630}
2631
2632/// docling's `_RICH_CELL_PICTURE_COVERAGE_THRESHOLD`.
2633const RICH_CELL_PICTURE_COVERAGE: f32 = 0.8;
2634
2635/// docling `ReadingOrderModel._match_table_pictures` (#3906, 2.118.1): every
2636/// picture ≥ 80 % inside a TableFormer-structured table on the page is matched
2637/// to the cell covering it, and returned per table as `cell index → pictures`.
2638/// A picture that pairs with a caption stays a standalone figure (upstream
2639/// would nest it and lose the caption; keeping the caption is the better
2640/// failure). Tables without first-class cells (geometric fallback) have no cell
2641/// boxes to match against and nest nothing.
2642fn match_table_pictures(
2643 regions: &[Region],
2644 table_rows: &[Option<TableGrid>],
2645 caption_for: &[Option<usize>],
2646) -> std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> {
2647 let mut out: std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> =
2648 std::collections::HashMap::new();
2649 for (p, pic) in regions.iter().enumerate() {
2650 if pic.label != "picture" || caption_for.get(p).is_some_and(Option::is_some) {
2651 continue;
2652 }
2653 let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2654 let mut best: Option<(f32, usize, usize)> = None; // (coverage, table, cell)
2655 for (t, tbl) in regions.iter().enumerate() {
2656 if !is_table_like(tbl.label) {
2657 continue;
2658 }
2659 let Some(grid) = table_rows.get(t).and_then(Option::as_ref) else {
2660 continue;
2661 };
2662 if inter(pic, tbl.l, tbl.t, tbl.r, tbl.b) / pa < RICH_CELL_PICTURE_COVERAGE {
2663 continue;
2664 }
2665 if let Some((cov, cell)) = match_picture_to_cell(pic, &grid.cells) {
2666 if best.is_none_or(|(b, _, _)| cov > b) {
2667 best = Some((cov, t, cell));
2668 }
2669 }
2670 }
2671 if let Some((_, t, cell)) = best {
2672 let entry = out.entry(t).or_default();
2673 match entry.iter_mut().find(|(c, _)| *c == cell) {
2674 Some((_, pics)) => pics.push(p),
2675 None => entry.push((cell, vec![p])),
2676 }
2677 }
2678 }
2679 out
2680}
2681
2682/// docling `_match_picture_to_table_cell`: among the cells covering ≥ 80 % of
2683/// the picture, prefer the one at the picture's inferred grid position (the
2684/// row / column whose median cell center is nearest the picture's center —
2685/// cell boxes can overlap across logical rows and columns), else the best
2686/// coverage. Returns `(coverage, cell index)`.
2687fn match_picture_to_cell(pic: &Region, cells: &[docling_core::TableCell]) -> Option<(f32, usize)> {
2688 let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2689 let cover = |b: &[f32; 4]| inter(pic, b[0], b[1], b[2], b[3]) / pa;
2690 let eligible: Vec<(f32, usize)> = cells
2691 .iter()
2692 .enumerate()
2693 .filter_map(|(i, c)| {
2694 let b = c.bbox.as_ref()?;
2695 let cov = cover(b);
2696 (cov >= RICH_CELL_PICTURE_COVERAGE).then_some((cov, i))
2697 })
2698 .collect();
2699 if eligible.is_empty() {
2700 return None;
2701 }
2702 let mut row_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2703 let mut col_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2704 for c in cells {
2705 let Some(b) = c.bbox.as_ref() else { continue };
2706 for r in c.start_row..c.start_row + c.row_span {
2707 row_centers.entry(r).or_default().push((b[1] + b[3]) / 2.0);
2708 }
2709 for k in c.start_col..c.start_col + c.col_span {
2710 col_centers.entry(k).or_default().push((b[0] + b[2]) / 2.0);
2711 }
2712 }
2713 let median = |v: &mut Vec<f32>| -> f32 {
2714 v.sort_by(f32::total_cmp);
2715 let n = v.len();
2716 if n % 2 == 1 {
2717 v[n / 2]
2718 } else {
2719 (v[n / 2 - 1] + v[n / 2]) / 2.0
2720 }
2721 };
2722 let (px, py) = ((pic.l + pic.r) / 2.0, (pic.t + pic.b) / 2.0);
2723 let nearest = |centers: &mut std::collections::BTreeMap<usize, Vec<f32>>, target: f32| {
2724 centers
2725 .iter_mut()
2726 .map(|(&i, v)| (i, (median(v) - target).abs()))
2727 .min_by(|a, b| a.1.total_cmp(&b.1))
2728 .map(|(i, _)| i)
2729 };
2730 let row = nearest(&mut row_centers, py);
2731 let col = nearest(&mut col_centers, px);
2732 let logical: Vec<(f32, usize)> = eligible
2733 .iter()
2734 .copied()
2735 .filter(|&(_, i)| {
2736 let c = &cells[i];
2737 row.is_some_and(|r| c.start_row <= r && r < c.start_row + c.row_span)
2738 && col.is_some_and(|k| c.start_col <= k && k < c.start_col + c.col_span)
2739 })
2740 .collect();
2741 let pool = if logical.is_empty() {
2742 &eligible
2743 } else {
2744 &logical
2745 };
2746 // Python's `max` over `(coverage, cell_index, cell)` tuples: highest
2747 // coverage, ties to the higher index.
2748 pool.iter()
2749 .copied()
2750 .max_by(|a, b| a.0.total_cmp(&b.0).then(a.1.cmp(&b.1)))
2751}
2752
2753/// The DocLang structure overlay derived from first-class cells: span
2754/// continuations (`lcel`/`ucel`/`xcel`) and per-cell header roles, so the
2755/// PDF path's DCLX carries real spans instead of a flat grid.
2756fn structure_from_cells(
2757 cells: &[docling_core::TableCell],
2758 nrows: usize,
2759 ncols: usize,
2760) -> docling_core::TableStructure {
2761 let grid = || vec![vec![false; ncols]; nrows];
2762 let mut col_cont = grid();
2763 let mut row_cont = grid();
2764 let mut row_header = grid();
2765 let mut col_header = grid();
2766 for c in cells {
2767 for r in c.start_row..(c.start_row + c.row_span).min(nrows) {
2768 for k in c.start_col..(c.start_col + c.col_span).min(ncols) {
2769 col_cont[r][k] = k > c.start_col;
2770 row_cont[r][k] = r > c.start_row;
2771 row_header[r][k] = c.row_header;
2772 col_header[r][k] = c.column_header;
2773 }
2774 }
2775 }
2776 docling_core::TableStructure {
2777 header_row: Vec::new(),
2778 col_continuation: col_cont,
2779 row_continuation: row_cont,
2780 row_header,
2781 col_header,
2782 }
2783}
2784
2785pub fn assemble_page(
2786 page: &PdfPage,
2787 regions: Vec<Region>,
2788 table_rows: &[Option<TableGrid>],
2789 enrichments: &[Option<Enrichment>],
2790 // Picture-crop scale in px/pt (docling's `images_scale`, #520); `None`
2791 // keeps the page render's own scale.
2792 picture_scale: Option<f32>,
2793) -> (Vec<Node>, Vec<(String, String)>) {
2794 // Without pixels (the text-layer-only wasm build) no picture is cropped.
2795 #[cfg(not(feature = "ocr-prep"))]
2796 let _ = picture_scale;
2797 let mut nodes: Vec<Node> = Vec::new();
2798 // Every page opens with an invisible page marker carrying its size in
2799 // points — what the JSON export needs to build docling's `pages` map and
2800 // denormalize the 0–511 `<location>` grid into point bboxes (#171). The
2801 // page *number* is stamped by the document-level collector (which knows
2802 // the real 1-based index, `--pages` windows included); every serializer
2803 // except JSON skips the marker, so Markdown/DocLang stay byte-identical.
2804 nodes.push(Node::PageInfo {
2805 page_no: 0,
2806 width: page.width,
2807 height: page.height,
2808 });
2809 // Recover this page's hyperlinks (anchor-precise pairs for strict
2810 // Markdown; whole-item docling-parity links are baked below and their
2811 // pairs dropped from this list so strict output doesn't double-wrap).
2812 let mut links = resolve_link_anchors(page);
2813 // Pair each region with its precomputed TableFormer grid and enrichment
2814 // (indexed by original order) and order by reading order together, so they
2815 // stay aligned.
2816 // A picture's children (docling's `_set_cluster_children`: the regulars
2817 // > 80 % inside it) are not page elements — they leave the reading order
2818 // here and ride with their picture, to be written under it in the JSON.
2819 let parents = picture_parents(®ions);
2820 let mut kids: Vec<Vec<Region>> = vec![Vec::new(); regions.len()];
2821 let mut top: Vec<(usize, Region)> = Vec::with_capacity(regions.len());
2822 for (i, (r, parent)) in regions.into_iter().zip(parents).enumerate() {
2823 match parent {
2824 Some(p) => kids[p].push(r),
2825 None => top.push((i, r)),
2826 }
2827 }
2828 // docling's assembly order of the regions — what its reading-order
2829 // predictor knows as `cid` (#424) — before they are shuffled.
2830 let top_regions: Vec<Region> = top.iter().map(|(_, r)| r.clone()).collect();
2831 let cids = cluster_cids(&top_regions, &page.cells);
2832 type RegionItem = (Region, Option<TableGrid>, Option<Enrichment>, Vec<Region>);
2833 let mut items: Vec<RegionItem> = top
2834 .into_iter()
2835 .map(|(i, r)| {
2836 (
2837 r,
2838 table_rows.get(i).cloned().flatten(),
2839 enrichments.get(i).cloned().flatten(),
2840 std::mem::take(&mut kids[i]),
2841 )
2842 })
2843 .collect();
2844 order_with_containers(&mut items, &cids, page.width, page.height, |it| &it.0);
2845 // Float a margin page number to the front of reading order (docling parity:
2846 // right_to_left_02's bottom `11` is its first item). Stable, so everything
2847 // else keeps its order; no-op on pages without such a region.
2848 let page_h = page.height;
2849 items.sort_by_key(|(r, _, _, _)| !is_page_number(r, &page.cells, page_h));
2850 let table_rows: Vec<Option<TableGrid>> = items.iter().map(|(_, t, _, _)| t.clone()).collect();
2851 let enrichments: Vec<Option<Enrichment>> = items.iter().map(|(_, _, e, _)| e.clone()).collect();
2852 let mut picture_children: Vec<Vec<Region>> = items
2853 .iter_mut()
2854 .map(|it| std::mem::take(&mut it.3))
2855 .collect();
2856 let regions: Vec<Region> = items.into_iter().map(|(r, _, _, _)| r).collect();
2857 // Children in docling's `_sort_clusters(mode="id")` order: first source
2858 // cell, then top, then left.
2859 for kids in picture_children.iter_mut().filter(|k| k.len() > 1) {
2860 let rank = cluster_cids(kids, &page.cells);
2861 let mut ranked: Vec<(usize, Region)> = rank.into_iter().zip(kids.drain(..)).collect();
2862 ranked.sort_by_key(|(k, _)| *k);
2863 kids.extend(ranked.into_iter().map(|(_, r)| r));
2864 }
2865 // docling emits a figure's caption *before* the image marker. Pair each
2866 // picture with the caption region nearest below it and consume that caption,
2867 // so it isn't also emitted in its own (lower) reading-order position.
2868 let caption_for = pair_captions(®ions);
2869 let code_caption_for = pair_code_captions(®ions);
2870 let mut consumed = vec![false; regions.len()];
2871 for ci in caption_for.iter().flatten() {
2872 consumed[*ci] = true;
2873 }
2874 for ci in code_caption_for.iter().flatten() {
2875 consumed[*ci] = true;
2876 }
2877 // Table captions (#265) claim from what the picture/code pairings left.
2878 let mut caption_taken = consumed.clone();
2879 let table_caption_for = pair_table_captions(®ions, &mut caption_taken);
2880 for ci in table_caption_for.iter().flatten() {
2881 consumed[*ci] = true;
2882 }
2883 // Pictures inside a table become rich-cell content (docling#3906, 2.118.1):
2884 // the picture is nested in the cell it covers and not emitted standalone.
2885 let rich_cell_pictures = match_table_pictures(®ions, &table_rows, &caption_for);
2886 for (_, pics) in rich_cell_pictures.values().flatten() {
2887 for &p in pics {
2888 consumed[p] = true;
2889 }
2890 }
2891 // A code block's language label (`XML`, `C#`, …) is chrome, not content — the
2892 // detector emits it as its own region above the code; consume it.
2893 for (i, is_label) in code_language_labels(®ions, &page.cells)
2894 .into_iter()
2895 .enumerate()
2896 {
2897 if is_label {
2898 consumed[i] = true;
2899 }
2900 }
2901
2902 // docling `ReadingOrderPredictor.predict_merges`: join a text fragment with a
2903 // following text fragment strictly to its right (an author column that wraps
2904 // into the next, a paragraph continuing in the next column) into one block —
2905 // the intra-page half of docling's reading-order merges (cross-page/vertical
2906 // continuations stay with [`merge_continuations`]). Already-consumed regions
2907 // (paired captions, code labels) are excluded.
2908 // Exclusive docling cell assignment: computed once for the ordered region
2909 // list and reused for every serialization below, so a cell can never render
2910 // in two regions. The picture children take part (docling assigns cells to
2911 // every regular cluster before it nests any); their texts are split off.
2912 let with_children: Vec<Region> = regions
2913 .iter()
2914 .chain(picture_children.iter().flatten())
2915 .cloned()
2916 .collect();
2917 let mut region_texts: Vec<String> = region_texts_exclusive(&with_children, &page.cells);
2918 let mut kid_texts = region_texts.split_off(regions.len()).into_iter();
2919 let child_texts: Vec<Vec<String>> = picture_children
2920 .iter()
2921 .map(|k| kid_texts.by_ref().take(k.len()).collect())
2922 .collect();
2923 let is_text: Vec<bool> = regions
2924 .iter()
2925 .enumerate()
2926 .map(|(i, r)| r.label == "text" && !consumed[i])
2927 .collect();
2928 let is_skip: Vec<bool> = regions
2929 .iter()
2930 .enumerate()
2931 .map(|(i, r)| {
2932 consumed[i]
2933 || matches!(
2934 r.label,
2935 "page_header" | "page_footer" | "table" | "picture" | "caption" | "footnote"
2936 )
2937 })
2938 .collect();
2939 let boxes: Vec<(f32, f32, f32, f32)> = regions.iter().map(|r| (r.l, r.t, r.r, r.b)).collect();
2940 if docling_core::env::flag("DOCLING_RS_DEBUG_MERGES") {
2941 for (i, r) in regions.iter().enumerate() {
2942 eprintln!(
2943 "MRG {i:2} {} text={} skip={} [{:.0},{:.0},{:.0},{:.0}] {:?}",
2944 r.label,
2945 is_text[i],
2946 is_skip[i],
2947 r.l,
2948 r.t,
2949 r.r,
2950 r.b,
2951 region_texts[i].chars().take(40).collect::<String>()
2952 );
2953 }
2954 }
2955 let mut merge_suffix: Vec<String> = vec![String::new(); regions.len()];
2956 for (head, children) in
2957 crate::reading_order::predict_merges(&boxes, ®ion_texts, &is_text, &is_skip)
2958 .into_iter()
2959 .enumerate()
2960 {
2961 for c in children {
2962 let t = region_texts[c].trim();
2963 if !t.is_empty() {
2964 merge_suffix[head].push(' ');
2965 merge_suffix[head].push_str(t);
2966 }
2967 consumed[c] = true;
2968 }
2969 }
2970
2971 for (i, region) in regions.iter().enumerate() {
2972 if consumed[i] {
2973 continue;
2974 }
2975 // Page headers/footers: docling emits them as furniture blocks
2976 // (`<page_header>`/`<page_footer>` with a layer + location + text) at
2977 // their reading-order position, not as body — emit them, don't skip.
2978 if matches!(region.label, "page_header" | "page_footer") {
2979 let text = region_texts[i].clone();
2980 if !text.is_empty() {
2981 nodes.push(Node::PageFurniture {
2982 footer: region.label == "page_footer",
2983 location: norm_loc(region, page.width, page_h),
2984 text: md_escape(&text),
2985 });
2986 }
2987 continue;
2988 }
2989 if is_skipped(region.label) {
2990 continue;
2991 }
2992 // Layout provenance for this region, normalized to docling's 0–511 grid.
2993 let loc = norm_loc(region, page.width, page_h);
2994 if region.label == "picture" {
2995 // The figure pixels are cropped from the page render for image export.
2996 // Captions are prose: markdown-escaped like a paragraph (the JSON
2997 // export unescapes back to the raw text, matching docling).
2998 let caption = caption_for[i]
2999 .map(|ci| md_escape(®ion_texts[ci]))
3000 .filter(|t| !t.is_empty());
3001 let classification = match &enrichments[i] {
3002 Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
3003 _ => None,
3004 };
3005 // Without the page render (text-layer-only build) a picture keeps
3006 // its caption/classification but carries no cropped pixels.
3007 #[cfg(feature = "ocr-prep")]
3008 let image =
3009 crate::timing::timed("crop_region", || crop_region(page, region, picture_scale));
3010 #[cfg(not(feature = "ocr-prep"))]
3011 let image: Option<PictureImage> = None;
3012 nodes.push(located(
3013 loc,
3014 Node::Picture {
3015 caption,
3016 caption_href: None,
3017 image,
3018 classification,
3019 // docling's layout pipeline parents a figure's caption to
3020 // the picture itself (#390) — the one backend that does.
3021 caption_parent: CaptionParent::Item,
3022 },
3023 ));
3024 let children: Vec<Node> = picture_children[i]
3025 .iter()
3026 .zip(&child_texts[i])
3027 .filter_map(|(r, text)| {
3028 picture_child_node(r, text, norm_loc(r, page.width, page_h))
3029 })
3030 .collect();
3031 if !children.is_empty() {
3032 nodes.push(Node::PictureChildren(children));
3033 }
3034 continue;
3035 }
3036 let mut text = region_texts[i].clone();
3037 text.push_str(&merge_suffix[i]);
3038 if text.is_empty() {
3039 continue;
3040 }
3041 match region.label {
3042 // docling assembles checkboxes as TEXT_ELEM items (the region's
3043 // cells are the option label, e.g. right_to_left_03's بلی/خير)
3044 // and its Markdown serializer renders them as task-list lines
3045 // (`- [x] …`) — mirrored by [`Node::CheckboxItem`].
3046 "checkbox_selected" | "checkbox_unselected" => nodes.push(Node::CheckboxItem {
3047 checked: region.label == "checkbox_selected",
3048 text: md_escape(&text),
3049 }),
3050 // docling renders both the document title and section headers as
3051 // `##` (it never emits a top-level `#` for PDFs), so match that.
3052 "title" | "section_header" => nodes.push(located(
3053 loc,
3054 Node::Heading {
3055 level: 2,
3056 text: md_escape(&text),
3057 },
3058 )),
3059 // docling's `ListItemMarkerProcessor.process_list_item` runs on
3060 // every PDF list item: a leading bullet glyph or enumeration marker
3061 // followed by whitespace is split off into the item's `marker`, and
3062 // docling-core's Markdown then prints `- text` for a bullet, `N. text`
3063 // for an `N.` marker and `- a) text` for any other marker holding a
3064 // letter or digit (see [`list_item_node`]). The symbol-font bullets
3065 // docling-parse filters out of its cells are stripped first.
3066 "list_item" => nodes.push(list_item_node(&text, loc, false)),
3067 // TableFormer structure (cells + spans, text matched from word cells)
3068 // when available; otherwise geometric grid reconstruction; finally a
3069 // single cell.
3070 "table" | "document_index" => {
3071 // TableFormer grids carry first-class cells (#240: text +
3072 // page-point bbox + span rectangle + OTSL header roles) into
3073 // the public model, and the DocLang structure overlay derives
3074 // from them so DCLX emits real span/header tokens. The
3075 // geometric fallback has no per-cell records.
3076 let (mut rows, cells, structure) = match table_rows[i].clone() {
3077 Some(grid) => {
3078 let nrows = grid.rows.len();
3079 let ncols = grid.rows.first().map_or(0, Vec::len);
3080 let structure = structure_from_cells(&grid.cells, nrows, ncols);
3081 (grid.rows, Some(grid.cells), Some(structure))
3082 }
3083 None => {
3084 let rows = reconstruct_table(region, &page.cells);
3085 let rows = if rows.iter().any(|r| r.len() > 1) {
3086 rows
3087 } else {
3088 vec![vec![text.clone()]]
3089 };
3090 (rows, None, None)
3091 }
3092 };
3093 // The paired caption (#265) rides on the table — docling's
3094 // TableItem.captions ref; Markdown prints it above the grid,
3095 // the JSON export emits the $ref, DocLang the <caption>.
3096 let caption = table_caption_for[i]
3097 .map(|ci| md_escape(®ion_texts[ci]))
3098 .filter(|t| !t.is_empty());
3099 // Rich cells (docling#3906): the covering cell's blocks are its
3100 // text followed by the nested picture(s). docling's Markdown
3101 // renders a `RichTableCell` through the serializer — the
3102 // group's children joined by blank lines, newlines flattened
3103 // to spaces — so the flat `rows` text becomes
3104 // `text <!-- image -->`; the first-class `cells` (the JSON
3105 // `table_cells` / `grid`) keep the plain text, as upstream.
3106 let mut cell_blocks: Option<Vec<Vec<Vec<Node>>>> = None;
3107 if let (Some(by_cell), Some(fc)) = (rich_cell_pictures.get(&i), cells.as_ref()) {
3108 let nrows = rows.len();
3109 let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
3110 let mut blocks = vec![vec![Vec::<Node>::new(); ncols]; nrows];
3111 for (cell_idx, pics) in by_cell {
3112 let cell = &fc[*cell_idx];
3113 let (r, c) = (cell.start_row, cell.start_col);
3114 if r >= nrows || c >= ncols {
3115 continue;
3116 }
3117 let mut parts: Vec<String> = Vec::new();
3118 let mut cell_nodes: Vec<Node> = Vec::new();
3119 if !cell.text.trim().is_empty() {
3120 parts.push(cell.text.clone());
3121 cell_nodes.push(Node::Paragraph {
3122 text: cell.text.clone(),
3123 });
3124 }
3125 for &p in pics {
3126 parts.push("<!-- image -->".to_string());
3127 let classification = match &enrichments[p] {
3128 Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
3129 _ => None,
3130 };
3131 #[cfg(feature = "ocr-prep")]
3132 let image = crop_region(page, ®ions[p], picture_scale);
3133 #[cfg(not(feature = "ocr-prep"))]
3134 let image: Option<PictureImage> = None;
3135 cell_nodes.push(located(
3136 norm_loc(®ions[p], page.width, page_h),
3137 Node::Picture {
3138 caption: None,
3139 caption_href: None,
3140 image,
3141 classification,
3142 caption_parent: Default::default(),
3143 },
3144 ));
3145 }
3146 let rendered = parts.join(" ");
3147 for row in rows.iter_mut().skip(r).take(cell.row_span) {
3148 for slot in row.iter_mut().skip(c).take(cell.col_span) {
3149 *slot = rendered.clone();
3150 }
3151 }
3152 blocks[r][c] = cell_nodes;
3153 }
3154 cell_blocks = Some(blocks);
3155 }
3156 nodes.push(located(
3157 loc,
3158 Node::Table(Table {
3159 rows,
3160 location: None,
3161 structure,
3162 cell_blocks,
3163 cells,
3164 caption,
3165 // As for pictures: the caption is the table's child.
3166 caption_parent: CaptionParent::Item,
3167 }),
3168 ));
3169 }
3170 // With formula enrichment the CodeFormula model decodes the region
3171 // to LaTeX; otherwise docling emits a placeholder comment rather
3172 // than the (garbled) raw glyph text.
3173 "formula" => match &enrichments[i] {
3174 Some(Enrichment::Formula { latex }) => nodes.push(Node::Formula {
3175 latex: latex.clone(),
3176 orig: text.clone(),
3177 location: Some(loc),
3178 }),
3179 _ => nodes.push(Node::Paragraph {
3180 text: "<!-- formula-not-decoded -->".into(),
3181 }),
3182 },
3183 // Code blocks: use the space-glyph-only grouping (monospace keeps its
3184 // source spacing) and emit a fenced block, preserving the line breaks
3185 // and indentation of the source (unlike prose, which reflows). pdfium
3186 // still inserts spaces around tight punctuation (`console .log`,
3187 // `add (3 , 5)`); tighten them to match docling-parse's source spacing.
3188 "code" => {
3189 // `code_region_text` preserves line breaks/indentation and tightens
3190 // each line itself; the fallback prose `text` is tightened here.
3191 let code = code_region_text(region, &page.code_cells);
3192 let code = if code.is_empty() {
3193 tighten_code_punct(&text)
3194 } else {
3195 code
3196 };
3197 // With code enrichment the CodeFormula model rewrites the block
3198 // (and names its language); `orig` keeps the raw extraction in
3199 // docling's shape — its parser has no line-preserving code
3200 // path, so its `orig` is the same code with the lines joined
3201 // by single spaces (indentation collapsed).
3202 // docling's parser has no line-preserving code path — its code
3203 // items carry the lines joined by single spaces. That flat
3204 // form is what every byte-conformance surface serializes
3205 // (legacy Markdown, JSON, DocLang); the line-preserving
3206 // extraction rides in `pretty` for strict Markdown only.
3207 let flat = code
3208 .lines()
3209 .map(str::trim)
3210 .filter(|l| !l.is_empty())
3211 .collect::<Vec<_>>()
3212 .join(" ");
3213 let node = match &enrichments[i] {
3214 Some(Enrichment::Code {
3215 language,
3216 text: enriched,
3217 }) => Node::Code {
3218 language: language.clone(),
3219 text: enriched.clone(),
3220 orig: Some(flat),
3221 pretty: None,
3222 },
3223 _ => Node::Code {
3224 language: None,
3225 text: flat,
3226 orig: None,
3227 pretty: Some(code),
3228 },
3229 };
3230 nodes.push(located(loc, node));
3231 // docling emits the `Listing N:` caption after the code block.
3232 if let Some(ci) = code_caption_for[i] {
3233 let cap = md_escape(®ion_texts[ci]);
3234 if !cap.is_empty() {
3235 nodes.push(Node::Paragraph { text: cap });
3236 }
3237 }
3238 }
3239 // text, caption, footnote → paragraph
3240 _ => {
3241 // docling parity (`PageAssembleModel._match_hyperlink`): when
3242 // link annotations cover ≥ half of the region's box, the
3243 // hyperlink attaches to the item and the legacy Markdown
3244 // serializer wraps its full text — 2206.01062's footnote URLs
3245 // render as `[1 https://…](https://…)`. Sparse in-paragraph
3246 // citation links stay below the 0.5 coverage threshold and
3247 // remain plain text, exactly like docling.
3248 //
3249 // Scope: **footnote regions only.** Upstream's page_assemble
3250 // matches every TEXT_ELEM label, but published docling
3251 // observably carries the hyperlink into the document only for
3252 // footnote items — in both committed groundtruth generations
3253 // (docling-JSON and Markdown, independent runs) the fully
3254 // covered plain-text DOI line of 2206.01062 page 1 has
3255 // `hyperlink: None` while the equally covered footnotes carry
3256 // theirs. The corpus is the conformance reference, so match
3257 // the observed behavior; widen the label set if a future
3258 // groundtruth refresh starts linking plain text too.
3259 let escaped = md_escape(&text);
3260 let hyperlink = (region.label == "footnote")
3261 .then(|| region_hyperlink(region, &page.links))
3262 .flatten();
3263 let text = match hyperlink {
3264 Some(uri) => {
3265 // The strict-mode anchor pairs this item covers are
3266 // superseded by the baked whole-item link.
3267 links.retain(|(anchor, href)| {
3268 !(href == &uri && region_texts[i].contains(anchor.as_str()))
3269 });
3270 format!("[{escaped}]({uri})")
3271 }
3272 None => escaped,
3273 };
3274 nodes.push(located(loc, Node::Paragraph { text }))
3275 }
3276 }
3277 }
3278 // A `/Rotate`-normalized scanned page (see `pdfium_backend`) was assembled
3279 // in upright space; rotate the finished geometry back so locations and the
3280 // page size are display-space, like docling and every viewer report them.
3281 if page.rotation != 0 {
3282 rotate_nodes_to_display(&mut nodes, page.rotation);
3283 }
3284 (nodes, links)
3285}
3286
3287/// One child of a picture as docling's `ReadingOrderModel._add_child_elements`
3288/// writes it under the `PictureItem`: a heading for a `section_header` /
3289/// `title` (upstream remaps title to section header), a list item for a
3290/// `list_item`, a furniture-layer text for a page header/footer, a caption,
3291/// otherwise a text item. `None` for a child that claimed no text.
3292fn picture_child_node(region: &Region, text: &str, loc: [u16; 4]) -> Option<Node> {
3293 if text.is_empty() {
3294 return None;
3295 }
3296 Some(match region.label {
3297 "title" | "section_header" => located(
3298 loc,
3299 Node::Heading {
3300 level: 2,
3301 text: md_escape(text),
3302 },
3303 ),
3304 // docling-core's `add_list_item` under a non-list parent opens a
3305 // list group per item, so every child item starts its own list;
3306 // `_add_child_elements` runs the marker processor on it too.
3307 "list_item" => list_item_node(text, loc, true),
3308 "page_header" | "page_footer" => Node::PageFurniture {
3309 footer: region.label == "page_footer",
3310 location: loc,
3311 text: md_escape(text),
3312 },
3313 "caption" => located(
3314 loc,
3315 Node::Caption {
3316 text: md_escape(text),
3317 href: None,
3318 },
3319 ),
3320 _ => located(
3321 loc,
3322 Node::Paragraph {
3323 text: md_escape(text),
3324 },
3325 ),
3326 })
3327}
3328
3329/// Rotate one 0–511 location bbox 90° clockwise on the grid (top-left origin):
3330/// `(x, y) → (511 - y, x)`.
3331fn rot_loc_cw(l: [u16; 4]) -> [u16; 4] {
3332 [511 - l[3], l[0], 511 - l[1], l[2]]
3333}
3334
3335/// Map upright-space geometry back to display space for a page whose `/Rotate`
3336/// was normalized away before inference: every `<location>` rotates `rot`°
3337/// clockwise on the 0–511 grid (the grid is per-axis normalized, so no page
3338/// dims are needed), and the `PageInfo` size returns to the display box. Node
3339/// text and order are untouched — reading order was decided upright, which is
3340/// the whole point.
3341fn rotate_nodes_to_display(nodes: &mut [Node], rot: u16) {
3342 let quarter_turns = (rot / 90) as usize;
3343 let rot_loc = |l: &mut [u16; 4]| {
3344 for _ in 0..quarter_turns {
3345 *l = rot_loc_cw(*l);
3346 }
3347 };
3348 fn walk(node: &mut Node, rot_loc: &impl Fn(&mut [u16; 4]), swap_dims: bool) {
3349 match node {
3350 Node::PageInfo { width, height, .. } => {
3351 if swap_dims {
3352 std::mem::swap(width, height);
3353 }
3354 }
3355 Node::Located { location, inner } => {
3356 rot_loc(location);
3357 walk(inner, rot_loc, swap_dims);
3358 }
3359 Node::Furniture { inner, .. } => walk(inner, rot_loc, swap_dims),
3360 Node::Group { children, .. } | Node::PictureChildren(children) => {
3361 for c in children {
3362 walk(c, rot_loc, swap_dims);
3363 }
3364 }
3365 Node::ListItem { location, .. }
3366 | Node::Formula { location, .. }
3367 | Node::Chart { location, .. } => {
3368 if let Some(l) = location {
3369 rot_loc(l);
3370 }
3371 }
3372 Node::PageFurniture { location, .. } => rot_loc(location),
3373 Node::Table(t) => {
3374 if let Some(l) = &mut t.location {
3375 rot_loc(l);
3376 }
3377 }
3378 _ => {}
3379 }
3380 }
3381 let swap_dims = quarter_turns % 2 == 1;
3382 for node in nodes {
3383 walk(node, &rot_loc, swap_dims);
3384 }
3385}
3386
3387/// Merge paragraph fragments split across a column or page break. docling joins a
3388/// paragraph whose previous fragment ends mid-sentence (a letter, not sentence
3389/// punctuation) with a lowercase continuation: `…definition of` + `lists in…` →
3390/// `…definition of lists in…`. The fragments are consecutive paragraphs, or
3391/// separated only by figure(s) the text wraps around: a column whose body flows
3392/// past a figure resumes below it (`…The wing type that is` ⟶[figure]⟶ `the most
3393/// common…`), and docling emits the whole paragraph before the figure. A heading,
3394/// table, or list between them ends the paragraph (no merge).
3395/// A paragraph that is really a figure/table caption (`Fig. 1. …`, `Table 2 …`).
3396/// Used to skip an unpaired caption when stitching a paragraph that wraps around
3397/// a figure.
3398fn looks_like_caption(text: &str) -> bool {
3399 let head: String = text.trim_start().chars().take(14).collect();
3400 (head.starts_with("Fig") || head.starts_with("Table"))
3401 && head.contains(|c: char| c.is_ascii_digit())
3402}
3403
3404/// A paragraph fragment is "open" — i.e. it might continue into the next
3405/// paragraph — when it ends mid-word (a letter) or with a wrap hyphen/dash.
3406/// docling joins `vocab-` + `ulary` → `vocab- ulary`.
3407fn paragraph_is_open(text: &str) -> bool {
3408 // docling's merge head test (`.+([a-z,\-\u00AD])\s*`): at least two chars,
3409 // ending in an ASCII lowercase letter, a comma, a hyphen, or a soft
3410 // hyphen. The comma matters: 2206's "…In phase four," resumes across the
3411 // page break. Uppercase/non-Latin endings do not merge, exactly as
3412 // upstream (the dash family is already `-` here — clean_text normalized).
3413 let t = text.trim_end();
3414 t.chars().count() >= 2
3415 && t.chars()
3416 .next_back()
3417 .is_some_and(|c| matches!(c, 'a'..='z' | ',' | '-' | '\u{ad}'))
3418}
3419
3420/// The paragraph text inside a node, looking through a [`Node::Located`]
3421/// provenance wrapper (PDF body paragraphs are wrapped since they carry a
3422/// `<location>`). Returns `None` for non-paragraph nodes.
3423fn as_paragraph(n: &Node) -> Option<&str> {
3424 match n {
3425 Node::Paragraph { text } => Some(text),
3426 Node::Located { inner, .. } => match inner.as_ref() {
3427 Node::Paragraph { text } => Some(text),
3428 _ => None,
3429 },
3430 _ => None,
3431 }
3432}
3433
3434/// Whether a node is a picture, looking through a [`Node::Located`] wrapper.
3435fn is_picture_node(n: &Node) -> bool {
3436 match n {
3437 Node::Picture { .. } => true,
3438 Node::Located { inner, .. } => matches!(inner.as_ref(), Node::Picture { .. }),
3439 _ => false,
3440 }
3441}
3442
3443/// A node a forward paragraph merge looks straight past: a figure or *table*
3444/// the text wraps around, or a page header/footer that falls between the two
3445/// fragments of a paragraph continuing across a page break (docling's merge
3446/// skip-labels: page_header, page_footer, table, picture, caption, footnote —
3447/// 2206's "…In phase four," resumes after a full caption+table+figure block).
3448fn is_merge_trailer(n: &Node) -> bool {
3449 is_picture_node(n)
3450 || matches!(
3451 n,
3452 Node::PageFurniture { .. }
3453 | Node::PageInfo { .. }
3454 | Node::Table(_)
3455 | Node::PictureChildren(_)
3456 )
3457 || matches!(n, Node::Located { inner, .. } if matches!(inner.as_ref(), Node::Table(_)))
3458 || as_paragraph(n).is_some_and(looks_like_caption)
3459}
3460
3461/// Rebuild node `i` as a paragraph with `text`, preserving its `<location>`
3462/// wrapper (and thus provenance) if it had one.
3463fn reparagraph(node: &Node, text: String) -> Node {
3464 match node {
3465 Node::Located { location, .. } => located(*location, Node::Paragraph { text }),
3466 _ => Node::Paragraph { text },
3467 }
3468}
3469
3470pub(crate) fn merge_continuations(nodes: &mut Vec<Node>) {
3471 let mut i = 0;
3472 while i + 1 < nodes.len() {
3473 let Some(a) = as_paragraph(&nodes[i]) else {
3474 i += 1;
3475 continue;
3476 };
3477 // A figure/table caption is a self-contained unit; body text resuming
3478 // after a figure is the continuation case, not the caption itself. Never
3479 // stitch *from* a caption — otherwise a caption that ends in a lone glyph
3480 // (`Fig. 5. … PubTabNet. μ`) would swallow a following stray figure label
3481 // (a standalone `μ`) into `… μ μ`.
3482 if looks_like_caption(a) {
3483 i += 1;
3484 continue;
3485 }
3486 if !paragraph_is_open(a) {
3487 i += 1;
3488 continue;
3489 }
3490 // The continuation is the next paragraph, looking past any figures the
3491 // text wraps around — and a figure/table caption that was emitted as its
3492 // own paragraph (an above-the-figure caption that didn't pair), since the
3493 // body text resumes after the whole figure+caption block.
3494 let mut j = i + 1;
3495 while nodes.get(j).is_some_and(is_merge_trailer) {
3496 j += 1;
3497 }
3498 // docling's continuation regex allows either case, but its merge runs
3499 // over the pre-assembly element stream; at node level an uppercase
3500 // start is overwhelmingly a new sentence/heading fragment (allowing it
3501 // swallowed 2305's formula blocks and redp's chapter openers), so the
3502 // continuation stays lowercase-start here.
3503 let cont = nodes.get(j).and_then(as_paragraph).is_some_and(|b| {
3504 b.trim_start()
3505 .chars()
3506 .next()
3507 .is_some_and(char::is_lowercase)
3508 });
3509 if cont {
3510 let a = as_paragraph(&nodes[i]).unwrap().trim_end().to_string();
3511 let b = as_paragraph(&nodes[j]).unwrap().trim_start().to_string();
3512 // A soft hyphen -- or a hard hyphen followed by a lowercase
3513 // continuation (guaranteed lowercase by the `cont` gate above) --
3514 // is a word split across the break: strip it and join without a
3515 // space, docling#3888 ("vocab-" + "ulary" -> "vocabulary");
3516 // docling's older serializer kept the artifact ("vocab- ulary").
3517 // Everything else joins with the space, as before.
3518 let merged = match a.strip_suffix('\u{ad}').or_else(|| a.strip_suffix('-')) {
3519 Some(stem) => format!("{stem}{b}"),
3520 None => format!("{a} {b}"),
3521 };
3522 // Keep node i's provenance wrapper; docling's merged paragraph keeps
3523 // the first fragment's geometry as its primary location.
3524 nodes[i] = reparagraph(&nodes[i], merged);
3525 nodes.remove(j);
3526 // Re-check i: the merged paragraph may continue further.
3527 } else {
3528 i += 1;
3529 }
3530 }
3531}
3532
3533/// How many leading nodes of `nodes` are safe to flush now — i.e. cannot be
3534/// rewritten by a future [`merge_continuations`] once more pages are appended.
3535///
3536/// A forward merge can only start from an "open" paragraph (ends mid-word) and
3537/// only reaches across trailing pictures and figure/table captions. So we scan
3538/// from the end past those skippable trailers: if the first non-skippable node is
3539/// an open paragraph, it (and the trailers after it) must be held; anything else —
3540/// a closed paragraph, a heading, a table, a list — blocks any forward merge, so
3541/// the whole buffer is safe to flush.
3542fn hold_start(nodes: &[Node]) -> usize {
3543 for k in (0..nodes.len()).rev() {
3544 // Skippable trailers (figures, page furniture, captions): a forward merge
3545 // looks straight past them.
3546 if is_merge_trailer(&nodes[k]) {
3547 continue;
3548 }
3549 match as_paragraph(&nodes[k]) {
3550 // An open body paragraph might still pull a continuation off the next
3551 // page — hold from here to the end.
3552 Some(text) if paragraph_is_open(text) => return k,
3553 // A closed paragraph, heading, table, list, etc. ends the paragraph:
3554 // nothing after it can merge backwards across it. Flush everything.
3555 _ => return nodes.len(),
3556 }
3557 }
3558 // Only skippable trailers (or empty) and no open paragraph to anchor a merge.
3559 nodes.len()
3560}
3561
3562/// Streaming counterpart of [`merge_continuations`]: feed per-page node batches in
3563/// document order and get back the prefix that is final (its cross-page merges are
3564/// resolved and no future page can change it), holding back only the small tail
3565/// that might still merge into the next page. Concatenating every flushed batch
3566/// (then [`finish`](Self::finish)) yields exactly the same nodes as running
3567/// [`merge_continuations`] once over the whole document.
3568pub(crate) struct StreamAssembler {
3569 pending: Vec<Node>,
3570}
3571
3572impl StreamAssembler {
3573 pub(crate) fn new() -> Self {
3574 Self {
3575 pending: Vec::new(),
3576 }
3577 }
3578
3579 /// Append one page's nodes, resolve merges within the buffer, and return the
3580 /// now-final prefix to emit (possibly empty).
3581 pub(crate) fn push(&mut self, mut nodes: Vec<Node>) -> Vec<Node> {
3582 self.pending.append(&mut nodes);
3583 merge_continuations(&mut self.pending);
3584 let cut = hold_start(&self.pending);
3585 let tail = self.pending.split_off(cut);
3586 std::mem::replace(&mut self.pending, tail)
3587 }
3588
3589 /// Flush whatever is left after the last page (the held tail is final once no
3590 /// more pages can follow).
3591 pub(crate) fn finish(self) -> Vec<Node> {
3592 self.pending
3593 }
3594}
3595
3596#[cfg(test)]
3597mod tests {
3598 use super::{cells_text, clean_text, merge_overlapping_regulars, reclaim_heading_body_footers};
3599
3600 /// docling drops a picture covering > 90 % of the page (its labels then
3601 /// read out as text); a dominant-but-not-full figure and any other label
3602 /// stay whatever their size.
3603 #[test]
3604 fn full_page_pictures_are_dropped_like_docling() {
3605 use super::drop_full_page_pictures;
3606 use crate::layout::Region;
3607 let region = |label: &'static str, l, t, r, b| Region {
3608 label,
3609 score: 0.99,
3610 l,
3611 t,
3612 r,
3613 b,
3614 };
3615 let mut regions = vec![
3616 region("picture", 0.0, 0.5, 478.9, 241.8),
3617 region("picture", 10.0, 10.0, 400.0, 200.0),
3618 region("table", 0.0, 0.0, 480.0, 243.0),
3619 region("text", 5.0, 5.0, 100.0, 20.0),
3620 ];
3621 drop_full_page_pictures(&mut regions, 480.75, 243.75);
3622 let labels: Vec<_> = regions.iter().map(|r| (r.label, r.l)).collect();
3623 assert_eq!(
3624 labels,
3625 vec![("picture", 10.0), ("table", 0.0), ("text", 5.0)]
3626 );
3627 }
3628 use super::{code_region_text, merge_continuations, resolve_link_anchors, StreamAssembler};
3629 use crate::layout::Region;
3630 use crate::pdfium_backend::{LinkAnnot, PdfPage, TextCell};
3631 use docling_core::Node;
3632
3633 /// The int8-layout guard's coverage metric: cells under detections count,
3634 /// cells outside don't, whitespace cells are ignored, and a cell-less page
3635 /// reads as fully covered (nothing to rescue).
3636 #[test]
3637 fn layout_cell_coverage_counts_claimed_text_cells() {
3638 let cell = |text: &str, l: f32, t: f32| TextCell {
3639 text: text.into(),
3640 l,
3641 t,
3642 r: l + 40.0,
3643 b: t + 10.0,
3644 };
3645 let region = Region {
3646 label: "text",
3647 score: 0.9,
3648 l: 0.0,
3649 t: 0.0,
3650 r: 100.0,
3651 b: 50.0,
3652 };
3653 let cells = vec![
3654 cell("inside", 10.0, 10.0),
3655 cell("also inside", 10.0, 30.0),
3656 cell("outside", 10.0, 200.0),
3657 cell(" ", 10.0, 210.0), // whitespace: not counted at all
3658 ];
3659 let cov = super::layout_cell_coverage(std::slice::from_ref(®ion), &cells);
3660 assert!((cov - 2.0 / 3.0).abs() < 1e-6, "got {cov}");
3661 assert_eq!(super::layout_cell_coverage(&[], &[]), 1.0);
3662 assert_eq!(super::layout_cell_coverage(&[], &cells), 0.0);
3663 }
3664
3665 /// #165: a picture no longer claims cells at 0.2 intersection-over-self.
3666 /// A line straddling the figure border (≤80 % contained) becomes an orphan
3667 /// region and is emitted as page text — before the fix its cells were
3668 /// silently erased. A line fully inside the picture is the picture's child
3669 /// (docling's `_set_cluster_children`): it survives the containment drop,
3670 /// leaves the page's reading order, and is written only under the picture
3671 /// in the JSON — never in the Markdown, like docling's picture serializer.
3672 #[test]
3673 fn border_straddlers_are_page_text_picture_interior_is_a_picture_child() {
3674 let pic = Region {
3675 label: "picture",
3676 score: 0.9,
3677 l: 0.0,
3678 t: 0.0,
3679 r: 100.0,
3680 b: 100.0,
3681 };
3682 // ~35 % of this cell overlaps the picture (l=90..120 of 0..100): above
3683 // the old 0.2 claim (was swallowed), below full containment (survives).
3684 let straddler = TextCell {
3685 text: "axis label".into(),
3686 l: 90.0,
3687 t: 40.0,
3688 r: 120.0,
3689 b: 48.0,
3690 };
3691 let interior = TextCell {
3692 text: "in-figure callout".into(),
3693 l: 10.0,
3694 t: 10.0,
3695 r: 60.0,
3696 b: 18.0,
3697 };
3698 let cells = vec![straddler, interior];
3699 let mut regions = vec![pic];
3700 super::add_orphan_regions(&mut regions, &cells);
3701 super::drop_contained_regulars(&mut regions);
3702 assert_eq!(
3703 regions.iter().filter(|r| r.label == "text").count(),
3704 2,
3705 "both unclaimed lines become orphans, and a picture swallows neither"
3706 );
3707 let parents = super::picture_parents(®ions);
3708 let parent_of = |l: f32| {
3709 regions
3710 .iter()
3711 .zip(&parents)
3712 .find(|(r, _)| r.label == "text" && r.l == l)
3713 .and_then(|(_, p)| *p)
3714 };
3715 assert_eq!(
3716 parent_of(10.0),
3717 Some(0),
3718 "the callout is the picture's child"
3719 );
3720 assert_eq!(parent_of(90.0), None, "the straddler is a page element");
3721
3722 let page = PdfPage::from_cells(200.0, 200.0, 2.0, cells);
3723 let n = regions.len();
3724 let (nodes, _) = super::assemble_page(&page, regions, &vec![None; n], &vec![None; n], None);
3725 let children: Vec<&Node> = nodes
3726 .iter()
3727 .filter_map(|n| match n {
3728 Node::PictureChildren(c) => Some(c),
3729 _ => None,
3730 })
3731 .flatten()
3732 .collect();
3733 assert!(
3734 matches!(children.as_slice(), [Node::Located { inner, .. }]
3735 if matches!(inner.as_ref(), Node::Paragraph { text } if text == "in-figure callout")),
3736 "{children:?}"
3737 );
3738 let mut doc = docling_core::DoclingDocument::new("t");
3739 doc.nodes = nodes;
3740 let md = doc.export_to_markdown();
3741 assert!(md.contains("axis label"), "{md}");
3742 assert!(!md.contains("in-figure callout"), "{md}");
3743 let json = doc.export_to_json_value();
3744 let pic = &json["pictures"][0];
3745 let child = pic["children"][0]["$ref"].as_str().expect("a child ref");
3746 let idx: usize = child.rsplit('/').next().unwrap().parse().unwrap();
3747 assert_eq!(json["texts"][idx]["text"], "in-figure callout");
3748 assert_eq!(json["texts"][idx]["parent"]["$ref"], "#/pictures/0");
3749 assert_eq!(json["texts"][idx]["content_layer"], "body");
3750 }
3751
3752 /// docling#3906's concern, pinned on our side: a picture detected fully
3753 /// inside a table region must survive the containment drop (upstream now
3754 /// attaches it to the table's cell; we keep it as a body sibling — either
3755 /// way it must not vanish). The text region inside the same table is the
3756 /// control: regulars are the ones the drop swallows.
3757 #[test]
3758 fn picture_inside_a_table_region_survives_the_containment_drop() {
3759 let mut regions = vec![
3760 region("table", 0.9, 0.0, 0.0, 200.0, 200.0),
3761 region("picture", 0.9, 20.0, 20.0, 120.0, 120.0),
3762 region("text", 0.9, 20.0, 140.0, 180.0, 180.0),
3763 ];
3764 super::drop_contained_regulars(&mut regions);
3765 let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
3766 assert_eq!(
3767 labels,
3768 ["table", "picture"],
3769 "the in-table picture stays; the in-table regular is the special's child"
3770 );
3771 }
3772
3773 /// Table–caption pairing (#265) is reading-order adjacency, docling's
3774 /// `_find_to_captions`: a caption binds the table directly next to it in
3775 /// the region sequence — above-caption and below-caption both work, and
3776 /// geometry is irrelevant (a same-page caption in the other column of a
3777 /// two-column layout is *not* adjacent, however close its box is). A
3778 /// caption with media on both sides, or separated from the table by a
3779 /// text paragraph, stays unattached.
3780 #[test]
3781 fn table_captions_pair_by_reading_order_adjacency() {
3782 // caption → table (above-caption), then table → caption (below-caption),
3783 // then a caption fenced off by a paragraph, then one between two tables.
3784 let regions = vec![
3785 region("text", 0.9, 0.0, 0.0, 100.0, 10.0), // 0 body text
3786 region("caption", 0.9, 0.0, 12.0, 60.0, 20.0), // 1 above-caption
3787 region("table", 0.9, 20.0, 22.0, 90.0, 60.0), // 2 ← pairs with 1
3788 region("table", 0.9, 0.0, 70.0, 100.0, 110.0), // 3 ← pairs with 4
3789 region("caption", 0.9, 0.0, 112.0, 60.0, 120.0), // 4 below-caption
3790 region("text", 0.9, 0.0, 130.0, 100.0, 140.0), // 5 body text
3791 region("caption", 0.9, 0.0, 142.0, 60.0, 150.0), // 6 fenced by 5/7
3792 region("text", 0.9, 0.0, 152.0, 100.0, 162.0), // 7 body text
3793 region("table", 0.9, 0.0, 170.0, 100.0, 200.0), // 8 unpaired
3794 region("caption", 0.9, 0.0, 202.0, 60.0, 210.0), // 9 ambiguous
3795 region("table", 0.9, 0.0, 212.0, 100.0, 240.0), // 10 unpaired
3796 ];
3797 let mut taken = vec![false; regions.len()];
3798 let pairs = super::pair_table_captions(®ions, &mut taken);
3799 assert_eq!(pairs[2], Some(1), "caption directly above its table pairs");
3800 assert_eq!(pairs[3], Some(4), "caption directly below its table pairs");
3801 assert_eq!(
3802 pairs[8], None,
3803 "a text paragraph between caption and table breaks the bond"
3804 );
3805 assert_eq!(
3806 pairs[10], None,
3807 "a caption between two tables is ambiguous and stays loose"
3808 );
3809 assert!(taken[1] && taken[4] && !taken[6] && !taken[9]);
3810 }
3811
3812 /// A colored terms-and-conditions panel detected as `picture` demotes into
3813 /// per-paragraph `text` regions (the blank line between C.7 and C.8 splits
3814 /// them); a chart whose only text is a few narrow axis labels keeps its
3815 /// crop untouched.
3816 #[test]
3817 fn text_panels_demote_to_paragraphs_but_charts_keep_their_crop() {
3818 let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3819 text: text.to_string(),
3820 l,
3821 t,
3822 r,
3823 b,
3824 };
3825 let panel = Region {
3826 label: "picture",
3827 score: 0.9,
3828 l: 0.0,
3829 t: 0.0,
3830 r: 100.0,
3831 b: 100.0,
3832 };
3833 // Three tight lines, a blank-line gap, two more: two paragraphs.
3834 let cells = vec![
3835 cell(
3836 "C.7. Wenn Sie diesen Vertrag widerrufen,",
3837 5.0,
3838 10.0,
3839 95.0,
3840 18.0,
3841 ),
3842 cell(
3843 "haben wir Ihnen alle Zahlungen, die wir",
3844 5.0,
3845 20.0,
3846 95.0,
3847 28.0,
3848 ),
3849 cell(
3850 "von Ihnen erhalten haben, zurückzuzahlen.",
3851 5.0,
3852 30.0,
3853 90.0,
3854 38.0,
3855 ),
3856 cell(
3857 "C.8. Wir können die Rückzahlung verweigern,",
3858 5.0,
3859 52.0,
3860 95.0,
3861 60.0,
3862 ),
3863 cell(
3864 "bis wir die Waren wieder zurückerhalten haben.",
3865 5.0,
3866 62.0,
3867 92.0,
3868 70.0,
3869 ),
3870 ];
3871 let mut regions = vec![panel.clone()];
3872 super::recover_text_panels(&mut regions, &cells);
3873 assert_eq!(
3874 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3875 ["text", "text"],
3876 "dense panel must demote into one text region per paragraph"
3877 );
3878 assert!(regions[0].b < regions[1].t, "paragraphs split at the gap");
3879 // Sparse narrow labels (a chart): picture survives.
3880 let labels = vec![
3881 cell("0", 5.0, 90.0, 8.0, 95.0),
3882 cell("50", 5.0, 50.0, 10.0, 55.0),
3883 cell("100", 5.0, 10.0, 12.0, 15.0),
3884 cell("t, s", 45.0, 96.0, 55.0, 100.0),
3885 ];
3886 let mut regions = vec![panel];
3887 super::recover_text_panels(&mut regions, &labels);
3888 assert_eq!(
3889 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3890 ["picture"]
3891 );
3892 }
3893
3894 /// An uncaptioned chart on a scanned page whose title, axis labels, and
3895 /// OCR boxes over the plot area are dense and wide enough to pass the
3896 /// coverage/width gates still keeps its crop: its line heights are ragged
3897 /// (title face vs tick labels vs bar-area OCR), failing the uniform-leading
3898 /// gate — a real text panel is set with constant leading (#173).
3899 #[test]
3900 fn dense_titled_chart_keeps_its_crop() {
3901 let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3902 text: text.to_string(),
3903 l,
3904 t,
3905 r,
3906 b,
3907 };
3908 let chart = Region {
3909 label: "picture",
3910 score: 0.9,
3911 l: 0.0,
3912 t: 0.0,
3913 r: 100.0,
3914 b: 100.0,
3915 };
3916 // Five wide lines at wildly different heights: a 12-pt title, 20-pt OCR
3917 // boxes over the bars, 4–5-pt tick/axis labels. Coverage and median
3918 // width both clear the panel thresholds.
3919 let cells = vec![
3920 cell("Underground Water Storage", 10.0, 5.0, 90.0, 17.0),
3921 cell("aquifer recharge zone", 15.0, 30.0, 75.0, 50.0),
3922 cell("confined | unconfined | perched", 12.0, 55.0, 80.0, 59.0),
3923 cell("saturated thickness", 8.0, 70.0, 60.0, 90.0),
3924 cell("distance from well, km", 20.0, 92.0, 85.0, 97.0),
3925 ];
3926 let mut regions = vec![chart];
3927 super::recover_text_panels(&mut regions, &cells);
3928 assert_eq!(
3929 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
3930 ["picture"],
3931 "ragged line heights mark a figure, not a text panel"
3932 );
3933 }
3934
3935 /// docling serializes a cluster's cells in docling-parse index order
3936 /// (`_sort_cells`) and joins them with `PageAssembleModel.sanitize_text`:
3937 /// a space after every line except one ending in `-`, which either fuses a
3938 /// wrapped word (alnum on both sides — dash dropped) or glues verbatim (a
3939 /// bare `-` cell: `[0000` `-` `0002` → `[0000 -0002`, the 2305 ORCID line;
3940 /// `-` + `"C" cell -` + `a new table cell` → `-"C" cell a new table cell`,
3941 /// its OTSL list). Verified against the corpus: pure index order beats any
3942 /// geometric re-sort (normal_4pages' heading numerals paint after their
3943 /// text and belong last: `## 들어가며 1`).
3944 #[test]
3945 fn cells_join_in_index_order_with_sanitize_text_rules() {
3946 let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
3947 text: text.to_string(),
3948 l,
3949 t,
3950 r,
3951 b,
3952 };
3953 let region = Region {
3954 label: "text",
3955 score: 1.0,
3956 l: 0.0,
3957 t: 95.0,
3958 r: 200.0,
3959 b: 130.0,
3960 };
3961 // ORCID superscript: a bare dash cell is a *detached* dash — kept, and
3962 // since docling#4052 (2.122) it joins with the ordinary space on both
3963 // sides (`[0000 -0002 -6960]` before that fix).
3964 let orcid = vec![
3965 cell("[0000", 10.0, 100.0, 30.0, 110.0),
3966 cell("−", 30.0, 100.0, 34.0, 110.0),
3967 cell("0002", 34.0, 100.0, 50.0, 110.0),
3968 cell("−", 50.0, 100.0, 54.0, 110.0),
3969 cell("6960]", 54.0, 100.0, 70.0, 110.0),
3970 ];
3971 assert_eq!(super::region_text(®ion, &orcid), "[0000 - 0002 - 6960]");
3972 // Wrapped word: dash dropped, lines fused (both boundary words alnum).
3973 let wrapped = vec![
3974 cell("platforms-", 10.0, 100.0, 60.0, 110.0),
3975 cell("reflects the design", 10.0, 112.0, 90.0, 122.0),
3976 ];
3977 assert_eq!(
3978 super::region_text(®ion, &wrapped),
3979 "platformsreflects the design"
3980 );
3981 // Dash-ending lines that are *detached* dashes (a bare bullet cell, a
3982 // `cell -` separator): the dash stays and the lines join with a space
3983 // — docling#4052; before it they glued (`-"C" cell a new table cell`,
3984 // 2305's OTSL list bullets).
3985 let otsl = vec![
3986 cell("–", 10.0, 100.0, 14.0, 110.0),
3987 cell("\"C\" cell -", 16.0, 100.0, 60.0, 110.0),
3988 cell("a new table cell", 10.0, 112.0, 80.0, 122.0),
3989 ];
3990 assert_eq!(
3991 super::region_text(®ion, &otsl),
3992 "- \"C\" cell - a new table cell"
3993 );
3994 // Index order is authoritative — no geometric re-sort.
3995 let numeral = vec![
3996 cell("들어가며", 30.0, 100.0, 80.0, 110.0),
3997 cell("1", 10.0, 98.0, 25.0, 112.0), // big numeral painted last
3998 ];
3999 assert_eq!(super::region_text(®ion, &numeral), "들어가며 1");
4000 }
4001
4002 /// The geometric-reliability gate, on the two shapes it has to tell apart.
4003 #[test]
4004 fn geometric_reliability_rejects_split_column_grids() {
4005 let g = |rows: &[&[&str]]| -> Vec<Vec<String>> {
4006 rows.iter()
4007 .map(|r| r.iter().map(|c| c.to_string()).collect())
4008 .collect()
4009 };
4010 // A genuine grid: dense, every column carrying entries. Nothing for
4011 // TableFormer to improve, so geometry is used as-is.
4012 assert!(super::geometric_table_is_reliable(&g(&[
4013 &["Datum", "Leistung", "Anzahl", "Kosten"],
4014 &["04.07", "Internet", "1", "40.30"],
4015 &["04.07", "Telefon", "2", "8.06"],
4016 ])));
4017 // The left-edge split artefact (the shape a scanned invoice produced):
4018 // one real label column plus values scattered across three sparse ones.
4019 assert!(!super::geometric_table_is_reliable(&g(&[
4020 &["www.magenta.at/faq", "", "", ""],
4021 &["Serviceteam", "", "", ""],
4022 &["Telefon", "0676/2000", "", ""],
4023 &["Kundennummer", "", "", "1.21699482"],
4024 &["Rechnungsnummer", "", "922769430725", ""],
4025 &["Rechnungsdatum", "", "", "04.07.2025"],
4026 ])));
4027 // A column only one row ever uses is a split artefact even when the
4028 // grid is otherwise dense.
4029 assert!(!super::geometric_table_is_reliable(&g(&[
4030 &["a", "b", ""],
4031 &["c", "d", ""],
4032 &["e", "f", "g"],
4033 ])));
4034 // Degenerate shapes are never vouched for — TableFormer may recover
4035 // structure a collapsed reconstruction lost.
4036 assert!(!super::geometric_table_is_reliable(&g(&[&[
4037 "only one column"
4038 ]])));
4039 assert!(!super::geometric_table_is_reliable(&[]));
4040 }
4041
4042 /// A `picture` region is cropped out of the rendered page, whatever built
4043 /// that page. The browser pipeline (#157) has no pdfium but does hand over
4044 /// the rasterized bitmap through `from_cells_with_image`, so it must get
4045 /// the same figure bytes the native path does — that is what makes
4046 /// `images = "embedded"` inline real pixels instead of a placeholder.
4047 #[cfg(feature = "ocr-prep")]
4048 #[test]
4049 fn picture_regions_are_cropped_from_a_host_supplied_page_image() {
4050 let mut img = image::RgbImage::new(200, 200);
4051 // Paint the figure area so the crop is distinguishable from the page.
4052 for y in 100..160 {
4053 for x in 20..120 {
4054 img.put_pixel(x, y, image::Rgb([255, 0, 0]));
4055 }
4056 }
4057 // scale 2.0: the region is in page points, the bitmap in pixels.
4058 let page = PdfPage::from_cells_with_image(100.0, 100.0, 2.0, Vec::new(), img);
4059 let region = Region {
4060 label: "picture",
4061 score: 0.9,
4062 l: 10.0,
4063 t: 50.0,
4064 r: 60.0,
4065 b: 80.0,
4066 };
4067 let (nodes, _) = super::assemble_page(&page, vec![region], &[None], &[None], None);
4068 // Layout-derived nodes carry provenance, so the picture arrives wrapped.
4069 let image = nodes
4070 .iter()
4071 .find_map(|n| match n {
4072 Node::Located { inner, .. } => match &**inner {
4073 Node::Picture { image, .. } => image.as_ref(),
4074 _ => None,
4075 },
4076 Node::Picture { image, .. } => image.as_ref(),
4077 _ => None,
4078 })
4079 .expect("a picture node with cropped pixels");
4080 assert_eq!(image.mimetype, "image/png");
4081 assert_eq!((image.width, image.height), (100, 60), "region × scale");
4082 assert!(!image.data.is_empty(), "PNG bytes were encoded");
4083 }
4084
4085 #[test]
4086 fn link_anchors_split_a_shared_word_cell_between_adjacent_links() {
4087 // A common header layout: one text run holds several pipe-separated
4088 // labels, each carrying its own link annotation. Every link must get
4089 // its own label as the anchor (and the "|" separators must belong to
4090 // none), not the whole run.
4091 let annot = |l: f32, r: f32, uri: &str| LinkAnnot {
4092 l,
4093 t: 100.0,
4094 r,
4095 b: 114.0,
4096 uri: uri.into(),
4097 };
4098 let page = PdfPage {
4099 width: 600.0,
4100 height: 800.0,
4101 scale: 2.0,
4102 cells: Vec::new(),
4103 code_cells: Vec::new(),
4104 // "LinkedIn | GitHub | Credly" = 26 chars over x 100..360.
4105 word_cells: vec![cell(
4106 "LinkedIn | GitHub | Credly",
4107 100.0,
4108 100.0,
4109 360.0,
4110 114.0,
4111 )],
4112 image: image::RgbImage::new(1, 1),
4113 image_layout: None,
4114 links: vec![
4115 annot(100.0, 180.0, "https://l"),
4116 annot(200.0, 260.0, "https://g"),
4117 annot(290.0, 360.0, "https://c"),
4118 ],
4119 rotation: 0,
4120 };
4121 assert_eq!(
4122 resolve_link_anchors(&page),
4123 vec![
4124 ("LinkedIn".to_string(), "https://l".to_string()),
4125 ("GitHub".to_string(), "https://g".to_string()),
4126 ("Credly".to_string(), "https://c".to_string()),
4127 ]
4128 );
4129 }
4130
4131 /// A one-line code cell at `[l, r] × [t, b]` (top-left coords).
4132 fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
4133 TextCell {
4134 text: text.into(),
4135 l,
4136 t,
4137 r,
4138 b,
4139 }
4140 }
4141
4142 /// OCR-path grouping (docling's `_remove_overlapping_clusters("regular")`):
4143 /// the low-score paragraph box RT-DETR draws over its own high-score line
4144 /// boxes collapses to one region — the group's union, with the survivor's
4145 /// label and score — so region-scoped OCR reads each line once. Regions
4146 /// that merely sit near each other, and specials, are untouched.
4147 #[test]
4148 fn merge_overlapping_regulars_collapses_a_block_over_its_lines() {
4149 let mut regions = vec![
4150 region("text", 0.84, 60.0, 186.0, 270.0, 198.0),
4151 region("text", 0.80, 60.0, 160.0, 294.0, 172.0),
4152 region("text", 0.79, 59.0, 107.0, 272.0, 119.0),
4153 // The paragraph box, lower score, containing all three lines.
4154 region("text", 0.52, 59.0, 107.0, 295.0, 200.0),
4155 // Elsewhere on the page: stays as is.
4156 region("section_header", 0.77, 60.0, 71.0, 253.0, 86.0),
4157 // A picture the block overlaps is not a regular — never grouped.
4158 region("picture", 0.9, 50.0, 100.0, 300.0, 210.0),
4159 ];
4160 merge_overlapping_regulars(&mut regions);
4161 assert_eq!(regions.len(), 3, "{regions:?}");
4162 let block = regions
4163 .iter()
4164 .find(|r| r.label == "text")
4165 .expect("one text");
4166 // docling keeps the largest passing candidate unless a rival is both
4167 // comparable in size and > 0.05 more confident; the 16× larger block
4168 // passes, and a smaller line never replaces a larger current best.
4169 // Either way the survivor spans the whole group.
4170 assert_eq!(
4171 (block.l, block.t, block.r, block.b),
4172 (59.0, 107.0, 295.0, 200.0)
4173 );
4174 assert!(regions.iter().any(|r| r.label == "section_header"));
4175 assert!(regions.iter().any(|r| r.label == "picture"));
4176 }
4177
4178 /// The pairwise rules, each in the arrangement where it decides the
4179 /// outcome: docling seeds the survivor with the group's first passing
4180 /// cluster and a later one replaces it only when larger *and* within
4181 /// 0.05 confidence, so a rule that merely lets a cluster pass matters
4182 /// exactly when that cluster comes first — a same-sized list item ahead
4183 /// of a far more confident text box, a code box ahead of the text it
4184 /// contains. Without the rule either would be rejected outright (similar
4185 /// size, rival > 0.05 more confident) and the text box would win.
4186 #[test]
4187 fn merge_overlapping_regulars_follows_the_preference_rules() {
4188 let mut regions = vec![
4189 region("list_item", 0.6, 0.0, 0.0, 102.0, 20.0),
4190 region("text", 0.9, 0.0, 0.0, 100.0, 20.0),
4191 ];
4192 merge_overlapping_regulars(&mut regions);
4193 assert_eq!(regions.len(), 1);
4194 assert_eq!(regions[0].label, "list_item");
4195
4196 let mut regions = vec![
4197 region("code", 0.6, 0.0, 0.0, 100.0, 100.0),
4198 region("text", 0.9, 2.0, 2.0, 98.0, 98.0),
4199 ];
4200 merge_overlapping_regulars(&mut regions);
4201 assert_eq!(regions.len(), 1);
4202 assert_eq!(regions[0].label, "code");
4203
4204 // No rule applies: a near-identical rival that is > 0.05 more
4205 // confident rejects the candidate whatever the order.
4206 for order in [[0.9, 0.6], [0.6, 0.9]] {
4207 let mut regions = vec![
4208 region("text", order[0], 0.0, 0.0, 100.0, 20.0),
4209 region("text", order[1], 0.0, 0.0, 105.0, 21.0),
4210 ];
4211 merge_overlapping_regulars(&mut regions);
4212 assert_eq!(regions.len(), 1);
4213 assert_eq!(regions[0].score, 0.9, "the confident twin wins");
4214 assert_eq!(
4215 (regions[0].r, regions[0].b),
4216 (105.0, 21.0),
4217 "on the union box"
4218 );
4219 }
4220
4221 // Side by side (no containment, IoU 0): nothing to merge.
4222 let mut regions = vec![
4223 region("text", 0.9, 0.0, 0.0, 100.0, 20.0),
4224 region("text", 0.9, 0.0, 22.0, 100.0, 42.0),
4225 ];
4226 merge_overlapping_regulars(&mut regions);
4227 assert_eq!(regions.len(), 2);
4228 }
4229
4230 #[test]
4231 fn footer_under_a_body_less_heading_is_its_text() {
4232 // The reporting CV's last page: `## Languages` at t=773.7..783.4 and
4233 // its one-line body at 798.5..808.7, labelled page_footer (A4, 595 pt).
4234 let mut regions = vec![
4235 region("list_item", 0.9, 66.0, 754.0, 353.0, 765.0),
4236 region("section_header", 0.94, 43.0, 773.7, 89.5, 783.4),
4237 region("page_footer", 0.88, 43.0, 798.5, 527.8, 808.7),
4238 ];
4239 reclaim_heading_body_footers(&mut regions, 595.28);
4240 assert_eq!(regions[2].label, "text");
4241 assert_eq!(regions[1].label, "section_header");
4242
4243 // A heading with its own paragraph and a running footer below: kept.
4244 let mut regions = vec![
4245 region("section_header", 0.9, 43.0, 700.0, 120.0, 710.0),
4246 region("text", 0.9, 43.0, 714.0, 520.0, 780.0),
4247 region("page_footer", 0.9, 43.0, 798.0, 520.0, 808.0),
4248 ];
4249 reclaim_heading_body_footers(&mut regions, 595.28);
4250 assert_eq!(regions[2].label, "page_footer");
4251
4252 // A page number under a trailing heading is too narrow to be a body.
4253 let mut regions = vec![
4254 region("section_header", 0.9, 43.0, 773.0, 120.0, 783.0),
4255 region("page_footer", 0.9, 280.0, 798.0, 300.0, 808.0),
4256 ];
4257 reclaim_heading_body_footers(&mut regions, 595.28);
4258 assert_eq!(regions[1].label, "page_footer");
4259
4260 // Too far below the heading (a real footer after a heading that ends
4261 // the page): kept.
4262 let mut regions = vec![
4263 region("section_header", 0.9, 43.0, 700.0, 120.0, 710.0),
4264 region("page_footer", 0.9, 43.0, 798.0, 520.0, 808.0),
4265 ];
4266 reclaim_heading_body_footers(&mut regions, 595.28);
4267 assert_eq!(regions[1].label, "page_footer");
4268 }
4269
4270 fn region(label: &'static str, score: f32, l: f32, t: f32, r: f32, b: f32) -> Region {
4271 Region {
4272 label,
4273 score,
4274 l,
4275 t,
4276 r,
4277 b,
4278 }
4279 }
4280
4281 #[test]
4282 fn resolve_collapses_nested_code_keeping_the_larger_box() {
4283 // A tight high-score `code` box and a taller lower-score near-duplicate that
4284 // contains it must collapse to one — the *larger* box, so every cell stays
4285 // covered and nothing leaks out as orphan text.
4286 let tight = region("code", 0.95, 78.0, 292.0, 300.0, 330.0);
4287 let wide = region("code", 0.66, 63.0, 260.0, 320.0, 346.0);
4288 let kept = super::resolve(vec![tight, wide]);
4289 assert_eq!(kept.len(), 1, "nested code boxes must collapse to one");
4290 assert!(
4291 kept[0].l == 63.0 && kept[0].b == 346.0,
4292 "the larger containing box is kept"
4293 );
4294 }
4295
4296 #[test]
4297 fn resolve_keeps_distinct_and_differently_typed_regions() {
4298 // A text box fully inside a lower-score *table* must NOT be collapsed (the
4299 // code dedup is code-only), and two separate code blocks stay separate.
4300 let text = region("text", 0.95, 90.0, 210.0, 200.0, 230.0);
4301 let table = region("table", 0.60, 80.0, 200.0, 400.0, 500.0);
4302 assert_eq!(super::resolve(vec![text, table]).len(), 2);
4303
4304 let code_a = region("code", 0.9, 78.0, 100.0, 300.0, 140.0);
4305 let code_b = region("code", 0.9, 78.0, 300.0, 300.0, 360.0); // far below, no overlap
4306 assert_eq!(super::resolve(vec![code_a, code_b]).len(), 2);
4307 }
4308
4309 /// A two-column glossary page came out as three column
4310 /// tables *and* one low-score whole-page table over them. docling's wrapper
4311 /// `_remove_overlapping_clusters` keeps one table per overlapping group
4312 /// (here the whole-page one: > 2× every rival's area and ≤ 0.2 less
4313 /// confident than the running best); `greedy` alone kept all four and
4314 /// emitted every cell twice.
4315 #[test]
4316 fn resolve_keeps_one_table_per_nested_group() {
4317 let kept = super::resolve(vec![
4318 region("table", 0.71, 26.0, 203.0, 183.0, 558.0),
4319 region("table", 0.67, 26.0, 55.0, 183.0, 196.0),
4320 region("table", 0.66, 196.0, 56.0, 354.0, 561.0),
4321 region("table", 0.53, 25.0, 53.0, 354.0, 561.0),
4322 ]);
4323 assert_eq!(kept.len(), 1, "one survivor per overlapping group");
4324 assert_eq!((kept[0].l, kept[0].b), (25.0, 561.0));
4325 // Side-by-side tables that don't overlap stay separate.
4326 let kept = super::resolve(vec![
4327 region("table", 0.9, 26.0, 55.0, 183.0, 558.0),
4328 region("table", 0.9, 196.0, 56.0, 354.0, 561.0),
4329 ]);
4330 assert_eq!(kept.len(), 2);
4331 }
4332
4333 /// A dense data table detected as both picture (0.80) and table (0.62) on one box.
4334 /// The picture is ≥ 0.1 more confident, so `_handle_cross_type_overlaps`
4335 /// keeps both, and the dense table text passed the text-panel gates: the
4336 /// demoted paragraph repeated every cell the table grid renders. A
4337 /// paragraph > 80 % inside a surviving table is the table's child and is
4338 /// not emitted; a panel with no table under it still demotes.
4339 #[test]
4340 fn text_panel_over_a_table_does_not_repeat_its_cells() {
4341 let lines = |t0: f32| -> Vec<TextCell> {
4342 (0..4)
4343 .map(|i| {
4344 let t = t0 + 10.0 * i as f32;
4345 cell("1 2 3 4 5 6 7 8 9 10 11 12", 5.0, t, 95.0, t + 8.0)
4346 })
4347 .collect()
4348 };
4349 let mut cells = lines(0.0);
4350 cells.extend(lines(200.0));
4351 let mut regions = vec![
4352 region("picture", 0.80, 0.0, 0.0, 100.0, 45.0),
4353 region("table", 0.62, 0.0, 0.0, 100.0, 45.0),
4354 region("picture", 0.80, 0.0, 200.0, 100.0, 245.0),
4355 ];
4356 super::recover_text_panels(&mut regions, &cells);
4357 assert_eq!(
4358 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
4359 ["table", "text"]
4360 );
4361 assert_eq!(regions[1].t, 200.0, "the table-free panel still demotes");
4362 }
4363
4364 /// docling's `ListItemMarkerProcessor` + docling-core's Markdown list rules
4365 /// (see `list_item_node`): a bullet glyph or `-` is split off and the
4366 /// serializer's own `-` printed (2305's OTSL list reads `- "C" cell`, not
4367 /// `- - "C" cell`); `N.` is an ordered item; any other marker with a letter
4368 /// or digit stays in the text behind the bullet; no marker → plain bullet.
4369 #[test]
4370 fn list_item_markers_split_like_docling() {
4371 let text_of = |n: &Node| match n {
4372 Node::ListItem {
4373 ordered,
4374 number,
4375 text,
4376 marker,
4377 ..
4378 } => (*ordered, *number, text.clone(), marker.clone()),
4379 other => panic!("{other:?}"),
4380 };
4381 let loc = [0, 0, 100, 10];
4382 assert_eq!(
4383 text_of(&super::list_item_node(
4384 "- \"C\" cell - a new table cell",
4385 loc,
4386 false
4387 )),
4388 (
4389 false,
4390 0,
4391 "\"C\" cell - a new table cell".into(),
4392 Some("-".into())
4393 )
4394 );
4395 assert_eq!(
4396 text_of(&super::list_item_node("• Bullet text", loc, false)),
4397 (false, 0, "Bullet text".into(), Some("•".into()))
4398 );
4399 assert_eq!(
4400 text_of(&super::list_item_node("3. Third step", loc, false)),
4401 (true, 3, "Third step".into(), Some("3.".into()))
4402 );
4403 assert_eq!(
4404 text_of(&super::list_item_node("a) Option", loc, false)),
4405 (false, 0, "a) Option".into(), Some("a)".into()))
4406 );
4407 assert_eq!(
4408 text_of(&super::list_item_node("1.2 Nested outline", loc, false)),
4409 (false, 0, "1.2 Nested outline".into(), Some("1.2".into()))
4410 );
4411 // 2203's `3.a. If all IOU scores…`: a compound marker is not an item
4412 // number — docling prints `- 3.a. If all…`.
4413 assert_eq!(
4414 text_of(&super::list_item_node("3.a. If all IOU scores", loc, false)),
4415 (
4416 false,
4417 0,
4418 "3.a. If all IOU scores".into(),
4419 Some("3.a.".into())
4420 )
4421 );
4422 // A glued symbol-font bullet is stripped, a spaced one is the marker.
4423 assert_eq!(
4424 text_of(&super::list_item_node("•Glued", loc, false)),
4425 (false, 0, "Glued".into(), Some("·".into()))
4426 );
4427 // No whitespace after the glyph → not a marker (docling's `\s` is required).
4428 assert_eq!(
4429 text_of(&super::list_item_node("-5 degrees", loc, false)),
4430 (false, 0, "-5 degrees".into(), Some("·".into()))
4431 );
4432 // The remaining numbered shapes, first-wins like docling's list.
4433 for (input, marker, body) in [
4434 ("1.2.3. Deep", "1.2.3.", "Deep"),
4435 ("9a) Nine-a", "9a)", "Nine-a"),
4436 ("(3.a) Paren", "(3.a)", "Paren"),
4437 ("12) Twelve", "12)", "Twelve"),
4438 ("(4) Four", "(4)", "Four"),
4439 ("[7] Seven", "[7]", "Seven"),
4440 ("iv. Roman", "iv.", "Roman"),
4441 ("IX. Roman", "IX.", "Roman"),
4442 ("b. Letter", "b.", "Letter"),
4443 ("B) Letter", "B)", "Letter"),
4444 ] {
4445 assert_eq!(
4446 super::split_list_marker(input),
4447 Some((marker, body, true)),
4448 "{input}"
4449 );
4450 }
4451 // A `1.2.` whose optional dot would eat the separator backtracks like
4452 // Python's regex; a marker with nothing after the whitespace is none.
4453 assert_eq!(
4454 super::split_list_marker("1.2.\tx"),
4455 Some(("1.2.", "x", true))
4456 );
4457 assert_eq!(super::split_list_marker("1. "), None);
4458 assert_eq!(super::split_list_marker("• "), None);
4459 assert_eq!(
4460 text_of(&super::list_item_node("Plain item", loc, false)),
4461 (false, 0, "Plain item".into(), Some("·".into()))
4462 );
4463 }
4464
4465 #[test]
4466 fn code_language_label_above_code_is_detected() {
4467 // A bare "XML" token directly above a code box is a language label; a real
4468 // heading above the same code is not; a language word with no code below is
4469 // left alone.
4470 let label = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
4471 let code = region("code", 0.7, 77.0, 552.0, 290.0, 640.0);
4472 let heading = region("section_header", 0.9, 76.0, 500.0, 260.0, 512.0);
4473 let cells = vec![
4474 cell("XML", 78.0, 541.0, 94.0, 548.0), // inside `label`
4475 cell("Overview", 78.0, 501.0, 250.0, 511.0), // inside `heading`
4476 ];
4477 let drop = super::code_language_labels(&[label, code, heading], &cells);
4478 assert_eq!(drop, vec![true, false, false], "only the label is consumed");
4479
4480 // Same label with no code region present → not consumed.
4481 let label2 = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
4482 let only = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
4483 assert_eq!(super::code_language_labels(&[label2], &only), vec![false]);
4484
4485 // A label swallowed into the top of a wider code box (negative gap) is still
4486 // recognized.
4487 let inside_lbl = region("text", 0.9, 76.0, 540.0, 96.0, 549.0);
4488 let wide_code = region("code", 0.7, 63.0, 531.0, 320.0, 654.0);
4489 let cells2 = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
4490 assert_eq!(
4491 super::code_language_labels(&[inside_lbl, wide_code], &cells2),
4492 vec![true, false]
4493 );
4494
4495 assert!(super::is_code_language("XML") && super::is_code_language("c#"));
4496 assert!(!super::is_code_language("Configure") && !super::is_code_language("XML schema"));
4497 }
4498
4499 #[test]
4500 fn code_region_text_keeps_lines_and_indentation() {
4501 // Three source lines; each glyph is 6 units wide (width / chars = 6), so the
4502 // `int X;` line indented to x=22 is (22-10)/6 = 2 spaces in.
4503 let region = Region {
4504 label: "code",
4505 score: 1.0,
4506 l: 0.0,
4507 t: -5.0,
4508 r: 100.0,
4509 b: 40.0,
4510 };
4511 let cells = vec![
4512 cell("struct P {", 10.0, 0.0, 70.0, 10.0),
4513 cell("int X;", 22.0, 12.0, 58.0, 22.0),
4514 cell("}", 10.0, 24.0, 16.0, 34.0),
4515 ];
4516 assert_eq!(code_region_text(®ion, &cells), "struct P {\n int X;\n}");
4517 }
4518
4519 #[test]
4520 fn code_region_text_tightens_punctuation_without_eating_indentation() {
4521 // A fluent `.Foo()` line at x=22 (2 chars in). Per-line tightening must not
4522 // consume the leading indent space by matching " ." across it.
4523 let region = Region {
4524 label: "code",
4525 score: 1.0,
4526 l: 0.0,
4527 t: -5.0,
4528 r: 100.0,
4529 b: 40.0,
4530 };
4531 let cells = vec![
4532 cell("builder", 10.0, 0.0, 52.0, 10.0),
4533 // pdfium spaced the call: ".Foo (x)" tightens to ".Foo(x)", still 2-indented.
4534 cell(".Foo (x)", 22.0, 12.0, 70.0, 22.0),
4535 ];
4536 assert_eq!(code_region_text(®ion, &cells), "builder\n .Foo(x)");
4537 }
4538
4539 #[test]
4540 fn code_region_text_orders_out_of_order_cells_and_ignores_blank_lines() {
4541 let region = Region {
4542 label: "code",
4543 score: 1.0,
4544 l: 0.0,
4545 t: -5.0,
4546 r: 100.0,
4547 b: 60.0,
4548 };
4549 // Fed bottom-up and with a whitespace-only cell; output is top-down, no blank.
4550 let cells = vec![
4551 cell("b();", 10.0, 24.0, 34.0, 34.0),
4552 cell(" ", 10.0, 12.0, 20.0, 22.0),
4553 cell("a();", 10.0, 0.0, 34.0, 10.0),
4554 ];
4555 assert_eq!(code_region_text(®ion, &cells), "a();\nb();");
4556 // No code cells → empty, so the caller falls back to the prose text.
4557 assert_eq!(code_region_text(®ion, &[]), "");
4558 }
4559
4560 fn para(text: &str) -> Node {
4561 Node::Paragraph { text: text.into() }
4562 }
4563
4564 /// Run a node sequence through [`StreamAssembler`] with the given page splits
4565 /// and assert the flushed result equals one-shot [`merge_continuations`].
4566 fn assert_stream_eq(nodes: &[Node], splits: &[usize]) {
4567 let mut want = nodes.to_vec();
4568 merge_continuations(&mut want);
4569
4570 let mut asm = StreamAssembler::new();
4571 let mut got = Vec::new();
4572 let mut start = 0;
4573 for &end in splits {
4574 got.extend(asm.push(nodes[start..end].to_vec()));
4575 start = end;
4576 }
4577 got.extend(asm.push(nodes[start..].to_vec()));
4578 got.extend(asm.finish());
4579 assert_eq!(got, want, "stream assembly diverged (splits={splits:?})");
4580 }
4581
4582 #[test]
4583 fn stream_assembler_matches_merge_continuations() {
4584 // Open fragment + lowercase continuation split across a page boundary.
4585 let cross = [para("the definition of"), para("lists in scope")];
4586 assert_stream_eq(&cross, &[1]);
4587 assert_stream_eq(&cross, &[]);
4588
4589 // Continuation that wraps around a figure (+ its caption) on the boundary.
4590 let wrap = [
4591 para("the wing type that is"),
4592 Node::Picture {
4593 caption: None,
4594 caption_href: None,
4595 image: None,
4596 classification: None,
4597 caption_parent: Default::default(),
4598 },
4599 para("Fig. 1. a diagram"),
4600 para("the most common kind"),
4601 ];
4602 for splits in [&[][..], &[1][..], &[2][..], &[3][..], &[1, 3][..]] {
4603 assert_stream_eq(&wrap, splits);
4604 }
4605
4606 // A heading between fragments blocks the merge (must still flush correctly).
4607 let blocked = [
4608 para("ends mid word and"),
4609 Node::Heading {
4610 level: 2,
4611 text: "New Section".into(),
4612 },
4613 para("more body here"),
4614 ];
4615 for splits in [&[][..], &[1][..], &[2][..]] {
4616 assert_stream_eq(&blocked, splits);
4617 }
4618
4619 // A chain across three pages: each page is one open lowercase fragment.
4620 let chain = [
4621 para("alpha beta"),
4622 para("gamma delta"),
4623 para("epsilon zeta"),
4624 ];
4625 assert_stream_eq(&chain, &[1, 2]);
4626 }
4627
4628 #[test]
4629 fn clean_text_dehyphenates_and_normalizes_typography() {
4630 // U+0002 line-wrap hyphen + the join space → merged word (like docling).
4631 assert_eq!(clean_text("com\u{2} pact"), "compact");
4632 assert_eq!(clean_text("end-to\u{2} end deep"), "end-toend deep");
4633 // A stray wrap hyphen (no following join) is dropped.
4634 assert_eq!(clean_text("word\u{2}"), "word");
4635 // Typographic punctuation → ASCII: every curly quote becomes `'`
4636 // (docling-parse's sanitizer table), a literal `"` stays.
4637 assert_eq!(
4638 clean_text("Graph\u{2019}s \u{201c}x\u{201d} \"y\""),
4639 "Graph's 'x' \"y\""
4640 );
4641 assert_eq!(clean_text("a\u{2026}"), "a...");
4642 // The docling-parse sanitizer's internal spacing is preserved as
4643 // placed; line breaks/tabs normalize to a space, ends trim.
4644 assert_eq!(clean_text("a b\nc"), "a b c");
4645 }
4646
4647 /// docling#4064: a form's children are emitted together where the form
4648 /// sits in the top-level order, not interleaved with surrounding text.
4649 #[test]
4650 fn form_children_stay_together_in_reading_order() {
4651 let reg = |label: &'static str, l: f32, t: f32, r: f32, b: f32| Region {
4652 label,
4653 score: 0.9,
4654 l,
4655 t,
4656 r,
4657 b,
4658 };
4659 // Page: intro text, then a form spanning the left column with two
4660 // fields and a table inside, while a right-column paragraph sits
4661 // level with the form's first field (it would otherwise be read
4662 // between the form's children).
4663 let mut items = vec![
4664 reg("text", 50.0, 50.0, 550.0, 70.0), // 0 intro
4665 reg("form", 50.0, 100.0, 300.0, 400.0), // 1 container
4666 reg("text", 60.0, 110.0, 290.0, 130.0), // 2 field A (child)
4667 reg("text", 320.0, 110.0, 550.0, 130.0), // 3 right column paragraph
4668 reg("table", 60.0, 150.0, 290.0, 300.0), // 4 table (child)
4669 reg("text", 60.0, 320.0, 290.0, 340.0), // 5 field B (child)
4670 reg("text", 50.0, 450.0, 550.0, 470.0), // 6 outro
4671 ];
4672 let cids = super::cluster_cids(&items, &[]);
4673 super::order_with_containers(&mut items, &cids, 600.0, 800.0, |r| r);
4674 let order: Vec<(&str, f32)> = items.iter().map(|r| (r.label, r.t)).collect();
4675 // The form block (container, then its children top-down) is one unit.
4676 let form_pos = order.iter().position(|(l, _)| *l == "form").unwrap();
4677 assert_eq!(
4678 &order[form_pos..form_pos + 4],
4679 &[
4680 ("form", 100.0),
4681 ("text", 110.0),
4682 ("table", 150.0),
4683 ("text", 320.0)
4684 ]
4685 );
4686 assert_eq!(order[0], ("text", 50.0));
4687 assert_eq!(order[order.len() - 1], ("text", 450.0));
4688 // Without a container the plain order interleaves by geometry.
4689 let mut flat: Vec<Region> = items
4690 .iter()
4691 .filter(|r| r.label != "form")
4692 .cloned()
4693 .collect();
4694 let cids = super::cluster_cids(&flat, &[]);
4695 super::order_regions(&mut flat, &cids, 600.0, 800.0, |r| r);
4696 assert_ne!(
4697 flat.iter().map(|r| r.t).collect::<Vec<_>>(),
4698 order
4699 .iter()
4700 .filter(|(l, _)| *l != "form")
4701 .map(|(_, t)| *t)
4702 .collect::<Vec<_>>()
4703 );
4704 }
4705
4706 /// docling#3906: a picture inside a table lands in the covering cell,
4707 /// chosen by the picture's inferred grid position when cell boxes overlap.
4708 #[test]
4709 fn picture_matches_the_cell_at_its_grid_position() {
4710 let cell = |r: usize, c: usize, bbox: [f32; 4]| docling_core::TableCell {
4711 text: format!("r{r}c{c}"),
4712 bbox: Some(bbox),
4713 start_row: r,
4714 start_col: c,
4715 row_span: 1,
4716 col_span: 1,
4717 column_header: false,
4718 row_header: false,
4719 row_section: false,
4720 };
4721 // 2×2 grid; the (1,0) cell box is generous and also covers the picture.
4722 let cells = vec![
4723 cell(0, 0, [0.0, 0.0, 100.0, 50.0]),
4724 cell(0, 1, [100.0, 0.0, 200.0, 50.0]),
4725 cell(1, 0, [0.0, 50.0, 100.0, 100.0]),
4726 cell(1, 1, [100.0, 50.0, 200.0, 100.0]),
4727 ];
4728 let pic = Region {
4729 label: "picture",
4730 score: 0.9,
4731 l: 110.0,
4732 t: 60.0,
4733 r: 190.0,
4734 b: 95.0,
4735 };
4736 assert_eq!(super::match_picture_to_cell(&pic, &cells), Some((1.0, 3)));
4737 // A picture only half inside any cell is not nested.
4738 let straddling = Region {
4739 label: "picture",
4740 score: 0.9,
4741 l: 60.0,
4742 t: 60.0,
4743 r: 160.0,
4744 b: 95.0,
4745 };
4746 assert_eq!(super::match_picture_to_cell(&straddling, &cells), None);
4747 }
4748
4749 /// docling#4052 (2.122): a line-final dash fuses the wrapped word only
4750 /// when attached to it; a detached dash is a literal and the lines join
4751 /// with a space.
4752 #[test]
4753 fn line_final_hyphen_fuses_only_when_attached_to_a_word() {
4754 let line = |text: &str, t: f32| TextCell {
4755 text: text.to_string(),
4756 l: 0.0,
4757 t,
4758 r: 100.0,
4759 b: t + 10.0,
4760 };
4761 // `algo-` / `rithms`: attached hyphen, alnum on both sides → fused.
4762 assert_eq!(
4763 cells_text(vec![&line("algo-", 0.0), &line("rithms", 12.0)]),
4764 "algorithms"
4765 );
4766 // `pp. 545-` / `561`: attached, digits count as alnum → `545561` (upstream).
4767 assert_eq!(
4768 cells_text(vec![&line("pp. 545-", 0.0), &line("561", 12.0)]),
4769 "pp. 545561"
4770 );
4771 // A dash after whitespace — a separator or a lone `-` cell — is kept and
4772 // the lines take the ordinary joining space.
4773 assert_eq!(
4774 cells_text(vec![&line("range -", 0.0), &line("wide", 12.0)]),
4775 "range - wide"
4776 );
4777 assert_eq!(
4778 cells_text(vec![&line("-", 0.0), &line("item", 12.0)]),
4779 "- item"
4780 );
4781 // Attached but the next line opens with no word (`x-` / `...`): dash
4782 // kept and, as before, no separating space.
4783 assert_eq!(
4784 cells_text(vec![&line("x-", 0.0), &line("...", 12.0)]),
4785 "x-..."
4786 );
4787 }
4788
4789 #[test]
4790 fn lam_alef_only_swaps_a_genuinely_reversed_ligature() {
4791 // A mid-word `alef-variant + lam` is pdfium's reversed lam-alef ligature and
4792 // is swapped back to logical `lam + alef-variant` (`ب أ ل` → `ب ل أ`).
4793 assert_eq!(
4794 clean_text("\u{0628}\u{0623}\u{0644}"),
4795 "\u{0628}\u{0644}\u{0623}"
4796 );
4797 // But when the alef-variant is *already* preceded by a lam it is the logical
4798 // ligature `لآ`; the following lam is the next syllable's letter and must not
4799 // move. `التعلم الآلي` must stay `الآلي`, not become `اللآي`.
4800 assert_eq!(
4801 clean_text("\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"),
4802 "\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"
4803 );
4804 }
4805
4806 /// The #419 page, in points: three layout boxes over one paragraph, two of
4807 /// them ending partway through a line. The sliced lines miss the 0.2 claim
4808 /// and become orphans; the third model box starts above the second orphan,
4809 /// so unfitted the reading order emits that box first and strands the line.
4810 fn sliced_paragraph() -> (Vec<Region>, Vec<TextCell>) {
4811 let line = |text: &str, t: f32, r: f32| cell(text, 60.0, t, r, t + 11.0);
4812 let cells = vec![
4813 line("The mission of this series is to improve", 135.0, 458.0),
4814 line("The books in this series are technical,", 147.0, 458.0),
4815 line("substantial. The authors are", 159.0, 458.0),
4816 line("highly experienced craftsmen and", 171.5, 458.0), // sliced: 1.5/11 under box A
4817 line("actually works in practice, as opposed", 185.0, 458.0),
4818 line("about what the author has done, not", 197.0, 458.0),
4819 line("about programming, there will be lots", 210.5, 458.0), // sliced: 1.5/11 under box B
4820 line("will be lots of case studies from real", 223.0, 206.0), // C's line
4821 ];
4822 let regions = vec![
4823 region("text", 0.9, 60.0, 132.0, 458.0, 173.0), // A: three lines + a sliver of the 4th
4824 region("text", 0.9, 60.0, 184.0, 458.0, 212.0), // B: two lines + a sliver of the 7th
4825 region("text", 0.9, 60.0, 216.0, 206.0, 227.0), // C: last line, box opening 5.5pt too early
4826 ];
4827 (regions, cells)
4828 }
4829
4830 fn ordered_texts(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
4831 let mut items: Vec<Region> = regions.to_vec();
4832 let cids = super::cluster_cids(&items, cells);
4833 super::order_regions(&mut items, &cids, 500.0, 700.0, |r| r);
4834 super::region_texts_exclusive(&items, cells)
4835 .into_iter()
4836 .map(|t| t.chars().take(9).collect())
4837 .collect()
4838 }
4839
4840 /// #419: fitted to its cells, a model box that cut a line in half no longer
4841 /// overlaps the orphan that line became, so the orphan orders where it
4842 /// reads; unfitted, the same page strands the line after the paragraph.
4843 #[test]
4844 fn fitting_boxes_to_cells_puts_a_sliced_line_back_in_order() {
4845 let (mut regions, cells) = sliced_paragraph();
4846 super::add_orphan_regions(&mut regions, &cells);
4847 assert_eq!(regions.len(), 5, "two orphan lines");
4848 // The defect, for the record: C (top 216) is not strictly below the
4849 // orphan at 210.5–221.5, so the graph orders C first.
4850 assert_eq!(
4851 ordered_texts(®ions, &cells).last().map(String::as_str),
4852 Some("about pro")
4853 );
4854
4855 super::fit_regions_to_cells(&mut regions, &cells);
4856 assert_eq!(regions.len(), 5);
4857 // A ends on its last claimed line, C starts on its only one.
4858 assert_eq!((regions[0].t, regions[0].b), (135.0, 170.0));
4859 assert_eq!((regions[2].t, regions[2].b), (223.0, 234.0));
4860 assert_eq!(
4861 ordered_texts(®ions, &cells),
4862 [
4863 "The missi",
4864 "highly ex",
4865 "actually ",
4866 "about pro",
4867 "will be l"
4868 ]
4869 );
4870 }
4871
4872 /// An orphan the fitted paragraph box surrounds (a short middle line the
4873 /// narrow model box missed while claiming the lines around it) is folded
4874 /// into the paragraph; an empty regular box goes away, a formula stays, a
4875 /// picture is never refitted, and a page with no cells is left untouched.
4876 #[test]
4877 fn fitting_folds_surrounded_orphans_and_drops_empty_regulars() {
4878 let wide = |text: &str, t: f32| cell(text, 60.0, t, 400.0, t + 11.0);
4879 let cells = vec![
4880 wide("first line of the paragraph", 100.0),
4881 cell("stray", 250.0, 112.0, 400.0, 123.0), // clear of the narrow box
4882 wide("third line of the paragraph", 124.0),
4883 ];
4884 let mut regions = vec![
4885 // Narrow box: claims the wide lines at 0.41, misses the short one.
4886 region("text", 0.9, 60.0, 98.0, 200.0, 136.0),
4887 region("section_header", 0.8, 60.0, 300.0, 200.0, 320.0), // no cells
4888 region("formula", 0.8, 60.0, 340.0, 200.0, 360.0), // no cells, kept
4889 region("picture", 0.8, 0.0, 400.0, 500.0, 600.0),
4890 ];
4891 super::add_orphan_regions(&mut regions, &cells);
4892 assert_eq!(regions.len(), 5, "the short line became an orphan");
4893 super::fit_regions_to_cells(&mut regions, &cells);
4894 let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
4895 assert_eq!(labels, ["text", "formula", "picture"]);
4896 let para = ®ions[0];
4897 assert_eq!(
4898 (para.l, para.t, para.r, para.b),
4899 (60.0, 100.0, 400.0, 135.0)
4900 );
4901 assert_eq!(
4902 super::region_texts_exclusive(®ions, &cells)[0],
4903 "first line of the paragraph stray third line of the paragraph"
4904 );
4905 assert_eq!(
4906 (regions[2].t, regions[2].b),
4907 (400.0, 600.0),
4908 "picture untouched"
4909 );
4910
4911 let mut untouched = vec![region("text", 0.9, 0.0, 0.0, 10.0, 10.0)];
4912 super::fit_regions_to_cells(&mut untouched, &[]);
4913 assert_eq!(untouched.len(), 1, "no cells yet: nothing dropped");
4914 }
4915}