docling_pdf/assemble.rs
1//! Layout-driven assembly: map detected [`Region`]s + text cells to a
2//! [`DoclingDocument`], mirroring docling's page-assembly + reading-order.
3//!
4//! Overlapping detections are resolved greedily by score, each text cell is
5//! assigned to its best-containing region, regions are ordered in reading order
6//! (two-column aware), and each becomes a typed node by its layout label.
7
8use docling_core::{CaptionParent, Node, PictureClass, PictureImage, Table};
9#[cfg(any(feature = "ml", feature = "ocr-prep"))]
10use image::RgbImage;
11
12use crate::layout::Region;
13use crate::pdfium_backend::{PdfPage, TextCell};
14
15fn area(l: f32, t: f32, r: f32, b: f32) -> f32 {
16 ((r - l).max(0.0)) * ((b - t).max(0.0))
17}
18
19/// Intersection area of two boxes.
20fn inter(a: &Region, l: f32, t: f32, r: f32, b: f32) -> f32 {
21 let il = a.l.max(l);
22 let it = a.t.max(t);
23 let ir = a.r.min(r);
24 let ib = a.b.min(b);
25 area(il, it, ir, ib)
26}
27
28/// Wrapper (structured-region) labels, ported from docling
29/// `LayoutPostprocessor.WRAPPER_TYPES`: a region that *contains* other regions
30/// and renders as a structured block (a table / table-of-contents index), not as
31/// its own flat text.
32fn is_wrapper(label: &str) -> bool {
33 matches!(
34 label,
35 "table" | "document_index" | "form" | "key_value_region"
36 )
37}
38
39/// Labels docling's table-structure (TableFormer) model runs on and that render
40/// as a Markdown table: a plain `table` and a `document_index` (a table of
41/// contents), which docling assembles as a `TableItem` too.
42pub fn is_table_like(label: &str) -> bool {
43 matches!(label, "table" | "document_index")
44}
45
46/// Greedily keep regions by descending score, dropping a region that is mostly
47/// covered by an already-kept one (RT-DETR emits overlapping duplicates).
48fn greedy(mut regions: Vec<Region>) -> Vec<Region> {
49 regions.sort_by(|a, b| b.score.total_cmp(&a.score));
50 let mut kept: Vec<Region> = Vec::new();
51 for r in regions {
52 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
53 let covered = kept.iter().any(|k| {
54 let i = inter(&r, k.l, k.t, k.r, k.b);
55 let ka = area(k.l, k.t, k.r, k.b).max(1.0);
56 // drop if most of r is inside k, or they strongly mutually overlap
57 i / ra > 0.7 || i / (ra + ka - i) > 0.5
58 });
59 if !covered {
60 kept.push(r);
61 }
62 }
63 kept
64}
65
66/// Resolve overlapping RT-DETR detections, ported from the bucket structure of
67/// docling's `LayoutPostprocessor`: regular, picture and wrapper clusters live in
68/// **separate** spatial indexes and are de-overlapped independently, so a
69/// high-score picture never suppresses a lower-score table or table-of-contents
70/// index (the redp5110 TOC that was otherwise replaced by a picture box). A
71/// cross-type pass first drops a picture that nearly coincides with a table
72/// (`_handle_cross_type_overlaps`), keeping the structured table.
73/// docling's `_remove_overlapping_clusters("picture")`: same-label picture
74/// detections whose boxes heavily overlap (IoU > 0.8, or either box > 80 %
75/// contained in the other) form one group, and a single survivor is kept per
76/// group. Survivor selection ports `_should_prefer_cluster` /
77/// `_select_best_cluster_from_group` with the picture params
78/// (`area_threshold` 2.0, `conf_threshold` 0.3): a candidate is rejected only
79/// when a rival is both comparable in size (candidate ≤ 2× its area) and
80/// clearly more confident (> 0.3); among the survivors the *larger* box wins
81/// unless it is > 0.3 less confident. Net effect on the corpus: a figure the
82/// detector proposes both whole and as its sub-panels (2206's four-thumbnail
83/// Figure 1) collapses to the whole-figure box, exactly like docling.
84pub(crate) fn dedup_pictures(regions: &mut Vec<Region>) {
85 remove_overlapping_specials(regions, |l| l == "picture", 2.0, 0.3);
86}
87
88/// docling's `_remove_overlapping_clusters` for one special bucket (the
89/// regions whose label satisfies `in_bucket`), with that bucket's
90/// `OVERLAP_PARAMS`: picture (2.0, 0.3) — see [`dedup_pictures`] — or wrapper
91/// (2.0, 0.2) for the table bucket in [`resolve`]. Regions outside the bucket
92/// are untouched; a bucket member overlapping nothing is always kept.
93fn remove_overlapping_specials(
94 regions: &mut Vec<Region>,
95 in_bucket: impl Fn(&str) -> bool,
96 area_threshold: f32,
97 conf_threshold: f32,
98) {
99 let idx: Vec<usize> = (0..regions.len())
100 .filter(|&i| in_bucket(regions[i].label))
101 .collect();
102 if idx.len() < 2 {
103 return;
104 }
105 // Union-find over the bucket.
106 let mut parent: Vec<usize> = (0..idx.len()).collect();
107 fn find(parent: &mut [usize], i: usize) -> usize {
108 let mut root = i;
109 while parent[root] != root {
110 root = parent[root];
111 }
112 let mut cur = i;
113 while parent[cur] != root {
114 let next = parent[cur];
115 parent[cur] = root;
116 cur = next;
117 }
118 root
119 }
120 let boxed = |r: &Region| (r.l, r.t, r.r, r.b);
121 for a in 0..idx.len() {
122 for b in (a + 1)..idx.len() {
123 let (ra, rb) = (®ions[idx[a]], ®ions[idx[b]]);
124 let (al, at, ar, ab_) = boxed(ra);
125 let (bl, bt, br, bb) = boxed(rb);
126 let ix = (ar.min(br) - al.max(bl)).max(0.0);
127 let iy = (ab_.min(bb) - at.max(bt)).max(0.0);
128 let inter = ix * iy;
129 let aa = area(al, at, ar, ab_).max(f32::EPSILON);
130 let ba = area(bl, bt, br, bb).max(f32::EPSILON);
131 let iou = inter / (aa + ba - inter).max(f32::EPSILON);
132 if iou > 0.8 || inter / aa > 0.8 || inter / ba > 0.8 {
133 let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
134 if pa != pb {
135 parent[pa] = pb;
136 }
137 }
138 }
139 }
140 // Per group, run docling's pairwise preference + larger-wins selection.
141 let mut groups: std::collections::HashMap<usize, Vec<usize>> = std::collections::HashMap::new();
142 for i in 0..idx.len() {
143 let root = find(&mut parent, i);
144 groups.entry(root).or_default().push(i);
145 }
146 let mut drop = vec![false; regions.len()];
147 for group in groups.values() {
148 if group.len() < 2 {
149 continue;
150 }
151 let area_of = |i: usize| {
152 let r = ®ions[idx[i]];
153 area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
154 };
155 let mut best: Option<usize> = None;
156 for &cand in group {
157 let passes = group.iter().all(|&other| {
158 if other == cand {
159 return true;
160 }
161 let area_ratio = area_of(cand) / area_of(other);
162 let conf_diff = regions[idx[other]].score - regions[idx[cand]].score;
163 !(area_ratio <= area_threshold && conf_diff > conf_threshold)
164 });
165 if passes {
166 best = Some(match best {
167 None => cand,
168 Some(cur) => {
169 if area_of(cand) > area_of(cur)
170 && regions[idx[cur]].score - regions[idx[cand]].score <= conf_threshold
171 {
172 cand
173 } else {
174 cur
175 }
176 }
177 });
178 }
179 }
180 // Every candidate rejected can't happen with docling's rule (rejection
181 // needs a strictly better rival); guard with highest score anyway.
182 let keep = best.unwrap_or_else(|| {
183 *group
184 .iter()
185 .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(®ions[idx[b]].score))
186 .expect("non-empty group")
187 });
188 for &i in group {
189 if i != keep {
190 drop[idx[i]] = true;
191 }
192 }
193 }
194 let mut keep_iter = drop.into_iter();
195 regions.retain(|_| !keep_iter.next().expect("aligned"));
196}
197
198/// docling's `_remove_overlapping_clusters("regular")` for what [`greedy`]
199/// leaves standing, run on OCR'd pages before the region-scoped OCR (found
200/// with #471's synthetic scan: every line of the page came out twice).
201///
202/// `greedy` keeps regions by descending score and drops a candidate mostly
203/// inside an already-kept one — so a *lower*-score block that contains
204/// several higher-score line boxes (RT-DETR's favourite reading of a sparse
205/// scanned page: every line, plus the paragraph) survives alongside them.
206/// On a digital page that is harmless: [`fit_regions_to_cells`] hands each
207/// text cell to one owner and drops the regions left empty. On a scanned
208/// page the cells don't exist yet — they come from OCR of *each* region's
209/// crop — so the block and its lines were each recognized, and the block,
210/// which then owned both cell sets, read every line twice.
211///
212/// Upstream never has this problem because its OCR runs over the bitmap
213/// before layout postprocessing and each cell is assigned once; the
214/// postprocessor's regular pass then groups clusters that overlap (IoU > 0.8,
215/// or either > 80 % contained in the other) with a union-find, keeps one
216/// survivor per group via `_should_prefer_cluster` /
217/// `_select_best_cluster_from_group` (`area_threshold` 1.3, `conf_threshold`
218/// 0.05; a LIST_ITEM beats a same-sized TEXT, a CODE box beats what it
219/// contains) and merges the losers' cells into it, whose box is then fitted
220/// to those cells. The equivalent for region-scoped OCR: one survivor per
221/// group, keeping its label and score, with the group's **union** box so the
222/// single crop still covers every merged line. Pictures and wrappers are not
223/// regulars and are left alone ([`dedup_pictures`], [`resolve`]).
224pub(crate) fn merge_overlapping_regulars(regions: &mut Vec<Region>) {
225 let idx: Vec<usize> = (0..regions.len())
226 .filter(|&i| regions[i].label != "picture" && !is_wrapper(regions[i].label))
227 .collect();
228 if idx.len() < 2 {
229 return;
230 }
231 let mut parent: Vec<usize> = (0..idx.len()).collect();
232 fn find(parent: &mut [usize], i: usize) -> usize {
233 let mut root = i;
234 while parent[root] != root {
235 root = parent[root];
236 }
237 let mut cur = i;
238 while parent[cur] != root {
239 let next = parent[cur];
240 parent[cur] = root;
241 cur = next;
242 }
243 root
244 }
245 for a in 0..idx.len() {
246 for b in (a + 1)..idx.len() {
247 let (ra, rb) = (®ions[idx[a]], ®ions[idx[b]]);
248 let i = inter(ra, rb.l, rb.t, rb.r, rb.b);
249 let aa = area(ra.l, ra.t, ra.r, ra.b).max(f32::EPSILON);
250 let ba = area(rb.l, rb.t, rb.r, rb.b).max(f32::EPSILON);
251 if i / (aa + ba - i).max(f32::EPSILON) > 0.8 || i / aa > 0.8 || i / ba > 0.8 {
252 let (pa, pb) = (find(&mut parent, a), find(&mut parent, b));
253 if pa != pb {
254 parent[pa] = pb;
255 }
256 }
257 }
258 }
259 let mut groups: std::collections::BTreeMap<usize, Vec<usize>> =
260 std::collections::BTreeMap::new();
261 for i in 0..idx.len() {
262 let root = find(&mut parent, i);
263 groups.entry(root).or_default().push(i);
264 }
265 const AREA_THRESHOLD: f32 = 1.3;
266 const CONF_THRESHOLD: f32 = 0.05;
267 let area_of = |i: usize| {
268 let r = ®ions[idx[i]];
269 area(r.l, r.t, r.r, r.b).max(f32::EPSILON)
270 };
271 // `_should_prefer_cluster(candidate, other)` with the regular params.
272 let prefer = |cand: usize, other: usize| -> bool {
273 let (c, o) = (®ions[idx[cand]], ®ions[idx[other]]);
274 let area_ratio = area_of(cand) / area_of(other);
275 if c.label == "list_item" && o.label == "text" && (1.0 - area_ratio).abs() < 0.2 {
276 return true;
277 }
278 if c.label == "code" && inter(o, c.l, c.t, c.r, c.b) / area_of(other) > 0.8 {
279 return true;
280 }
281 !(area_ratio <= AREA_THRESHOLD && o.score - c.score > CONF_THRESHOLD)
282 };
283 let mut drop = vec![false; regions.len()];
284 let mut unions: Vec<(usize, (f32, f32, f32, f32))> = Vec::new();
285 for group in groups.values() {
286 if group.len() < 2 {
287 continue;
288 }
289 let mut best: Option<usize> = None;
290 for &cand in group {
291 if group
292 .iter()
293 .all(|&other| other == cand || prefer(cand, other))
294 {
295 best = Some(match best {
296 None => cand,
297 Some(cur)
298 if area_of(cand) > area_of(cur)
299 && regions[idx[cur]].score - regions[idx[cand]].score
300 <= CONF_THRESHOLD =>
301 {
302 cand
303 }
304 Some(cur) => cur,
305 });
306 }
307 }
308 // docling falls back to the group's first cluster; the highest score
309 // is the deterministic equivalent for a set with no insertion order.
310 let keep = best.unwrap_or_else(|| {
311 *group
312 .iter()
313 .max_by(|&&a, &&b| regions[idx[a]].score.total_cmp(®ions[idx[b]].score))
314 .expect("non-empty group")
315 });
316 let mut u = (
317 f32::INFINITY,
318 f32::INFINITY,
319 f32::NEG_INFINITY,
320 f32::NEG_INFINITY,
321 );
322 for &i in group {
323 let r = ®ions[idx[i]];
324 u = (u.0.min(r.l), u.1.min(r.t), u.2.max(r.r), u.3.max(r.b));
325 if i != keep {
326 drop[idx[i]] = true;
327 }
328 }
329 unions.push((idx[keep], u));
330 }
331 for (i, (l, t, r, b)) in unions {
332 let k = &mut regions[i];
333 (k.l, k.t, k.r, k.b) = (l, t, r, b);
334 }
335 let mut keep_iter = drop.into_iter();
336 regions.retain(|_| !keep_iter.next().expect("aligned"));
337}
338
339/// `intersection_over_union` of two regions.
340fn iou(a: &Region, b: &Region) -> f32 {
341 let i = inter(a, b.l, b.t, b.r, b.b);
342 let u = area(a.l, a.t, a.r, a.b) + area(b.l, b.t, b.r, b.b) - i;
343 if u > 0.0 {
344 i / u
345 } else {
346 0.0
347 }
348}
349
350/// docling's `_resolve_coincident_pairs` (#4059, 2.122): for every (loser,
351/// winner) pair at a near-identical box (IoU > 0.8) whose confidences are
352/// within 0.1 (`loser.score - winner.score < 0.1`), the loser label is dropped
353/// so the label with the richer downstream semantic survives. Nothing else —
354/// containment, area — is considered; a clearly more confident loser stays.
355fn coincident_losers(regions: &[Region], losers: &[usize], winners: &[usize]) -> Vec<usize> {
356 let mut out = Vec::new();
357 for &li in losers {
358 for &wi in winners {
359 if iou(®ions[li], ®ions[wi]) > 0.8 && regions[li].score - regions[wi].score < 0.1
360 {
361 out.push(li);
362 break;
363 }
364 }
365 }
366 out
367}
368
369/// docling's `_handle_cross_type_overlaps` (2.122/2.123 shape): the layout
370/// model can emit one grounded region under several labels, and the picture /
371/// table / container buckets are de-overlapped independently, so such a region
372/// survives twice. Elect a winner for the near-identical pairs:
373///
374/// | pair | loser | winner |
375/// |---------------------------------------|-----------|---------------------|
376/// | TABLE vs DOCUMENT_INDEX | table | document_index |
377/// | PICTURE vs TABLE / DOCUMENT_INDEX | picture | the table-like |
378/// | FORM / KEY_VALUE_REGION vs *surviving* TABLE / DOCUMENT_INDEX / PICTURE | container | structured element |
379///
380/// IoU (not containment) so a genuine small figure inside a large table region
381/// is not removed; the confidence tolerance keeps a clearly more confident
382/// loser (an earlier port dropped every coincident picture regardless).
383fn handle_cross_type_overlaps(regions: Vec<Region>) -> Vec<Region> {
384 let by = |pred: &dyn Fn(&str) -> bool| -> Vec<usize> {
385 (0..regions.len())
386 .filter(|&i| pred(regions[i].label))
387 .collect()
388 };
389 let tables = by(&|l| l == "table");
390 let doc_indices = by(&|l| l == "document_index");
391 let pictures = by(&|l| l == "picture");
392 let containers = by(&|l| matches!(l, "form" | "key_value_region"));
393 let mut drop = vec![false; regions.len()];
394 for i in coincident_losers(®ions, &tables, &doc_indices) {
395 drop[i] = true;
396 }
397 let table_like: Vec<usize> = tables.iter().chain(&doc_indices).copied().collect();
398 for i in coincident_losers(®ions, &pictures, &table_like) {
399 drop[i] = true;
400 }
401 let structured: Vec<usize> = table_like
402 .iter()
403 .chain(&pictures)
404 .copied()
405 .filter(|&i| !drop[i])
406 .collect();
407 for i in coincident_losers(®ions, &containers, &structured) {
408 drop[i] = true;
409 }
410 let mut drop = drop.into_iter();
411 let mut regions = regions;
412 regions.retain(|_| !drop.next().expect("aligned"));
413 regions
414}
415
416pub fn resolve(regions: Vec<Region>) -> Vec<Region> {
417 let regions = handle_cross_type_overlaps(regions);
418 // De-overlap each bucket on its own.
419 let pictures = greedy(
420 regions
421 .iter()
422 .filter(|r| r.label == "picture")
423 .cloned()
424 .collect(),
425 );
426 // Tables and containers are separate buckets since docling 2.123
427 // (`TABLE_TYPES` vs `CONTAINER_TYPES`, docling#4064): a form drawn around a
428 // table no longer competes with it for survival — the table nests inside
429 // the container instead (`order_with_containers`).
430 let mut tables = greedy(
431 regions
432 .iter()
433 .filter(|r| is_table_like(r.label))
434 .cloned()
435 .collect(),
436 );
437 // `greedy` only drops a table mostly inside a *more* confident one, so a
438 // low-score whole-page table proposed over the column tables it contains
439 // (a two-column glossary page: 0.53 over 0.71/0.67/0.66) survived next to them and every
440 // cell was emitted twice. docling's `_remove_overlapping_clusters(tables,
441 // "wrapper")` groups tables whose boxes overlap (IoU > 0.8, or either one
442 // > 80 % inside the other) and keeps one per group: run it on what
443 // `greedy` leaves, like `merge_overlapping_regulars` does for regulars.
444 remove_overlapping_specials(&mut tables, |_| true, 2.0, 0.2);
445 let containers = greedy(
446 regions
447 .iter()
448 .filter(|r| matches!(r.label, "form" | "key_value_region"))
449 .cloned()
450 .collect(),
451 );
452 let mut kept = greedy(
453 regions
454 .iter()
455 .filter(|r| r.label != "picture" && !is_wrapper(r.label))
456 .cloned()
457 .collect(),
458 );
459 dedup_nested_code(&mut kept);
460 kept.extend(pictures);
461 kept.extend(tables);
462 kept.extend(containers);
463 kept
464}
465
466/// Drop a regular region that is >80% contained in a surviving special region we
467/// render **as a single unit** — a table/table-of-contents index — ported from
468/// docling's "Remove regular clusters that are included in wrappers" step: the
469/// special absorbs it as a child (a table cell), so it must not also be emitted
470/// as its own paragraph/list-item. This stops the survey list-items from
471/// appearing both inside the detected table and again as bullets
472/// (`table_mislabeled_as_picture`).
473///
474/// `picture` regions are **not** in the swallow set: docling keeps a picture's
475/// contained clusters as the `PictureItem`'s *children* in the document JSON
476/// (`_set_cluster_children`, `ReadingOrderModel._add_child_elements`), while
477/// its `MarkdownPictureSerializer` prints only the caption and the image — the
478/// children never reach the Markdown (verified against the corpus groundtruth:
479/// `amt_handbook`'s in-figure callout labels are absent). So the regulars
480/// inside a picture stay in the region list, claim their cells and are fitted
481/// like any regular, and [`assemble_page`] lifts them out of the page's
482/// reading order into a [`Node::PictureChildren`] after their picture (see
483/// [`picture_parents`]). A line only partially under a figure box (straddling
484/// its border, ≤80 % contained) is no picture's child and is emitted, as
485/// docling emits it (#165).
486///
487/// `form` / `key_value_region` wrappers are deliberately **excluded**: this
488/// pipeline does not render them as a structured block (they are skipped), so
489/// their textual content comes precisely from the contained regular regions —
490/// A `page_footer` that is really the body of the last heading on the page
491/// becomes `text`.
492///
493/// The layout model labels the bottom margin by position as much as by
494/// content: a one-line paragraph that happens to sit where a running footer
495/// would — a CV's `Languages` line under its `## Languages` heading, the last
496/// entry of a section that runs to the page edge — comes back as
497/// `page_footer` (0.88 against 0.49 for `text` on the reporting file, with
498/// the fp32 model and either renderer alike), and Markdown drops furniture,
499/// so the document loses real content while the JSON keeps it under the
500/// wrong label. docling fails the same way. Deliberate deviation, gated so a
501/// real footer keeps its label: the nearest `section_header` above the footer
502/// must overlap it horizontally, end within 2.5 footer heights of its top,
503/// and be the **last body element** of its column — no other non-furniture
504/// region starts at or below the heading's bottom edge over the same span —
505/// and the footer must be a line of text, not a page number: at least 40 %
506/// of the page width. A running footer never follows a heading that has no
507/// body of its own, so the combination is the misread heading body.
508pub fn reclaim_heading_body_footers(regions: &mut [Region], page_w: f32) {
509 let n = regions.len();
510 let overlap_x = |a: &Region, b: &Region| a.r.min(b.r) - a.l.max(b.l) > 0.0;
511 for fi in 0..n {
512 let f = regions[fi].clone();
513 if f.label != "page_footer" || f.r - f.l < 0.4 * page_w {
514 continue;
515 }
516 let fh = (f.b - f.t).max(1.0);
517 // The nearest heading above the footer, over the footer's span.
518 let heading = (0..n)
519 .filter(|&j| {
520 let h = ®ions[j];
521 j != fi && h.label == "section_header" && h.b <= f.t + 0.5 * fh && overlap_x(h, &f)
522 })
523 .min_by(|&a, &b| regions[b].b.total_cmp(®ions[a].b));
524 let Some(hi) = heading else {
525 continue;
526 };
527 let h = regions[hi].clone();
528 if f.t - h.b > 2.5 * fh {
529 continue;
530 }
531 // The heading must have no body of its own: nothing but the footer
532 // starts at or below its bottom edge over the heading's or footer's
533 // span (a heading whose paragraph follows is not this case, and a
534 // heading with the footer far below it was filtered above).
535 let has_body = (0..n).any(|j| {
536 let r = ®ions[j];
537 j != fi
538 && j != hi
539 && !matches!(r.label, "page_footer" | "page_header")
540 && r.t >= h.b - 0.5 * fh
541 && (overlap_x(r, &h) || overlap_x(r, &f))
542 });
543 if has_body {
544 continue;
545 }
546 regions[fi].label = "text";
547 }
548}
549
550/// dropping those would erase the page (e.g. `right_to_left_03`'s form-heavy
551/// pages). Runs *after* [`drop_false_pictures`] so a phantom picture can't
552/// swallow real text on its way out.
553pub fn drop_contained_regulars(regions: &mut Vec<Region>) {
554 let specials: Vec<(f32, f32, f32, f32)> = regions
555 .iter()
556 .filter(|r| is_table_like(r.label))
557 .map(|r| (r.l, r.t, r.r, r.b))
558 .collect();
559 if specials.is_empty() {
560 return;
561 }
562 regions.retain(|r| {
563 if r.label == "picture" || is_wrapper(r.label) {
564 return true;
565 }
566 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
567 !specials
568 .iter()
569 .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.8)
570 });
571}
572
573/// docling's `_set_cluster_children` for pictures: a regular region (one that
574/// claims cells) whose box is > 80 % inside a picture's box
575/// (`intersection_over_self`) is that picture's child — not a page element
576/// (it leaves the reading order and the layout score), but an item under the
577/// `PictureItem` in the JSON. Returns, per region, the index of its parent
578/// picture: the smallest containing one when pictures nest (upstream would
579/// attach it to each).
580pub fn picture_parents(regions: &[Region]) -> Vec<Option<usize>> {
581 regions
582 .iter()
583 .map(|r| {
584 if !claims_cells(r) {
585 return None;
586 }
587 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
588 regions
589 .iter()
590 .enumerate()
591 .filter(|(_, p)| p.label == "picture" && inter(r, p.l, p.t, p.r, p.b) / ra > 0.8)
592 .min_by(|(_, a), (_, b)| {
593 area(a.l, a.t, a.r, a.b).total_cmp(&area(b.l, b.t, b.r, b.b))
594 })
595 .map(|(i, _)| i)
596 })
597 .collect()
598}
599
600/// True for a bare, single-token source-code language label (`XML`, `C#`, `JSON`,
601/// `bash`, …) — the little header the docs render above a code block. Matched
602/// case-insensitively; anything with whitespace or longer than a token is out.
603fn is_code_language(t: &str) -> bool {
604 let t = t.trim();
605 if t.is_empty() || t.chars().any(char::is_whitespace) || t.chars().count() > 12 {
606 return false;
607 }
608 const LANGS: &[&str] = &[
609 "xml",
610 "html",
611 "xhtml",
612 "json",
613 "jsonc",
614 "yaml",
615 "yml",
616 "toml",
617 "ini",
618 "c#",
619 "csharp",
620 "f#",
621 "fsharp",
622 "vb",
623 "c",
624 "c++",
625 "cpp",
626 "java",
627 "kotlin",
628 "scala",
629 "go",
630 "golang",
631 "rust",
632 "swift",
633 "javascript",
634 "js",
635 "typescript",
636 "ts",
637 "jsx",
638 "tsx",
639 "python",
640 "py",
641 "ruby",
642 "rb",
643 "php",
644 "perl",
645 "lua",
646 "r",
647 "dart",
648 "bash",
649 "sh",
650 "shell",
651 "powershell",
652 "zsh",
653 "batch",
654 "cmd",
655 "sql",
656 "tsql",
657 "plsql",
658 "graphql",
659 "dockerfile",
660 "makefile",
661 "css",
662 "scss",
663 "sass",
664 "less",
665 "markdown",
666 "md",
667 "tex",
668 "latex",
669 "diff",
670 "proto",
671 "razor",
672 "cshtml",
673 "xaml",
674 "aspx",
675 "http",
676 ];
677 let lower = t.to_ascii_lowercase();
678 LANGS.contains(&lower.as_str())
679}
680
681/// Mark the region indices that are a code block's **language label** — a bare
682/// `XML`/`C#`/… token sitting directly above a `code` region — so they are consumed
683/// rather than emitted as their own stray paragraph/heading. The label may also be
684/// captured inside a wider code box (rendered as the fence's first line); dropping
685/// the standalone copy just removes the duplicate.
686fn code_language_labels(regions: &[Region], cells: &[TextCell]) -> Vec<bool> {
687 let mut drop = vec![false; regions.len()];
688 for (i, r) in regions.iter().enumerate() {
689 if matches!(r.label, "code" | "picture" | "table") {
690 continue;
691 }
692 if !is_code_language(®ion_text(r, cells)) {
693 continue;
694 }
695 // The label sits just above the code (a blank line's gap) or is swallowed
696 // into the top of a wider code box; either way it is that block's label.
697 // The window is generous because the label's own font is small, so a
698 // one-line gap is several times its height.
699 let line_h = (r.b - r.t).abs().max(1.0);
700 let window = (line_h * 4.0).max(28.0);
701 let labels_code = regions.iter().enumerate().any(|(j, c)| {
702 if j == i || c.label != "code" {
703 return false;
704 }
705 let gap = c.t - r.b; // >0 when the code is below the label
706 let h_overlap = (r.r.min(c.r) - r.l.max(c.l)).max(0.0);
707 gap > -line_h * 3.0 && gap < window && h_overlap > 0.0
708 });
709 if labels_code {
710 drop[i] = true;
711 }
712 }
713 drop
714}
715
716/// Collapse `code` regions where one is nested inside another, keeping the larger.
717///
718/// RT-DETR sometimes emits a tight code box *and* a wider near-duplicate that also
719/// captures the block's language label (`XML`, `C#`, …). When the tight box scores
720/// higher it is kept first, and the wider container — not "mostly inside" the tight
721/// box — survives [`resolve`]'s greedy pass, so the block is emitted twice. Keeping
722/// the **larger** box (rather than dropping it) collapses the pair without leaking
723/// the container's extra cells back out as orphan text, since the larger box still
724/// covers every cell. Restricted to `code` so genuinely distinct nested regions of
725/// other kinds are untouched.
726fn dedup_nested_code(kept: &mut Vec<Region>) {
727 let mut drop = vec![false; kept.len()];
728 for i in 0..kept.len() {
729 if kept[i].label != "code" {
730 continue;
731 }
732 let ai = area(kept[i].l, kept[i].t, kept[i].r, kept[i].b).max(1.0);
733 for j in 0..kept.len() {
734 if i == j || drop[j] || kept[j].label != "code" {
735 continue;
736 }
737 let aj = area(kept[j].l, kept[j].t, kept[j].r, kept[j].b).max(1.0);
738 // Drop i when it is mostly inside a strictly larger code box j.
739 let overlap = inter(&kept[i], kept[j].l, kept[j].t, kept[j].r, kept[j].b);
740 if aj > ai && overlap / ai > 0.7 {
741 drop[i] = true;
742 break;
743 }
744 }
745 }
746 let mut keep = drop.iter();
747 kept.retain(|_| !*keep.next().unwrap());
748}
749
750/// Fraction of the page's non-empty text cells that some detected region
751/// claims (>0.2 intersection-over-self, docling's assignment rule). 1.0 for a
752/// page without text cells.
753///
754/// The int8-layout guard keys off this: a dense digital page whose detections
755/// cover almost none of its text is the signature of quantized confidences
756/// flipping under the 0.5 label thresholds on this CPU's kernels — not of a
757/// genuinely empty layout — and is worth re-running on the fp32 graph.
758pub fn layout_cell_coverage(regions: &[Region], cells: &[TextCell]) -> f32 {
759 let mut total = 0usize;
760 let mut covered = 0usize;
761 for c in cells {
762 if c.text.trim().is_empty() {
763 continue;
764 }
765 total += 1;
766 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
767 if regions
768 .iter()
769 .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
770 {
771 covered += 1;
772 }
773 }
774 if total == 0 {
775 1.0
776 } else {
777 covered as f32 / total as f32
778 }
779}
780
781/// Append `text` regions for cells the layout left uncovered ("orphan cells"),
782/// the way docling's `LayoutPostprocessor` does (`create_orphan_clusters`): any
783/// non-empty cell that no kept region covers (>50% of the cell's area) becomes a
784/// text region of its own, so text the detector missed (a stray `.`, a small
785/// label) is still emitted instead of silently dropped. Adjacent orphan cells on a
786/// line are merged so a missed paragraph doesn't shatter into one block per line.
787pub fn add_orphan_regions(regions: &mut Vec<Region>, cells: &[TextCell]) {
788 // docling assigns each cell to its single best-overlapping cluster at
789 // intersection-over-self > 0.2 and serializes exactly the assigned cells —
790 // and since [`region_texts_exclusive`] now emits under that very rule, the
791 // claim test here matches it: any cell over 0.2 will actually render in
792 // its best region, everything else becomes an orphan. Completeness by
793 // construction, with no (0.2, 0.5] hole (the old > 0.5 serializer needed
794 // the claim test raised to > 0.5 to keep right_to_left_03's `20300` from
795 // vanishing; the exclusive port closes that structurally).
796 //
797 // Only *regular* clusters claim cells: docling's `_find_unassigned_cells`
798 // walks `regular_clusters` alone, so a cell under a `picture` or a wrapper
799 // (`table`/`document_index`/`form`/`key_value_region`) that no regular
800 // cluster covers still becomes an orphan text cluster (#165). The orphans
801 // that end up *fully* inside the special are re-dropped by
802 // [`drop_contained_regulars`] (docling's Markdown drops them the same way
803 // — a picture's children never reach its `MarkdownPictureSerializer`
804 // output, a table's text renders through the reconstructed grid). The
805 // observable fix is the border-straddlers: a line only partially under a
806 // figure box used to lose its cells to the picture's 0.2 claim and vanish
807 // — now it forms an orphan region and is emitted, as docling does.
808 let assigned = |c: &TextCell| {
809 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
810 regions
811 .iter()
812 .filter(|r| r.label != "picture" && !is_wrapper(r.label))
813 .any(|r| inter(r, c.l, c.t, c.r, c.b) / ca > 0.2)
814 };
815 // Collect orphan cells (non-empty, unassigned), in page order.
816 let mut orphans: Vec<&TextCell> = cells
817 .iter()
818 .filter(|c| !c.text.trim().is_empty() && !assigned(c))
819 .collect();
820 if orphans.is_empty() {
821 return;
822 }
823 orphans.sort_by(|a, b| a.t.total_cmp(&b.t).then(a.l.total_cmp(&b.l)));
824 // Merge cells that sit on the same line and nearly touch into one region, so a
825 // dropped multi-word line stays one block (docling's refinement merges these).
826 let mut merged: Vec<Region> = Vec::new();
827 for c in orphans {
828 let h = (c.b - c.t).abs().max(1.0);
829 if let Some(last) = merged.last_mut() {
830 let same_line = (last.t - c.t).abs() < h * 0.5;
831 let touching = c.l <= last.r + h && c.l >= last.l - h;
832 // Both tolerances scale with the cell's own height, so a run set
833 // vertically (the arXiv stamp up the margin, #528 — a cell a few
834 // points wide and hundreds tall) would read as "on the line" of
835 // whatever precedes it and glue a whole column into one region.
836 // Lines of one row differ by a drop cap's few multiples at most.
837 let lh = (last.b - last.t).abs().max(1.0);
838 let comparable = h <= 4.0 * lh && lh <= 4.0 * h;
839 if same_line && touching && comparable {
840 last.l = last.l.min(c.l);
841 last.r = last.r.max(c.r);
842 last.t = last.t.min(c.t);
843 last.b = last.b.max(c.b);
844 continue;
845 }
846 }
847 merged.push(Region {
848 label: "text",
849 score: 0.0,
850 l: c.l,
851 t: c.t,
852 r: c.r,
853 b: c.b,
854 });
855 }
856 regions.extend(merged);
857}
858
859/// Demote a `picture` region that is really a **text panel** — a paragraph block
860/// the layout model boxed as a figure because it is typeset on a colored
861/// background (terms-and-conditions callouts, quote boxes) — into ordinary
862/// `text` regions, one per paragraph, so its words are read instead of shipped
863/// as pixels. docling loses this text the same way (cells assigned to a picture
864/// cluster are never serialized); this is a deliberate improvement, not parity.
865///
866/// The gate is conservative so a genuine figure keeps its crop: the region must
867/// contain at least three text lines whose median width spans most of the panel
868/// (axis labels and chat bubbles are narrow and varied) and whose cells cover a
869/// substantial fraction of its area (a photo or chart with sparse labels does
870/// not). Paragraph boundaries are re-derived from the line pitch: a vertical gap
871/// clearly larger than the panel's own leading starts a new `text` region, so
872/// the panel doesn't collapse into one giant paragraph.
873///
874/// Works on any cell source — the digital text layer or OCR lines recognized
875/// from the picture crop — so the native and browser paths, with or without
876/// force-OCR, demote identically.
877pub fn recover_text_panels(regions: &mut Vec<Region>, cells: &[TextCell]) {
878 // A *captioned* picture is a genuine figure whatever it contains — the
879 // corpus is full of document screenshots ("Figure 3: …" above a page
880 // image) that are exactly as dense and wide as a text panel. Only an
881 // uncaptioned picture is a demotion candidate.
882 let captioned: Vec<bool> = regions
883 .iter()
884 .map(|r| {
885 r.label == "picture"
886 && regions.iter().any(|c| {
887 c.label == "caption" && c.r.min(r.r) - c.l.max(r.l) > 0.0 && {
888 let gap = if c.t >= r.b {
889 c.t - r.b
890 } else if r.t >= c.b {
891 r.t - c.b
892 } else {
893 f32::MAX // vertically overlapping: not a caption
894 };
895 gap <= 25.0
896 }
897 })
898 })
899 .collect();
900 let mut out: Vec<Region> = Vec::with_capacity(regions.len());
901 // Synthesized paragraphs and the demoted panels' boxes are kept separate
902 // from `out` until the end: the dedup filter below must not confuse a
903 // paragraph we just built with a pre-existing region inside the panel.
904 let mut demoted_paras: Vec<Region> = Vec::new();
905 let mut demoted_boxes: Vec<(f32, f32, f32, f32)> = Vec::new();
906 for (i, r) in regions.drain(..).enumerate() {
907 if r.label != "picture" || captioned[i] {
908 out.push(r);
909 continue;
910 }
911 let inside: Vec<&TextCell> = cells
912 .iter()
913 .filter(|c| {
914 !c.text.trim().is_empty() && {
915 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
916 inter(&r, c.l, c.t, c.r, c.b) / ca > 0.5
917 }
918 })
919 .collect();
920 // Group the contained cells into lines by vertical overlap (the same
921 // rule region_text orders by), tracking each line's union box.
922 let mut lines: Vec<(f32, f32, f32, f32)> = Vec::new(); // (t, b, l, r)
923 for c in &inside {
924 let (ct, cb) = (c.t.min(c.b), c.t.max(c.b));
925 match lines.iter_mut().find(|(lt, lb, _, _)| {
926 let ov = cb.min(*lb) - ct.max(*lt);
927 ov > 0.5 * (cb - ct).min(*lb - *lt).max(1.0)
928 }) {
929 Some((lt, lb, ll, lr)) => {
930 *lt = lt.min(ct);
931 *lb = lb.max(cb);
932 *ll = ll.min(c.l);
933 *lr = lr.max(c.r);
934 }
935 None => lines.push((ct, cb, c.l, c.r)),
936 }
937 }
938 if lines.len() < 3 {
939 out.push(r);
940 continue;
941 }
942 let panel_w = (r.r - r.l).max(1.0);
943 let coverage = inside.iter().map(|c| area(c.l, c.t, c.r, c.b)).sum::<f32>()
944 / area(r.l, r.t, r.r, r.b).max(1.0);
945 let mut widths: Vec<f32> = lines.iter().map(|(_, _, l, rr)| rr - l).collect();
946 widths.sort_by(f32::total_cmp);
947 // A figure's text is ragged: a title line, small axis/tick labels, and
948 // OCR boxes over the plot area come out at wildly different heights,
949 // whereas a real text panel is set in one face with constant leading.
950 // Require near-uniform line heights (median absolute deviation ≤ 35%
951 // of the median) so an uncaptioned chart keeps its crop even when its
952 // labels are dense enough to pass the coverage gate (#173) — garbled
953 // OCR of its bars is not content.
954 let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
955 heights.sort_by(f32::total_cmp);
956 let h_med = heights[heights.len() / 2].max(1.0);
957 let mut devs: Vec<f32> = heights.iter().map(|h| (h - h_med).abs()).collect();
958 devs.sort_by(f32::total_cmp);
959 let uniform = devs[devs.len() / 2] <= 0.35 * h_med;
960 let text_panel = coverage >= 0.2 && widths[widths.len() / 2] >= 0.45 * panel_w && uniform;
961 if !text_panel {
962 out.push(r);
963 continue;
964 }
965 lines.sort_by(|a, b| a.0.total_cmp(&b.0));
966 let mut heights: Vec<f32> = lines.iter().map(|(t, b, _, _)| b - t).collect();
967 heights.sort_by(f32::total_cmp);
968 let h = heights[heights.len() / 2].max(1.0);
969 let mut gaps: Vec<f32> = lines
970 .windows(2)
971 .map(|w| (w[1].0 - w[0].1).max(0.0))
972 .collect();
973 gaps.sort_by(f32::total_cmp);
974 let leading = if gaps.is_empty() {
975 0.0
976 } else {
977 gaps[gaps.len() / 2]
978 };
979 let brk = (1.8 * leading).max(0.75 * h);
980 let mut para: Option<(f32, f32, f32, f32)> = None; // (l, t, r, b) union
981 for (t, b, l, rr) in &lines {
982 match &mut para {
983 Some((pl, _, pr, pb)) if *t - *pb <= brk => {
984 *pl = pl.min(*l);
985 *pr = pr.max(*rr);
986 *pb = pb.max(*b);
987 }
988 _ => {
989 if let Some((pl, pt, pr, pb)) = para.take() {
990 demoted_paras.push(Region {
991 label: "text",
992 score: r.score,
993 l: pl,
994 t: pt,
995 r: pr,
996 b: pb,
997 });
998 }
999 para = Some((*l, *t, *rr, *b));
1000 }
1001 }
1002 }
1003 if let Some((pl, pt, pr, pb)) = para {
1004 demoted_paras.push(Region {
1005 label: "text",
1006 score: r.score,
1007 l: pl,
1008 t: pt,
1009 r: pr,
1010 b: pb,
1011 });
1012 }
1013 demoted_boxes.push((r.l, r.t, r.r, r.b));
1014 }
1015 // The paragraphs are rebuilt from *all* of the panel's cells, so any
1016 // surviving text region inside a demoted panel (an orphan cluster or a
1017 // layout-detected fragment — pictures no longer swallow them, #165) would
1018 // say the same words twice. Consume those; wrappers and pictures stay.
1019 if !demoted_boxes.is_empty() {
1020 out.retain(|r| {
1021 r.label == "picture" || is_wrapper(r.label) || {
1022 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
1023 !demoted_boxes
1024 .iter()
1025 .any(|&(l, t, rr, b)| inter(r, l, t, rr, b) / ra > 0.5)
1026 }
1027 });
1028 }
1029 // docling's "Remove regular clusters that are included in wrappers" (a
1030 // regular > 80 % inside a table is absorbed by it) already ran as
1031 // [`drop_contained_regulars`], but before this demotion created new
1032 // regulars. Apply it to them too: a panel that coincides with a table (a
1033 // dense data table detected as picture 0.80 and table 0.62 on one box;
1034 // `_handle_cross_type_overlaps` keeps both once the picture is ≥ 0.1 more
1035 // confident) rebuilds the table's words as a paragraph the grid already
1036 // renders. A panel inside another picture is left as it was.
1037 demoted_paras.retain(|p| {
1038 let pa = area(p.l, p.t, p.r, p.b).max(1.0);
1039 !out.iter()
1040 .any(|s| is_table_like(s.label) && inter(p, s.l, s.t, s.r, s.b) / pa > 0.8)
1041 });
1042 out.extend(demoted_paras);
1043 *regions = out;
1044}
1045
1046/// Drop a `picture` detection covering more than 90 % of the page — docling's
1047/// `LayoutPostprocessor._process_special_clusters` "Filter out full-page
1048/// pictures" (upstream since 2.15), applied to the thresholded detections
1049/// before overlap resolution. A box that big is the page itself, not a figure
1050/// on it: the layout model emits one for a whole-page diagram (a LaTeX figure
1051/// PDF cropped to its drawing), a plate, or a scan, and keeping it would swallow
1052/// every text cell on the page as picture children — the diagram's labels and
1053/// caption vanish behind a lone `<!-- image -->`, where docling reads them out as
1054/// text. `page_w`/`page_h` is the display-frame page box.
1055pub fn drop_full_page_pictures(regions: &mut Vec<Region>, page_w: f32, page_h: f32) {
1056 let page_area = (page_w * page_h).max(1.0);
1057 regions.retain(|r| r.label != "picture" || area(r.l, r.t, r.r, r.b) / page_area <= 0.90);
1058}
1059
1060/// Drop a `picture` detection that is a small, empty, low-confidence margin box on
1061/// a **text page** — a false positive the RT-DETR layout sometimes emits (e.g.
1062/// `right_to_left_02`'s phantom right-column picture, score 0.40); docling does not
1063/// emit it. The gate is deliberately narrow so a genuine figure is never dropped:
1064/// (1) only on pages with a digital text layer — image/scanned/figure pages have
1065/// no `cells` yet at this point (OCR runs later), so their pictures, which *are*
1066/// the content, are kept; (2) only a box covering < 25 % of the page (a margin
1067/// artifact, not a dominant figure); (3) only when it contains no text and scores
1068/// below 0.5 (real empty figures in the corpus all score ≥ 0.86).
1069pub fn drop_false_pictures(
1070 regions: &mut Vec<Region>,
1071 cells: &[TextCell],
1072 page_w: f32,
1073 page_h: f32,
1074) {
1075 if cells.iter().all(|c| c.text.trim().is_empty()) {
1076 return; // no digital text layer (image/scanned page) — keep all pictures
1077 }
1078 // A text-document page carries several text-bearing non-picture regions (so a
1079 // spurious margin picture is clearly extra). A slide / figure page has at most
1080 // one — there the picture is the content, so never drop it.
1081 let content_regions = regions
1082 .iter()
1083 .filter(|r| r.label != "picture" && !region_text(r, cells).trim().is_empty())
1084 .count();
1085 if content_regions < 2 {
1086 return;
1087 }
1088 let page_area = (page_w * page_h).max(1.0);
1089 regions.retain(|r| {
1090 if r.label != "picture" || r.score >= 0.5 {
1091 return true;
1092 }
1093 if area(r.l, r.t, r.r, r.b) / page_area >= 0.25 {
1094 return true; // a dominant figure, not a margin artifact
1095 }
1096 // Keep it if any text cell falls mostly inside (a real captioned/labelled
1097 // figure); drop only the genuinely empty low-confidence boxes.
1098 cells.iter().any(|c| {
1099 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1100 !c.text.trim().is_empty() && inter(r, c.l, c.t, c.r, c.b) / ca > 0.5
1101 })
1102 });
1103}
1104
1105/// A small digit-only region in the top/bottom margin: a page number. docling
1106/// emits `right_to_left_02`'s bottom `11` as the page's *first* text item (its
1107/// reading-order model floats the page number to the front), whereas our
1108/// position-based ordering would place a bottom region last.
1109fn is_page_number(region: &Region, cells: &[TextCell], page_h: f32) -> bool {
1110 let t = region_text(region, cells);
1111 let t = t.trim();
1112 !t.is_empty()
1113 && t.chars().all(|c| c.is_ascii_digit())
1114 && (region.b - region.t).abs() < 30.0
1115 && (region.t < page_h * 0.12 || region.b > page_h * 0.88)
1116}
1117
1118/// docling's `form` / `key_value_region` *containers* (2.123, docling#4064):
1119/// every region sitting > 0.8 inside one — text, list items, and since #4064
1120/// tables and pictures too — is that container's child. Children are
1121/// reading-ordered among themselves and emitted as one block where the
1122/// container falls in the page's top-level order (a `form_area` /
1123/// `key_value_area` group upstream), instead of interleaving with the text
1124/// around the form. A child inside several containers belongs to the smallest
1125/// (then most confident, then first); a container with children shrinks to
1126/// their union for the top-level ordering, like upstream's bbox adjustment.
1127///
1128/// The containers themselves are still not emitted (`is_skipped`), so the
1129/// Markdown is exactly upstream's — a group prints only its children.
1130///
1131/// `cids` are the items' positions in docling's assembly order
1132/// ([`cluster_cids`]) — the reading-order predictor's same-row rule (#424)
1133/// pairs consecutive ones, within the top level and within each container.
1134fn order_with_containers<T: Clone>(
1135 items: &mut Vec<T>,
1136 cids: &[usize],
1137 page_w: f32,
1138 page_h: f32,
1139 reg: impl Fn(&T) -> &Region,
1140) {
1141 let is_container = |r: &Region| matches!(r.label, "form" | "key_value_region");
1142 let containers: Vec<usize> = (0..items.len())
1143 .filter(|&i| is_container(reg(&items[i])))
1144 .collect();
1145 if containers.is_empty() {
1146 order_regions(items, cids, page_w, page_h, reg);
1147 return;
1148 }
1149 // Parent container per item (containers never nest in each other here —
1150 // upstream assigns regulars and tables/pictures only).
1151 let mut parent: Vec<Option<usize>> = vec![None; items.len()];
1152 for i in 0..items.len() {
1153 let r = reg(&items[i]);
1154 if is_container(r) {
1155 continue;
1156 }
1157 let ra = area(r.l, r.t, r.r, r.b).max(1.0);
1158 let mut best: Option<(usize, f32, f32)> = None; // (idx, area, -score)
1159 for &c in &containers {
1160 let cr = reg(&items[c]);
1161 if inter(r, cr.l, cr.t, cr.r, cr.b) / ra > 0.8 {
1162 let key = (area(cr.l, cr.t, cr.r, cr.b), -cr.score);
1163 if best.is_none_or(|(_, a, s)| key.0 < a || (key.0 == a && key.1 < s)) {
1164 best = Some((c, key.0, key.1));
1165 }
1166 }
1167 }
1168 parent[i] = best.map(|(c, _, _)| c);
1169 }
1170 // Top-level pass: non-children plus the containers, the latter shrunk to
1171 // their children's union.
1172 let mut top: Vec<(usize, Region)> = Vec::new();
1173 for i in 0..items.len() {
1174 if parent[i].is_some() {
1175 continue;
1176 }
1177 let mut r = reg(&items[i]).clone();
1178 if is_container(&r) {
1179 let kids: Vec<&Region> = (0..items.len())
1180 .filter(|&k| parent[k] == Some(i))
1181 .map(|k| reg(&items[k]))
1182 .collect();
1183 if !kids.is_empty() {
1184 r.l = kids.iter().map(|k| k.l).fold(f32::INFINITY, f32::min);
1185 r.t = kids.iter().map(|k| k.t).fold(f32::INFINITY, f32::min);
1186 r.r = kids.iter().map(|k| k.r).fold(f32::NEG_INFINITY, f32::max);
1187 r.b = kids.iter().map(|k| k.b).fold(f32::NEG_INFINITY, f32::max);
1188 }
1189 }
1190 top.push((i, r));
1191 }
1192 let top_cids: Vec<usize> = top.iter().map(|(i, _)| cids[*i]).collect();
1193 order_regions(&mut top, &top_cids, page_w, page_h, |it| &it.1);
1194 let mut out: Vec<T> = Vec::with_capacity(items.len());
1195 for (i, _) in top {
1196 if is_container(reg(&items[i])) {
1197 let kid_idx: Vec<usize> = (0..items.len()).filter(|&k| parent[k] == Some(i)).collect();
1198 let mut kids: Vec<T> = kid_idx.iter().map(|&k| items[k].clone()).collect();
1199 let kid_cids: Vec<usize> = kid_idx.iter().map(|&k| cids[k]).collect();
1200 order_regions(&mut kids, &kid_cids, page_w, page_h, ®);
1201 out.push(items[i].clone());
1202 out.extend(kids);
1203 } else {
1204 out.push(items[i].clone());
1205 }
1206 }
1207 *items = out;
1208}
1209
1210/// Furniture / not-yet-emitted labels.
1211fn is_skipped(label: &str) -> bool {
1212 matches!(
1213 label,
1214 "page_header" | "page_footer" | "form" | "key_value_region"
1215 )
1216}
1217
1218/// Reading-order sort of a page's regions, via the ported rule-based
1219/// [`reading_order`](crate::reading_order) predictor (docling's
1220/// `ReadingOrderPredictor`): an up/down geometry graph with same-row links
1221/// between `cids`-consecutive elements (#424), horizontal dilation and a
1222/// depth-first traversal, with `page_header`/`page_footer` ordered as their own
1223/// groups (first/last) as docling does.
1224fn order_regions<T: Clone>(
1225 items: &mut Vec<T>,
1226 cids: &[usize],
1227 page_w: f32,
1228 page_h: f32,
1229 reg: impl Fn(&T) -> &Region,
1230) {
1231 let boxes: Vec<(f32, f32, f32, f32)> = items
1232 .iter()
1233 .map(|it| {
1234 let r = reg(it);
1235 (r.l, r.t, r.r, r.b)
1236 })
1237 .collect();
1238 let is_header: Vec<bool> = items
1239 .iter()
1240 .map(|it| reg(it).label == "page_header")
1241 .collect();
1242 let is_footer: Vec<bool> = items
1243 .iter()
1244 .map(|it| reg(it).label == "page_footer")
1245 .collect();
1246 let order =
1247 crate::reading_order::order_page(&boxes, cids, &is_header, &is_footer, page_w, page_h);
1248 *items = order.iter().map(|&i| items[i].clone()).collect();
1249}
1250
1251/// docling's assembly order of a page's clusters (`LayoutPostprocessor`'s
1252/// final `_sort_clusters(mode="id")`, #424): each region's rank when sorted by
1253/// its first source cell, then by top edge, then left edge; a region with no
1254/// cells sorts after every one that has some. docling numbers its page
1255/// elements (`cid`) in this order, and the reading-order predictor's same-row
1256/// rule pairs elements with consecutive numbers, so the ranks are what
1257/// [`order_with_containers`] hands the predictor.
1258///
1259/// A regular region's first cell is the smallest index among the cells it
1260/// claims. A table, picture or container has no cells of its own upstream
1261/// either — its cells are its *children's*: the regular clusters > 0.8 inside
1262/// it, and upstream every cell no regular cluster claimed is an orphan cluster
1263/// of its own, so a table's interior text (which no regular cluster claims)
1264/// reaches the table through those orphans. Here that is the cells > 0.8
1265/// inside the region plus the claimed cells of the regular regions > 0.8
1266/// inside it. Without the interior cells every table would sort last, and two
1267/// side-by-side tables would then be consecutive and row-linked — reading the
1268/// right table's caption ahead of the left column's headings (2206 page 8).
1269pub fn cluster_cids(regions: &[Region], cells: &[TextCell]) -> Vec<usize> {
1270 let owned = assign_cells(regions, cells);
1271 let first_cell: Vec<usize> = regions
1272 .iter()
1273 .enumerate()
1274 .map(|(i, r)| {
1275 if claims_cells(r) {
1276 return owned[i].iter().copied().min().unwrap_or(usize::MAX);
1277 }
1278 let interior = cells
1279 .iter()
1280 .enumerate()
1281 .filter(|(_, c)| {
1282 !c.text.trim().is_empty()
1283 && inter(r, c.l, c.t, c.r, c.b) / area(c.l, c.t, c.r, c.b).max(1.0) > 0.8
1284 })
1285 .map(|(ci, _)| ci)
1286 .min();
1287 let children = regions
1288 .iter()
1289 .enumerate()
1290 .filter(|(j, child)| {
1291 *j != i && claims_cells(child) && {
1292 let ca = area(child.l, child.t, child.r, child.b).max(1.0);
1293 inter(r, child.l, child.t, child.r, child.b) / ca > 0.8
1294 }
1295 })
1296 .filter_map(|(j, _)| owned[j].iter().copied().min())
1297 .min();
1298 interior
1299 .into_iter()
1300 .chain(children)
1301 .min()
1302 .unwrap_or(usize::MAX)
1303 })
1304 .collect();
1305 let mut by_source: Vec<usize> = (0..regions.len()).collect();
1306 // Stable, like Python's `sorted`: full ties keep the layout order.
1307 by_source.sort_by(|&a, &b| {
1308 first_cell[a]
1309 .cmp(&first_cell[b])
1310 .then(regions[a].t.total_cmp(®ions[b].t))
1311 .then(regions[a].l.total_cmp(®ions[b].l))
1312 });
1313 let mut cids = vec![0; regions.len()];
1314 for (rank, &i) in by_source.iter().enumerate() {
1315 cids[i] = rank;
1316 }
1317 cids
1318}
1319
1320/// Clean a region's assembled text: undo soft-hyphen line wraps, map curly
1321/// quotes and the ellipsis to ASCII (matching docling), and collapse runs of
1322/// whitespace. pdfium emits the line-wrap hyphen as U+0002 in this corpus
1323/// (U+00AD elsewhere), so `word\u{2} continuation` is one hyphenated word —
1324/// drop the hyphen + the joining space and merge (`com\u{2} pact` → `compact`,
1325/// `end-to\u{2} end` → `end-toend`), exactly as docling does.
1326///
1327/// Token spacing is otherwise left as the geometric join produced it. We do not
1328/// tighten punctuation spacing: docling preserves the PDF's own spaces (it keeps
1329/// `{ ahn }`, `Name 1 .`, `[ 9 ]`), and a geometric gap heuristic diverges from
1330/// it more than a plain single-space join does.
1331/// An ordered-list enumeration marker at the start of a list item: leading ASCII
1332/// digits followed by `.`, e.g. `1. Undo/Redo` → `(1, "Undo/Redo")`. Returns
1333/// `None` when the text doesn't start with `digits.`.
1334/// docling's `ListItemMarkerProcessor` bullet patterns
1335/// (`docling/models/postprocessing/list_marker_processor.py`), one glyph each.
1336const LIST_BULLET_MARKERS: &str = "\u{2022}\u{2023}\u{25E6}\u{2043}\u{204C}\u{204D}\u{2219}\u{25AA}\u{25AB}\u{25CF}\u{25CB}-*+•·‣⁃►▶▸➤➢✓✔✗✘";
1337
1338/// docling's numbered-marker patterns as byte-length scanners over the start
1339/// of the text, in its first-wins order (the compound ones first, as they are
1340/// the more specific). Each returns the marker's candidate lengths, longest
1341/// (greedy) first — the alternatives Python's regex would backtrack through
1342/// before the `\s(.+)` tail. Hand-rolled rather than `regex` because this file
1343/// is part of the `pdf-text` (wasm) build, where the `regex` crate is not.
1344/// `\d` is Unicode-aware in Python, hence `char::is_numeric`; the letters are
1345/// ASCII classes in both.
1346const LIST_NUMBERED_MARKERS: &[fn(&str) -> Vec<usize>] = &[
1347 // `\d+(?:\.\d+)+\.?` — 1.1 1.2.3 1.1.
1348 |s| {
1349 let mut i = digits(s, 0);
1350 if i == 0 {
1351 return Vec::new();
1352 }
1353 let mut groups = 0;
1354 while s[i..].starts_with('.') && digits(s, i + 1) > i + 1 {
1355 i = digits(s, i + 1);
1356 groups += 1;
1357 }
1358 if groups == 0 {
1359 return Vec::new();
1360 }
1361 if s[i..].starts_with('.') {
1362 vec![i + 1, i]
1363 } else {
1364 vec![i]
1365 }
1366 },
1367 // `\d+\.?[a-zA-Z]\.` — 9a. 3.a.
1368 |s| digits_dot_letter(s, 0, '.').into_iter().collect(),
1369 // `\d+\.?[a-zA-Z]\)` — 9a) 3.a)
1370 |s| digits_dot_letter(s, 0, ')').into_iter().collect(),
1371 // `\(\d+\.?[a-zA-Z]\)` — (9a) (3.a)
1372 |s| {
1373 if !s.starts_with('(') {
1374 return Vec::new();
1375 }
1376 digits_dot_letter(s, 1, ')').into_iter().collect()
1377 },
1378 // `\d+\.` — 1. 2. 3.
1379 |s| digits_then(s, 0, '.').into_iter().collect(),
1380 // `\d+\)` — 1) 2) 3)
1381 |s| digits_then(s, 0, ')').into_iter().collect(),
1382 // `\(\d+\)` — (1) (2) (3)
1383 |s| {
1384 if !s.starts_with('(') {
1385 return Vec::new();
1386 }
1387 digits_then(s, 1, ')').into_iter().collect()
1388 },
1389 // `\[\d+\]` — [1] [2] [3]
1390 |s| {
1391 if !s.starts_with('[') {
1392 return Vec::new();
1393 }
1394 digits_then(s, 1, ']').into_iter().collect()
1395 },
1396 // `[ivxlcdm]+\.` — i. ii. iii.
1397 |s| class_run_then(s, "ivxlcdm", '.').into_iter().collect(),
1398 // `[IVXLCDM]+\.` — I. II. III.
1399 |s| class_run_then(s, "IVXLCDM", '.').into_iter().collect(),
1400 // `[a-z]\.` / `[A-Z]\.` / `[a-z]\)` / `[A-Z]\)`
1401 |s| {
1402 letter_then(s, char::is_ascii_lowercase, '.')
1403 .into_iter()
1404 .collect()
1405 },
1406 |s| {
1407 letter_then(s, char::is_ascii_uppercase, '.')
1408 .into_iter()
1409 .collect()
1410 },
1411 |s| {
1412 letter_then(s, char::is_ascii_lowercase, ')')
1413 .into_iter()
1414 .collect()
1415 },
1416 |s| {
1417 letter_then(s, char::is_ascii_uppercase, ')')
1418 .into_iter()
1419 .collect()
1420 },
1421];
1422
1423/// Byte offset just past the run of `\d` characters starting at `from`
1424/// (`from` itself when there is none).
1425fn digits(s: &str, from: usize) -> usize {
1426 s[from..]
1427 .char_indices()
1428 .find(|(_, c)| !c.is_numeric())
1429 .map_or(s.len(), |(i, _)| from + i)
1430}
1431
1432/// `\d+<close>` from `from`: the length through `close`, if it matches.
1433fn digits_then(s: &str, from: usize, close: char) -> Option<usize> {
1434 let end = digits(s, from);
1435 (end > from && s[end..].starts_with(close)).then(|| end + close.len_utf8())
1436}
1437
1438/// `\d+\.?[a-zA-Z]<close>` from `from`.
1439fn digits_dot_letter(s: &str, from: usize, close: char) -> Option<usize> {
1440 let mut i = digits(s, from);
1441 if i == from {
1442 return None;
1443 }
1444 if s[i..].starts_with('.') {
1445 i += 1;
1446 }
1447 let letter = s[i..].chars().next().filter(char::is_ascii_alphabetic)?;
1448 i += letter.len_utf8();
1449 s[i..].starts_with(close).then(|| i + close.len_utf8())
1450}
1451
1452/// `[<class>]+<close>` at the start.
1453fn class_run_then(s: &str, class: &str, close: char) -> Option<usize> {
1454 let end = s
1455 .char_indices()
1456 .find(|(_, c)| !class.contains(*c))
1457 .map_or(s.len(), |(i, _)| i);
1458 (end > 0 && s[end..].starts_with(close)).then(|| end + close.len_utf8())
1459}
1460
1461/// `[<letter class>]<close>` at the start.
1462fn letter_then(s: &str, class: fn(&char) -> bool, close: char) -> Option<usize> {
1463 let letter = s.chars().next().filter(class)?;
1464 let i = letter.len_utf8();
1465 s[i..].starts_with(close).then(|| i + close.len_utf8())
1466}
1467
1468/// docling's `ListItemMarkerProcessor.process_list_item`: the item's original
1469/// text is matched against `^(marker)\s(.+)` (DOTALL) for the bullet patterns,
1470/// then the numbered ones in order; a hit splits it into
1471/// `(marker, text, enumerated)`. `\s` is one Unicode whitespace character and
1472/// `.+` everything after it, which must be non-empty.
1473fn split_list_marker(text: &str) -> Option<(&str, &str, bool)> {
1474 let tail_after = |len: usize| -> Option<&str> {
1475 let ws = text[len..].chars().next()?;
1476 if !ws.is_whitespace() {
1477 return None;
1478 }
1479 let rest = &text[len + ws.len_utf8()..];
1480 (!rest.is_empty()).then_some(rest)
1481 };
1482 let first = text.chars().next()?;
1483 if LIST_BULLET_MARKERS.contains(first) {
1484 if let Some(rest) = tail_after(first.len_utf8()) {
1485 return Some((&text[..first.len_utf8()], rest, false));
1486 }
1487 }
1488 for matcher in LIST_NUMBERED_MARKERS {
1489 for len in matcher(text) {
1490 if let Some(rest) = tail_after(len) {
1491 return Some((&text[..len], rest, true));
1492 }
1493 }
1494 }
1495 None
1496}
1497
1498/// A PDF `list_item` region as docling emits it: `ListItemMarkerProcessor`
1499/// splits the marker off (see [`split_list_marker`]), and docling-core's
1500/// Markdown list serializer (default `orig_list_item_marker_mode = AUTO`,
1501/// `ensure_valid_list_item_marker`) prints it as `N. text` for an `N.` marker
1502/// (`case_already_valid`: the marker verbatim); as `- text` for a bullet glyph
1503/// (no letter or digit in the marker: only the `-` the serializer adds); and
1504/// as `- a) text` / `- 1.2 text` / `- [3] text` for any other marker
1505/// (`case_auto`: the serializer's `-`, then the original marker) — spelled
1506/// here as a bullet item whose text carries the marker, the way the DOCX and
1507/// DOC backends already spell theirs. An item without a recognizable marker is
1508/// a plain bullet. The symbol-font bullets docling-parse filters out of its
1509/// cells (`•◦▪·*` glued to the text) are stripped before the match, as before.
1510fn list_item_node(text: &str, loc: [u16; 4], first_in_list: bool) -> Node {
1511 // docling's match runs on the text as docling-parse hands it over; the
1512 // glued symbol-font bullets it never sees are stripped only when the raw
1513 // text matches no marker (`•Text` → `Text`, but `• Text` → marker `•`).
1514 let stripped = text
1515 .trim_start_matches(['•', '◦', '▪', '·', '*'])
1516 .trim_start();
1517 let split = split_list_marker(text).or_else(|| split_list_marker(stripped));
1518 let bullet = |text: String, marker: &str| Node::ListItem {
1519 ordered: false,
1520 number: 0,
1521 first_in_list,
1522 text: md_escape(&text),
1523 level: 0,
1524 // docling keeps the marker as the DocLang list marker
1525 // (`<ldiv><marker>·</marker></ldiv>`); Markdown prints its own `-`.
1526 marker: Some(marker.to_string()),
1527 location: Some(loc),
1528 dclx: None,
1529 href: None,
1530 layer: None,
1531 };
1532 // docling-core's `case_already_valid` is a *full* `\d+\.` match: `3.a.`
1533 // and `1.2` are `case_auto` markers (`- 3.a. text`), not item numbers.
1534 let is_number_dot = |m: &str| {
1535 m.strip_suffix('.')
1536 .is_some_and(|d| !d.is_empty() && d.chars().all(char::is_numeric))
1537 };
1538 match split {
1539 Some((marker, body, true)) if is_number_dot(marker) => {
1540 let number = parse_ordered_marker(marker).map_or(0, |(n, _)| n);
1541 Node::ListItem {
1542 ordered: true,
1543 number,
1544 first_in_list,
1545 text: md_escape(body),
1546 level: 0,
1547 marker: Some(marker.to_string()),
1548 location: Some(loc),
1549 dclx: None,
1550 href: None,
1551 layer: None,
1552 }
1553 }
1554 // `case_auto`: a marker holding a letter or digit rides in the text.
1555 Some((marker, body, true)) => bullet(format!("{marker} {body}"), marker),
1556 Some((marker, body, false)) => bullet(body.to_string(), marker),
1557 None => bullet(stripped.to_string(), "·"),
1558 }
1559}
1560
1561fn parse_ordered_marker(s: &str) -> Option<(u64, String)> {
1562 let digits: String = s.chars().take_while(|c| c.is_ascii_digit()).collect();
1563 if digits.is_empty() {
1564 return None;
1565 }
1566 let rest = s[digits.len()..].strip_prefix('.')?;
1567 let number = digits.parse().ok()?;
1568 Some((number, rest.trim_start().to_string()))
1569}
1570
1571/// Escape markdown special characters the way docling-core's markdown serializer
1572/// does (`markdown.py` post_process): `_` → `\_`, then HTML-escape `&`, `<`, `>`
1573/// (quote=False, so quotes are left). Applied to prose (headings, list items,
1574/// paragraphs); code blocks, the formula placeholder, and table cells are left raw.
1575fn md_escape(text: &str) -> String {
1576 text.replace('_', "\\_")
1577 .replace('&', "&")
1578 .replace('<', "<")
1579 .replace('>', ">")
1580}
1581
1582fn clean_text(text: &str) -> String {
1583 // Typographic-quote normalization follows docling-parse's sanitizer table
1584 // (`pdf_sanitators/constants.h`): every curly quote — single *and double* —
1585 // becomes the ASCII apostrophe `'`, and `‚` a comma. A `"` in docling's
1586 // output only ever comes from a literal `quotedbl` glyph, never from `“ ”`
1587 // (2206's `'text in the wild"` pairs a curly open with a literal-quote
1588 // close). This replaces an earlier Hangul-only special case that patched
1589 // one symptom of mapping `“ ”` to `"`.
1590 let replaced = text
1591 .replace("\u{2} ", "")
1592 .replace("\u{ad} ", "")
1593 .replace(['\u{2}', '\u{ad}'], "") // any stray wrap hyphens not at a join
1594 .replace(
1595 [
1596 '\u{2018}', '\u{2019}', '\u{201b}', '\u{201c}', '\u{201d}', '\u{201e}', '\u{201f}',
1597 ],
1598 "'",
1599 ) // ‘ ’ ‛ “ ” „ ‟ → '
1600 .replace('\u{201a}', ",") // ‚ → ,
1601 .replace(
1602 [
1603 '\u{2010}', '\u{2011}', '\u{2012}', '\u{2013}', '\u{2014}', '\u{2015}', '\u{2212}',
1604 ],
1605 "-",
1606 ) // hyphen/dash family → -
1607 .replace('\u{2044}', "/") // ⁄ fraction slash → /
1608 .replace('\u{2022}', "\u{b7}") // • → · (docling never emits •; inline CCS-concept separators)
1609 .replace('\u{2026}', "..."); // … → ...
1610 // The docling-parse sanitizer already placed the correct spacing (e.g.
1611 // justified double spaces); preserve internal runs of spaces, only
1612 // normalizing line breaks/tabs and trimming the ends.
1613 let out = replaced.replace(['\n', '\r', '\t'], " ").trim().to_string();
1614 fix_arabic_lam_alef(&out)
1615}
1616
1617/// pdfium decomposes the Arabic lam-alef ligature (لا / لإ / لأ / لآ) into its
1618/// glyph constituents in *visual* order — `alef-variant, lam` — but docling keeps
1619/// logical order, `lam, alef-variant`. Swap a mid-word `alef-variant + lam` back
1620/// to `lam + alef-variant`. "Mid-word" (the previous char is an Arabic letter)
1621/// distinguishes the ligature from the definite article `ال` (word-initial
1622/// `alef + lam`), which must stay. No-op for non-Arabic text.
1623fn fix_arabic_lam_alef(s: &str) -> String {
1624 let is_arabic_letter = |c: char| ('\u{0620}'..='\u{064A}').contains(&c);
1625 let chars: Vec<char> = s.chars().collect();
1626 if !chars.iter().any(|&c| is_arabic_letter(c)) {
1627 return s.to_string(); // no-op for non-Arabic text
1628 }
1629 // Pass 1: swap mid-word `alef-variant + lam` → `lam + alef-variant`. Only the
1630 // hamza/madda alef variants (إ أ آ) are safe: the definite article is always
1631 // plain `ا + ل`, so plain `alef + lam` is ambiguous (a legitimate `فعالة` vs a
1632 // reversed `لا` ligature look identical) — leaving plain alef alone avoids
1633 // corrupting legitimate words.
1634 let mut a: Vec<char> = Vec::with_capacity(chars.len());
1635 let mut i = 0;
1636 while i < chars.len() {
1637 let c = chars[i];
1638 if matches!(c, '\u{0622}' | '\u{0623}' | '\u{0625}')
1639 && chars.get(i + 1) == Some(&'\u{0644}')
1640 && i > 0
1641 && is_arabic_letter(chars[i - 1])
1642 // A preceding lam means this alef-variant is *already* the logical
1643 // `lam + alef` ligature; the following lam is the next syllable's
1644 // letter, not a reversed ligature — swapping it corrupts `لآل` → `للآ`
1645 // (e.g. التعلم الآلي → الآلي, not اللآي).
1646 && chars[i - 1] != '\u{0644}'
1647 {
1648 a.push('\u{0644}');
1649 a.push(c);
1650 i += 2;
1651 continue;
1652 }
1653 a.push(c);
1654 i += 1;
1655 }
1656 // Pass 2: insert a space at Arabic↔Latin boundaries (bidi script switch) that
1657 // pdfium runs together — docling separates the embedded Latin run (`وPython`
1658 // → `و Python`).
1659 let mut out: Vec<char> = Vec::with_capacity(a.len());
1660 for (j, &c) in a.iter().enumerate() {
1661 if j > 0 {
1662 let p = a[j - 1];
1663 if (is_arabic_letter(p) && c.is_ascii_alphabetic())
1664 || (p.is_ascii_alphabetic() && is_arabic_letter(c))
1665 {
1666 out.push(' ');
1667 }
1668 }
1669 out.push(c);
1670 }
1671 out.into_iter().collect()
1672}
1673
1674/// docling's `PageAssembleModel._match_hyperlink`: the URI whose link
1675/// annotations cover at least half of the region's box, or `None`. Coverage is
1676/// intersection-over-region-area, **accumulated per URI** — a URL that wraps
1677/// across lines carries several annotation rects that sum toward the same
1678/// target. Ties resolve to the first-seen URI (Python's `max` over dict
1679/// insertion order); the winner still needs `>= 0.5`
1680/// (`_HYPERLINK_COVERAGE_THRESHOLD`).
1681pub(crate) fn region_hyperlink(
1682 region: &Region,
1683 links: &[crate::pdfium_backend::LinkAnnot],
1684) -> Option<String> {
1685 if links.is_empty() {
1686 return None;
1687 }
1688 let area = (region.r - region.l).max(0.0) * (region.b - region.t).max(0.0);
1689 if area <= 0.0 {
1690 return None;
1691 }
1692 let mut coverage: Vec<(&str, f32)> = Vec::new();
1693 for link in links {
1694 let ix = (region.r.min(link.r) - region.l.max(link.l)).max(0.0);
1695 let iy = (region.b.min(link.b) - region.t.max(link.t)).max(0.0);
1696 let c = ix * iy / area;
1697 match coverage.iter_mut().find(|(uri, _)| *uri == link.uri) {
1698 Some((_, acc)) => *acc += c,
1699 None => coverage.push((&link.uri, c)),
1700 }
1701 }
1702 let mut best: Option<(&str, f32)> = None;
1703 for (uri, c) in coverage {
1704 // Strictly greater keeps the first-seen URI on ties, like Python's max.
1705 if best.is_none_or(|(_, bc)| c > bc) {
1706 best = Some((uri, c));
1707 }
1708 }
1709 let (uri, c) = best?;
1710 (c >= 0.5).then(|| normalize_uri(uri))
1711}
1712
1713/// The pydantic-`AnyUrl` normalization docling's hyperlink value passes
1714/// through on its way to the serializer: a URL with an authority but no path
1715/// gains a trailing `/` (`https://arxiv.org` → `https://arxiv.org/`). Other
1716/// AnyUrl canonicalizations (scheme/host lowercasing, percent-encoding) don't
1717/// occur in PDF link annotations in practice, so they are not reproduced.
1718fn normalize_uri(uri: &str) -> String {
1719 if let Some((_, rest)) = uri.split_once("://") {
1720 if !rest.is_empty() && !rest.contains(['/', '?', '#']) {
1721 return format!("{uri}/");
1722 }
1723 }
1724 uri.to_string()
1725}
1726
1727/// Resolve each page hyperlink to the visible text it covers, as `(anchor, uri)`
1728/// in reading order. The anchor is the cells whose centre falls in the link rect,
1729/// joined left-to-right and cleaned the same way prose is (so it matches the
1730/// serialized text), deduped against the immediately-preceding link so pdfium's
1731/// occasional duplicate annotation doesn't double-list. Empty anchors are dropped.
1732pub(crate) fn resolve_link_anchors(page: &PdfPage) -> Vec<(String, String)> {
1733 let mut out: Vec<(String, String)> = Vec::new();
1734 // Use per-word cells, not the line-merged `cells`: a link rect covers a few
1735 // words on a line, and a whole merged line cell would over-capture (its centre
1736 // lands in one link's rect, grabbing the entire line as that link's anchor).
1737 let words = if page.word_cells.is_empty() {
1738 &page.cells
1739 } else {
1740 &page.word_cells
1741 };
1742 for link in &page.links {
1743 // A cell participates when its centre row is inside the rect and it
1744 // overlaps the rect horizontally. A cell can be *wider* than the rect:
1745 // PDFs often draw a whole header line as one text run ("LinkedIn |
1746 // GitHub | Credly"), which docling-parse's word grouping keeps as one
1747 // cell even though each label carries its own link annotation —
1748 // centre-in-rect alone would hand the entire line to every link.
1749 // [`cell_text_in_rect`] clips such a cell to the tokens under the rect.
1750 let mut inside: Vec<(&TextCell, String)> = words
1751 .iter()
1752 .filter(|c| {
1753 let cy = (c.t + c.b) / 2.0;
1754 cy >= link.t && cy <= link.b && c.r.min(link.r) > c.l.max(link.l)
1755 })
1756 .filter_map(|c| {
1757 let text = cell_text_in_rect(c, link.l, link.r);
1758 (!text.is_empty()).then_some((c, text))
1759 })
1760 .collect();
1761 // Reading order: top band then left-to-right (link anchors are LTR).
1762 let band = inside
1763 .iter()
1764 .map(|(c, _)| (c.b - c.t).abs())
1765 .fold(0.0f32, f32::max)
1766 .max(1.0);
1767 inside.sort_by_key(|(c, _)| ((c.t / band).round() as i64, (c.l * 10.0) as i64));
1768 let anchor = clean_text(
1769 &inside
1770 .iter()
1771 .map(|(_, t)| t.trim())
1772 .filter(|t| !t.is_empty())
1773 .collect::<Vec<_>>()
1774 .join(" "),
1775 );
1776 if anchor.is_empty() {
1777 continue;
1778 }
1779 if out
1780 .last()
1781 .is_some_and(|(a, u)| a == &anchor && u == &link.uri)
1782 {
1783 continue;
1784 }
1785 out.push((anchor, link.uri.clone()));
1786 }
1787 out
1788}
1789
1790/// The part of a cell's text that lies under a link rect's x-range. A cell
1791/// fully inside the rect (by centre) returns its whole text. A wider cell is
1792/// split into whitespace tokens whose x-spans are estimated proportionally to
1793/// their character positions (kerning makes this approximate, so selection
1794/// snaps to whole tokens, never characters); tokens whose estimated centre
1795/// falls inside the rect are kept. Returns "" when nothing falls inside.
1796fn cell_text_in_rect(c: &TextCell, l: f32, r: f32) -> String {
1797 let cx = (c.l + c.r) / 2.0;
1798 if cx >= l && cx <= r && c.l >= l - (c.r - c.l) * 0.25 && c.r <= r + (c.r - c.l) * 0.25 {
1799 return c.text.trim().to_string();
1800 }
1801 let chars: Vec<char> = c.text.chars().collect();
1802 let n = chars.len();
1803 if n == 0 || c.r <= c.l {
1804 return String::new();
1805 }
1806 let per = (c.r - c.l) / n as f32;
1807 let mut out: Vec<String> = Vec::new();
1808 let mut token = String::new();
1809 let mut start = 0usize;
1810 // A trailing sentinel space flushes the last token.
1811 for (i, &ch) in chars.iter().enumerate().chain(std::iter::once((n, &' '))) {
1812 if ch.is_whitespace() {
1813 if !token.is_empty() {
1814 let mid = c.l + (start as f32 + (i - start) as f32 / 2.0) * per;
1815 if mid >= l && mid <= r {
1816 out.push(std::mem::take(&mut token));
1817 } else {
1818 token.clear();
1819 }
1820 }
1821 } else {
1822 if token.is_empty() {
1823 start = i;
1824 }
1825 token.push(ch);
1826 }
1827 }
1828 out.join(" ")
1829}
1830
1831/// Cells assigned to a region (best container), in reading order, joined.
1832fn region_text(region: &Region, cells: &[TextCell]) -> String {
1833 let inside: Vec<&TextCell> = cells
1834 .iter()
1835 .filter(|c| {
1836 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1837 inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
1838 })
1839 .collect();
1840 cells_text(inside)
1841}
1842
1843/// docling's exclusive cell assignment (`_assign_cells_to_clusters`): every
1844/// non-empty cell goes to the single best-overlapping *regular* region at
1845/// intersection-over-self > 0.2, and each region serializes exactly its
1846/// assigned cells. A cell under two overlapping boxes is emitted once (by the
1847/// better-covering one), and a cell only partially under its region — e.g.
1848/// normal_4pages' big section numeral, ~30 % inside the heading box — still
1849/// joins it (`## 들어가며 1`) instead of leaking as an orphan. Pictures and
1850/// wrappers never claim (docling walks regular clusters only); ties go to the
1851/// first region, like docling's strict `>` best-overlap scan.
1852pub fn region_texts_exclusive(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
1853 let owned = assign_cells(regions, cells);
1854 // Non-claimers (tables/wrappers/pictures) keep the inclusive > 0.5 text:
1855 // docling fills a special cluster's cells from its contained children, and
1856 // downstream table assembly gates on that text being non-empty.
1857 regions
1858 .iter()
1859 .zip(owned)
1860 .map(|(r, cs)| {
1861 if claims_cells(r) {
1862 cells_text(cs.iter().map(|&i| &cells[i]).collect())
1863 } else {
1864 region_text(r, cells)
1865 }
1866 })
1867 .collect()
1868}
1869
1870/// A *regular* region in docling's sense — one that claims cells. Pictures and
1871/// the wrappers (`table`, `document_index`, `form`, `key_value_region`) fill
1872/// their cells from contained children instead.
1873fn claims_cells(r: &Region) -> bool {
1874 r.label != "picture" && !is_wrapper(r.label)
1875}
1876
1877/// docling's `_assign_cells_to_clusters`: each non-empty cell's index goes to
1878/// the single best-overlapping regular region at intersection-over-self > 0.2
1879/// (ties to the first region, like docling's strict `>` scan). One entry per
1880/// region, in region order.
1881fn assign_cells(regions: &[Region], cells: &[TextCell]) -> Vec<Vec<usize>> {
1882 let mut owned: Vec<Vec<usize>> = vec![Vec::new(); regions.len()];
1883 for (ci, c) in cells.iter().enumerate() {
1884 if c.text.trim().is_empty() {
1885 continue;
1886 }
1887 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
1888 let mut best: Option<(usize, f32)> = None;
1889 for (i, r) in regions.iter().enumerate() {
1890 if !claims_cells(r) {
1891 continue;
1892 }
1893 let ov = inter(r, c.l, c.t, c.r, c.b) / ca;
1894 if ov > 0.2 && best.is_none_or(|(_, b)| ov > b) {
1895 best = Some((i, ov));
1896 }
1897 }
1898 if let Some((i, _)) = best {
1899 owned[i].push(ci);
1900 }
1901 }
1902 owned
1903}
1904
1905/// The ballot-box glyph a checkbox line can open with (#609): `Some(checked)`.
1906/// Empty boxes — `☐`, the white squares `□ ▢ ◻` and the shadowed `❏ ❐ ❑ ❒`
1907/// Word's checkbox bullets use — are unchecked; `☑ ☒ ⊠ ⌧ ▣ 🗹 🗷 🗵` are
1908/// checked. A bare tick or cross (`✓ ✗`) is a list bullet
1909/// ([`LIST_BULLET_MARKERS`]), not a box. Symbol-font boxes that reach the text
1910/// layer as Private Use Area codes or as their ASCII byte (Wingdings `o`,
1911/// `þ`) are not recognized: without the font the code is ambiguous (`U+F06F`
1912/// is Symbol's omicron too).
1913pub(crate) fn checkbox_glyph(c: char) -> Option<bool> {
1914 match c {
1915 '☐' | '□' | '▢' | '◻' | '❏' | '❐' | '❑' | '❒' => Some(false),
1916 '☑' | '☒' | '⊠' | '⌧' | '▣' | '\u{1F5F9}' | '\u{1F5F7}' | '\u{1F5F5}' => {
1917 Some(true)
1918 }
1919 _ => None,
1920 }
1921}
1922
1923/// `text` without its leading ballot-box glyph and the space after it — a
1924/// checkbox item's label is the option text, the box is its state.
1925pub(crate) fn strip_checkbox_glyph(text: &str) -> &str {
1926 let t = text.trim_start();
1927 match t.chars().next() {
1928 Some(c) if checkbox_glyph(c).is_some() => t[c.len_utf8()..].trim_start(),
1929 _ => text,
1930 }
1931}
1932
1933/// Give every checklist line its own checkbox region (#609). The layout
1934/// model often reads a checklist as one `text` (or `list_item`) block — on
1935/// the reporter's ReportLab page all four options are one region, which then
1936/// printed as `First option Second option …`, and docling itself splits it
1937/// into two garbled paragraphs. What marks an item is on the page: a drawn
1938/// square just left of the line ([`crate::checkbox`]), or a ballot-box glyph
1939/// the line opens with ([`checkbox_glyph`]: `☐ Yes`, `☒ Done`). A region
1940/// whose lines carry either splits into a `checkbox_selected` /
1941/// `checkbox_unselected` region per marked line (its box spans the square
1942/// and the line; a following unmarked line — a wrapped label — stays with
1943/// it), so assembly emits [`Node::CheckboxItem`]s exactly as for the model's
1944/// own checkbox labels, the glyph stripped from the label. Lines above the
1945/// first mark stay one region of the original label. The first piece
1946/// replaces the region in place and the rest are appended, so per-region
1947/// slices indexed by the original order (table grids, enrichments — neither
1948/// applies to text) stay aligned.
1949///
1950/// A square counts for a line when its vertical centre lies on the line and
1951/// it sits just left of the line's first cell: no more than 1.5 sides (at
1952/// least 6 pt) of gap, at most 1 pt of overlap. Text set *inside* squares (a
1953/// comb field) never matches. A glyph counts when it opens the line and is
1954/// the line's only box glyph; a line with several (`☐ Yes ☐ No`) stays a text
1955/// piece of its own rather than become one item labelled `Yes ☐ No`. A
1956/// region the model already labelled a checkbox is not split, but its state
1957/// follows its first line's mark when it has one — the glyph or the square
1958/// is what the page says, the model's label a guess from the pixels (it read
1959/// a `☑ Bread` line as unselected). With neither on the page this is a no-op.
1960pub fn split_checkbox_lines(
1961 regions: &mut Vec<Region>,
1962 cells: &[TextCell],
1963 boxes: &[crate::checkbox::CheckBox],
1964) {
1965 let has_glyph = || {
1966 cells
1967 .iter()
1968 .any(|c| c.text.chars().any(|ch| checkbox_glyph(ch).is_some()))
1969 };
1970 if boxes.is_empty() && !has_glyph() {
1971 return;
1972 }
1973 let owned = assign_cells(regions, cells);
1974 let mut used = vec![false; boxes.len()];
1975 let mut appended: Vec<Region> = Vec::new();
1976 for (i, mine) in owned.iter().enumerate() {
1977 let region = regions[i].clone();
1978 let model_checkbox = matches!(region.label, "checkbox_selected" | "checkbox_unselected");
1979 if !(model_checkbox || matches!(region.label, "text" | "list_item")) || mine.is_empty() {
1980 continue;
1981 }
1982 // The region's lines, top down: cells sharing a vertical centre band,
1983 // with the row's text in left-to-right order (for the glyph check).
1984 let mut sorted: Vec<&TextCell> = mine.iter().map(|&c| &cells[c]).collect();
1985 sorted.sort_by(|a, b| a.t.total_cmp(&b.t).then(a.l.total_cmp(&b.l)));
1986 // (l, t, r, b) of each line, and its cells.
1987 type Row<'c> = ((f32, f32, f32, f32), Vec<&'c TextCell>);
1988 let mut rows: Vec<Row> = Vec::new();
1989 for c in sorted {
1990 let mid = (c.t + c.b) / 2.0;
1991 match rows.last_mut() {
1992 Some((row, members)) if mid >= row.1 && mid <= row.3 => {
1993 *row = (
1994 row.0.min(c.l),
1995 row.1.min(c.t),
1996 row.2.max(c.r),
1997 row.3.max(c.b),
1998 );
1999 members.push(c);
2000 }
2001 _ => rows.push(((c.l, c.t, c.r, c.b), vec![c])),
2002 }
2003 }
2004 // What each row does to the grouping.
2005 enum Mark {
2006 /// Opens a checkbox item (`square`: the drawn one, if any).
2007 Item {
2008 square: Option<usize>,
2009 checked: bool,
2010 },
2011 /// Several box glyphs on one line (`☐ Yes ☐ No`): a text piece of
2012 /// its own, closing the open item.
2013 Plain,
2014 /// Unmarked: continues the open piece (a wrapped label).
2015 Continue,
2016 }
2017 let mut marks: Vec<Mark> = Vec::with_capacity(rows.len());
2018 for ((l, t, _, b), members) in &rows {
2019 let square = boxes
2020 .iter()
2021 .enumerate()
2022 .filter(|&(k, sq)| {
2023 let side = sq.r - sq.l;
2024 let mid = (sq.t + sq.b) / 2.0;
2025 let gap = l - sq.r;
2026 !used[k]
2027 && mid >= t - 2.0
2028 && mid <= b + 2.0
2029 && gap >= -1.0
2030 && gap <= (1.5 * side).max(6.0)
2031 })
2032 .min_by(|a, b| (l - a.1.r).total_cmp(&(l - b.1.r)))
2033 .map(|(k, _)| k);
2034 let mark = match square {
2035 Some(k) => {
2036 used[k] = true;
2037 Mark::Item {
2038 square: Some(k),
2039 checked: boxes[k].checked,
2040 }
2041 }
2042 None => {
2043 let mut members = members.clone();
2044 members.sort_by(|a, b| a.l.total_cmp(&b.l));
2045 let text: String = members.iter().map(|c| c.text.as_str()).collect();
2046 let first = text.trim_start().chars().next().and_then(checkbox_glyph);
2047 let glyphs = text
2048 .chars()
2049 .filter(|&c| checkbox_glyph(c).is_some())
2050 .count();
2051 match (first, glyphs) {
2052 (Some(checked), 1) => Mark::Item {
2053 square: None,
2054 checked,
2055 },
2056 (_, 0) => Mark::Continue,
2057 _ => Mark::Plain,
2058 }
2059 }
2060 };
2061 marks.push(mark);
2062 }
2063 // A region the model already labelled a checkbox keeps its extent;
2064 // only its state follows the mark — a glyph or a drawn square is the
2065 // page's own record, where the model reads it off the pixels (on a
2066 // `☑ Bread` line Heron says unselected).
2067 if model_checkbox {
2068 if let Some(Mark::Item { checked, .. }) = marks.first() {
2069 regions[i].label = if *checked {
2070 "checkbox_selected"
2071 } else {
2072 "checkbox_unselected"
2073 };
2074 }
2075 continue;
2076 }
2077 if !marks.iter().any(|m| matches!(m, Mark::Item { .. })) {
2078 continue;
2079 }
2080 // Group rows: a marked row opens an item; an unmarked one continues
2081 // the open group (the leading group keeps the region's own label).
2082 let mut pieces: Vec<Region> = Vec::new();
2083 for ((row, _), mark) in rows.iter().zip(&marks) {
2084 match mark {
2085 Mark::Item { square, checked } => {
2086 let (mut l, mut t, mut r, mut b) = *row;
2087 if let Some(s) = square.map(|k| &boxes[k]) {
2088 (l, t, r, b) = (l.min(s.l), t.min(s.t), r.max(s.r), b.max(s.b));
2089 }
2090 pieces.push(Region {
2091 label: if *checked {
2092 "checkbox_selected"
2093 } else {
2094 "checkbox_unselected"
2095 },
2096 score: region.score,
2097 l,
2098 t,
2099 r,
2100 b,
2101 });
2102 }
2103 Mark::Plain => pieces.push(Region {
2104 l: row.0,
2105 t: row.1,
2106 r: row.2,
2107 b: row.3,
2108 ..region.clone()
2109 }),
2110 Mark::Continue => match pieces.last_mut() {
2111 Some(p) => {
2112 p.l = p.l.min(row.0);
2113 p.t = p.t.min(row.1);
2114 p.r = p.r.max(row.2);
2115 p.b = p.b.max(row.3);
2116 }
2117 None => pieces.push(Region {
2118 l: row.0,
2119 t: row.1,
2120 r: row.2,
2121 b: row.3,
2122 ..region.clone()
2123 }),
2124 },
2125 }
2126 }
2127 let mut pieces = pieces.into_iter();
2128 if let Some(first) = pieces.next() {
2129 regions[i] = first;
2130 }
2131 appended.extend(pieces);
2132 }
2133 regions.extend(appended);
2134}
2135
2136/// docling's regular-cluster refinement after cell assignment
2137/// (`LayoutPostprocessor._process_regular_clusters`, #419), run once the page's
2138/// cells are final and before reading order:
2139///
2140/// 1. every regular region's box becomes the union of the cells it claimed
2141/// (`_adjust_cluster_bboxes` — a regular cluster's bbox *is* its cells'
2142/// bbox; a table's is the union with the model box, and pictures keep
2143/// theirs, so neither is touched here);
2144/// 2. a regular region that claimed no cell is dropped (`keep_empty_clusters`
2145/// is off; a `formula` is kept, as upstream keeps it);
2146/// 3. an orphan text region (`score == 0.0`, from [`add_orphan_regions`]) that
2147/// now sits > 0.8 inside another regular region's fitted box is folded into
2148/// it (`_remove_overlapping_clusters` at containment 0.8, the larger box
2149/// winning the group) — up to three rounds, like upstream's loop.
2150///
2151/// Why it matters: the layout model's box can end partway through a line. That
2152/// line fails the 0.2 claim and becomes an orphan — recoverable — but the
2153/// *model* box still overlaps the orphan's line by a few points, so the
2154/// reading-order graph, which links only strictly-above pairs, gets no edge
2155/// between them and may emit the next paragraph first, stranding the line
2156/// after the paragraph it belongs in (1540 of 6050 text blocks on the #419
2157/// book began mid-sentence). Fitted to its cells, the box ends on a line
2158/// boundary and the orphan slots in between; an orphan the fitted box
2159/// swallows joins the paragraph outright. Cell assignment is untouched: a
2160/// region's fitted box contains every cell it claimed, so
2161/// [`region_texts_exclusive`] hands it the same cells afterwards.
2162///
2163/// A page with no cells yet (a scan before OCR) is left alone: dropping every
2164/// text region for want of cells would be wrong, and the OCR paths call this
2165/// again once the cells exist.
2166pub fn fit_regions_to_cells(regions: &mut Vec<Region>, cells: &[TextCell]) {
2167 if !cells.iter().any(|c| !c.text.trim().is_empty()) {
2168 return;
2169 }
2170 for _ in 0..3 {
2171 let owned = assign_cells(regions, cells);
2172 let mut fitted: Vec<Region> = Vec::with_capacity(regions.len());
2173 for (r, own) in regions.iter().zip(&owned) {
2174 if !claims_cells(r) {
2175 fitted.push(r.clone());
2176 continue;
2177 }
2178 if own.is_empty() {
2179 if r.label == "formula" {
2180 fitted.push(r.clone());
2181 }
2182 continue;
2183 }
2184 let mut f = r.clone();
2185 f.l = own
2186 .iter()
2187 .map(|&i| cells[i].l)
2188 .fold(f32::INFINITY, f32::min);
2189 f.t = own
2190 .iter()
2191 .map(|&i| cells[i].t)
2192 .fold(f32::INFINITY, f32::min);
2193 f.r = own
2194 .iter()
2195 .map(|&i| cells[i].r)
2196 .fold(f32::NEG_INFINITY, f32::max);
2197 f.b = own
2198 .iter()
2199 .map(|&i| cells[i].b)
2200 .fold(f32::NEG_INFINITY, f32::max);
2201 fitted.push(f);
2202 }
2203 let mut changed = fitted.len() != regions.len();
2204 // Fold orphans into the regular region whose fitted box holds them.
2205 let mut drop = vec![false; fitted.len()];
2206 for i in 0..fitted.len() {
2207 let o = &fitted[i];
2208 if !(o.score == 0.0 && o.label == "text") {
2209 continue;
2210 }
2211 let oa = area(o.l, o.t, o.r, o.b).max(1.0);
2212 let mut best: Option<(usize, f32)> = None;
2213 for (j, r) in fitted.iter().enumerate() {
2214 if j == i || drop[j] || r.score == 0.0 || !claims_cells(r) {
2215 continue;
2216 }
2217 let ov = inter(r, o.l, o.t, o.r, o.b) / oa;
2218 if ov > 0.8 && best.is_none_or(|(_, b)| ov > b) {
2219 best = Some((j, ov));
2220 }
2221 }
2222 if let Some((j, _)) = best {
2223 let (l, t, r, b) = (o.l, o.t, o.r, o.b);
2224 let host = &mut fitted[j];
2225 host.l = host.l.min(l);
2226 host.t = host.t.min(t);
2227 host.r = host.r.max(r);
2228 host.b = host.b.max(b);
2229 drop[i] = true;
2230 changed = true;
2231 }
2232 }
2233 let mut drop = drop.into_iter();
2234 fitted.retain(|_| !drop.next().expect("aligned"));
2235 *regions = fitted;
2236 if !changed {
2237 break;
2238 }
2239 }
2240}
2241
2242/// Join a prefiltered cell list into the region's text (docling's
2243/// `sanitize_text` over the sanitizer's cell order).
2244fn cells_text(inside: Vec<&TextCell>) -> String {
2245 // docling orders a cluster's cells by their docling-parse cell index
2246 // alone (`LayoutPostprocessor._sort_cells`: `sorted(cells, key=c.index)`)
2247 // — the sanitizer's output order, which our `cells` slice already is.
2248 // No geometric re-sort: normal_4pages' big section numerals paint
2249 // *after* their heading text, and docling's `## 들어가며 1` (numeral
2250 // last) only falls out of pure index order — a band sort dragged the
2251 // numeral to the front. The overlap-grouped line restore this replaced
2252 // measured strictly worse on the corpus (it fixed nothing the index
2253 // order broke, and broke the numerals).
2254 let joined = {
2255 // docling's `PageAssembleModel.sanitize_text`, ported verbatim over the
2256 // parse-index-ordered lines: append a separating space to a line —
2257 // unless it ends with `-`. A dash-ending line whose last word and the
2258 // next line's first word are both alphanumeric is a wrapped word: the
2259 // dash is dropped and the lines fuse (`platforms-` + `reflects` →
2260 // `platformsreflects`, `pp. 545-` + `561` → `545561`). Any other
2261 // dash-ending line — e.g. the *bare* `-` cell a superscript ORCID or an
2262 // inline `–` bullet splits off (its word list is empty, so the fuse
2263 // test fails) — keeps its dash and still takes no trailing space:
2264 // `[0000` `-` `0002` joins as docling's `[0000 -0002`, and the OTSL
2265 // list's `-` + `"C" cell -` + `a new table cell` collapses to
2266 // `-"C" cell a new table cell`. Our cells still carry the raw dash
2267 // family (docling-parse normalizes to `-` before this; clean_text does
2268 // it after), so the endswith test matches them all.
2269 let texts: Vec<&str> = inside
2270 .iter()
2271 .map(|c| c.text.trim())
2272 // Skip whitespace-only cells (a justified line's trailing space
2273 // glyph): an empty line would double the separator.
2274 .filter(|t| !t.is_empty())
2275 .collect();
2276 let last_word_alnum = |s: &str| {
2277 s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
2278 .rfind(|w| !w.is_empty())
2279 .is_some_and(|w| w.chars().all(char::is_alphanumeric))
2280 };
2281 let first_word_alnum = |s: &str| {
2282 s.split(|c: char| !(c.is_alphanumeric() || c == '_'))
2283 .find(|w| !w.is_empty())
2284 .is_some_and(|w| w.chars().all(char::is_alphanumeric))
2285 };
2286 let mut out = String::new();
2287 for (i, t) in texts.iter().enumerate() {
2288 if i > 0 {
2289 let prev = texts[i - 1];
2290 let dashish = matches!(
2291 prev.chars().last(),
2292 Some(
2293 '-' | '\u{2010}'
2294 | '\u{2011}'
2295 | '\u{2012}'
2296 | '\u{2013}'
2297 | '\u{2014}'
2298 | '\u{2015}'
2299 | '\u{2212}'
2300 )
2301 );
2302 // docling#4052 (2.122): a dash only splits a word when it is
2303 // *attached* to one — the character before it is alphanumeric.
2304 // A dash that follows whitespace (a separator dash, a bullet
2305 // marker, a wrapped `-prefixed` token, the bare `-` cell an
2306 // ORCID splits off) is a literal character: it is kept and the
2307 // lines join with the ordinary space.
2308 let attached = prev.chars().rev().nth(1).is_some_and(char::is_alphanumeric);
2309 if dashish && attached {
2310 if last_word_alnum(prev) && first_word_alnum(t) {
2311 out.pop(); // wrapped word: fuse without the dash
2312 }
2313 // an attached dash never takes a separating space
2314 } else {
2315 out.push(' ');
2316 }
2317 }
2318 out.push_str(t);
2319 }
2320 out
2321 };
2322 clean_text(&joined)
2323}
2324
2325/// Tighten the spaces pdfium leaves around tight punctuation in a code line
2326/// (`console .log` → `console.log`, `add (3 , 5)` → `add(3, 5)`), matching
2327/// docling-parse's source spacing.
2328fn tighten_code_punct(s: &str) -> String {
2329 s.replace(" .", ".")
2330 .replace(" ,", ",")
2331 .replace(" ;", ";")
2332 .replace(" )", ")")
2333 .replace(" (", "(")
2334}
2335
2336/// Assemble a **code** region's text with its line structure preserved.
2337///
2338/// Unlike [`region_text`] — which joins every cell with a single space, the right
2339/// thing for prose reflow — a code block's line breaks and indentation are
2340/// significant. The `code_cells` are already one physical source line each
2341/// (grouped space-glyph-only, so monospace runs keep their spacing), so this:
2342///
2343/// 1. groups the cells into vertical line bands and orders them top→bottom,
2344/// left→right;
2345/// 2. joins the lines with `\n` (rather than spaces), keeping the carriage
2346/// returns; and
2347/// 3. reconstructs each line's leading indentation from its left offset, in units
2348/// of the block's estimated monospace character width, so nesting survives.
2349///
2350/// Typography is normalized per line via [`clean_text`] (smart quotes, dashes,
2351/// ellipsis), which never merges lines. Returns an empty string if the region has
2352/// no code cells (the caller falls back to the prose text).
2353fn code_region_text(region: &Region, cells: &[TextCell]) -> String {
2354 let mut inside: Vec<&TextCell> = cells
2355 .iter()
2356 .filter(|c| {
2357 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2358 inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
2359 })
2360 .filter(|c| !c.text.trim().is_empty())
2361 .collect();
2362 if inside.is_empty() {
2363 return String::new();
2364 }
2365
2366 // Quantize the top edge into ~line bands (like `region_text`), then order the
2367 // cells by band (top→bottom) and, within a band, by left edge.
2368 let band = inside
2369 .iter()
2370 .map(|c| (c.b - c.t).abs())
2371 .fold(0.0f32, f32::max)
2372 .max(1.0);
2373 let line_of = |c: &TextCell| (c.t / band).round() as i64;
2374 inside.sort_by_key(|c| (line_of(c), (c.l * 10.0) as i64));
2375
2376 // Estimate one monospace character's width (total ink width / total glyphs) to
2377 // convert a line's left offset into a count of leading spaces. Measured over
2378 // all lines so a single short line can't skew it.
2379 let (mut total_w, mut total_chars) = (0.0f32, 0usize);
2380 for c in &inside {
2381 let n = c.text.trim().chars().count();
2382 if n > 0 {
2383 total_w += (c.r - c.l).max(0.0);
2384 total_chars += n;
2385 }
2386 }
2387 let char_w = if total_chars > 0 {
2388 (total_w / total_chars as f32).max(1.0)
2389 } else {
2390 1.0
2391 };
2392 // The block's own left margin is the zero-indent baseline.
2393 let base_l = inside.iter().map(|c| c.l).fold(f32::INFINITY, f32::min);
2394
2395 let mut lines: Vec<String> = Vec::new();
2396 let mut cur: Option<i64> = None;
2397 for c in &inside {
2398 // Tighten pdfium's spaced punctuation per line (on the trimmed content, so
2399 // the reconstructed leading indentation is never nibbled).
2400 let text = tighten_code_punct(&clean_text(c.text.trim()));
2401 if Some(line_of(c)) == cur {
2402 // A second cell sharing this band (rare — e.g. split columns): keep it
2403 // on the same source line, separated by a space.
2404 if let Some(last) = lines.last_mut() {
2405 last.push(' ');
2406 last.push_str(&text);
2407 }
2408 continue;
2409 }
2410 let indent = ((c.l - base_l) / char_w).round().max(0.0) as usize;
2411 lines.push(format!("{}{}", " ".repeat(indent), text));
2412 cur = Some(line_of(c));
2413 }
2414 lines.join("\n")
2415}
2416
2417/// Reconstruct a table's grid geometrically from the text cells inside its
2418/// region: cluster cells into rows (by vertical centre) and columns (by clustered
2419/// left edges), then place each cell. A model-free stand-in for TableFormer that
2420/// recovers grid-aligned tables from the precise PDF text layer (it does not
2421/// resolve row/column spans).
2422pub fn reconstruct_table(region: &Region, cells: &[TextCell]) -> Vec<Vec<String>> {
2423 let mut inside: Vec<&TextCell> = cells
2424 .iter()
2425 .filter(|c| {
2426 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2427 inter(region, c.l, c.t, c.r, c.b) / ca > 0.5
2428 })
2429 .collect();
2430 if inside.is_empty() {
2431 return Vec::new();
2432 }
2433 inside.sort_by(|a, b| a.t.total_cmp(&b.t));
2434
2435 // Rows: consecutive cells whose vertical centre is within ~0.7 line height.
2436 let mut rows: Vec<(f32, Vec<&TextCell>)> = Vec::new();
2437 for c in &inside {
2438 let cyc = (c.t + c.b) / 2.0;
2439 let lh = (c.b - c.t).abs().max(1.0);
2440 if let Some((ryc, row)) = rows.last_mut() {
2441 if (cyc - *ryc).abs() < lh * 0.7 {
2442 row.push(c);
2443 continue;
2444 }
2445 }
2446 rows.push((cyc, vec![c]));
2447 }
2448
2449 // Columns: cluster left edges (merge those within a tolerance).
2450 let tol = {
2451 let mut hs: Vec<f32> = inside.iter().map(|c| (c.b - c.t).abs()).collect();
2452 hs.sort_by(f32::total_cmp);
2453 hs[hs.len() / 2].max(4.0) * 1.5
2454 };
2455 let mut lefts: Vec<f32> = inside.iter().map(|c| c.l).collect();
2456 lefts.sort_by(f32::total_cmp);
2457 let mut col_starts: Vec<f32> = Vec::new();
2458 for l in lefts {
2459 if col_starts.last().is_none_or(|&last| l - last > tol) {
2460 col_starts.push(l);
2461 }
2462 }
2463 let ncols = col_starts.len().max(1);
2464 let col_of = |l: f32| -> usize {
2465 col_starts
2466 .iter()
2467 .rposition(|&s| l + tol * 0.5 >= s)
2468 .unwrap_or(0)
2469 .min(ncols - 1)
2470 };
2471
2472 let mut grid = Vec::with_capacity(rows.len());
2473 for (_, mut row) in rows {
2474 row.sort_by(|a, b| a.l.total_cmp(&b.l));
2475 let mut cols = vec![String::new(); ncols];
2476 for c in row {
2477 let ci = col_of(c.l);
2478 // Strip the wrap-hyphen control char so it never lands in a cell.
2479 let t = c.text.trim().replace(['\u{2}', '\u{ad}'], "");
2480 if cols[ci].is_empty() {
2481 cols[ci] = t;
2482 } else {
2483 cols[ci].push(' ');
2484 cols[ci].push_str(&t);
2485 }
2486 }
2487 grid.push(cols);
2488 }
2489 grid
2490}
2491
2492/// Does the geometric reconstruction of a table look trustworthy enough to use
2493/// as-is, instead of paying for TableFormer?
2494///
2495/// [`reconstruct_table`] derives columns by clustering cell **left edges**. On a
2496/// clean grid that is exact, but when a column's entries are not left-aligned
2497/// (or the OCR boxes wobble) the clustering splits one real column into several,
2498/// and the result is a wide, mostly-empty grid — the "spurious empty columns"
2499/// failure TableFormer exists to fix.
2500///
2501/// Two symptoms separate the two cases, and both are properties of the grid
2502/// alone (no model needed):
2503/// * **density** — a real table is mostly full; a split-up one is mostly holes;
2504/// * **thin columns** — a column carrying at most one entry across several rows
2505/// is almost always a split artefact rather than a real column.
2506///
2507/// Deliberately conservative: it answers `true` only for grids that are plainly
2508/// well-formed, so the expensive path stays the default whenever there is doubt.
2509/// A caller that skips TableFormer on `true` trades no quality for the time.
2510pub fn geometric_table_is_reliable(rows: &[Vec<String>]) -> bool {
2511 let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
2512 // Fewer than two columns is not a grid this heuristic can vouch for: it is
2513 // exactly the shape a collapsed table takes, and TableFormer may recover
2514 // real structure from it.
2515 if rows.len() < 2 || ncols < 2 {
2516 return false;
2517 }
2518 let filled = |c: &String| !c.trim().is_empty();
2519 let total = rows.len() * ncols;
2520 let full = rows.iter().flatten().filter(|c| filled(c)).count();
2521 if (full as f32) < MIN_TABLE_FILL * total as f32 {
2522 return false;
2523 }
2524 // A column used by at most one row, when there are rows enough to tell.
2525 if rows.len() >= 3 {
2526 for ci in 0..ncols {
2527 let used = rows
2528 .iter()
2529 .filter(|r| r.get(ci).is_some_and(filled))
2530 .count();
2531 if used <= 1 {
2532 return false;
2533 }
2534 }
2535 }
2536 true
2537}
2538
2539/// Share of a geometric grid's cells that must carry text for it to be trusted
2540/// without TableFormer. Chosen well above the density a left-edge split
2541/// produces (those land nearer a third) and below what a genuine table with a
2542/// few blank cells reaches.
2543const MIN_TABLE_FILL: f32 = 0.6;
2544
2545/// The union bbox of the text cells assigned to a region (same >50%-overlap
2546/// rule as [`region_text`]), or `None` when no cell lands in it. docling's
2547/// LayoutPostprocessor shrinks a regular cluster's bbox to its cells, and the
2548/// enrichment crops are taken from that cell-tight box — cropping the raw
2549/// detector box instead hands the VLM surrounding chrome (e.g. the `Listing N:`
2550/// caption under a code block) that changes its output.
2551pub fn region_cell_bbox(region: &Region, cells: &[TextCell]) -> Option<[f32; 4]> {
2552 let mut bbox: Option<[f32; 4]> = None;
2553 for c in cells {
2554 let ca = area(c.l, c.t, c.r, c.b).max(1.0);
2555 if inter(region, c.l, c.t, c.r, c.b) / ca <= 0.5 {
2556 continue;
2557 }
2558 bbox = Some(match bbox {
2559 None => [c.l, c.t, c.r, c.b],
2560 Some([l, t, r, b]) => [l.min(c.l), t.min(c.t), r.max(c.r), b.max(c.b)],
2561 });
2562 }
2563 bbox
2564}
2565
2566/// One region's enrichment-model result, produced by the pipeline's opt-in
2567/// passes (issue #76) and applied during assembly.
2568#[derive(Debug, Clone)]
2569pub enum Enrichment {
2570 /// DocumentPictureClassifier predictions, descending confidence.
2571 PictureClasses(Vec<PictureClass>),
2572 /// CodeFormulaV2 output for a `code` region: the rewritten source text and
2573 /// the `<_language_>` prefix (when the model emitted one).
2574 Code {
2575 language: Option<String>,
2576 text: String,
2577 },
2578 /// CodeFormulaV2 output for a `formula` region: the decoded LaTeX.
2579 Formula { latex: String },
2580}
2581
2582/// Crop a region (page points, already expanded by the caller if needed) from
2583/// the rendered page image and resize it to `target_scale` pixels per point —
2584/// the enrichment-model equivalent of docling's
2585/// `page.get_image(scale=…, cropbox=…)`, sourced from the existing
2586/// [`crate::pdfium_backend::RENDER_SCALE`] render instead of a fresh pdfium
2587/// pass (the page bitmap is already the exact docling render at scale 2).
2588#[cfg(feature = "ml")]
2589pub fn crop_region_scaled(page: &PdfPage, bbox: [f32; 4], target_scale: f32) -> Option<RgbImage> {
2590 let s = page.scale;
2591 let [l, t, r, b] = bbox;
2592 let (iw, ih) = (page.image.width(), page.image.height());
2593 let x = (l * s).max(0.0) as u32;
2594 let y = (t * s).max(0.0) as u32;
2595 if x >= iw || y >= ih {
2596 return None;
2597 }
2598 let w = (((r - l.max(0.0)) * s) as u32).min(iw - x);
2599 let h = (((b - t.max(0.0)) * s) as u32).min(ih - y);
2600 if w == 0 || h == 0 {
2601 return None;
2602 }
2603 let crop = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
2604 // docling renders the crop at `target_scale` directly; from the scale-2
2605 // page render that is a resize to the same pixel geometry
2606 // (`round(width_points * scale)`, PIL's BICUBIC ≙ CatmullRom).
2607 let tw = ((w as f32 / s) * target_scale).round().max(1.0) as u32;
2608 let th = ((h as f32 / s) * target_scale).round().max(1.0) as u32;
2609 if (tw, th) == (w, h) {
2610 return Some(crop);
2611 }
2612 Some(image::imageops::resize(
2613 &crop,
2614 tw,
2615 th,
2616 image::imageops::FilterType::CatmullRom,
2617 ))
2618}
2619
2620/// Resample `img` (rendered at `from` px/pt, covering `w_pt`×`h_pt` points)
2621/// to `to` px/pt — docling's `round(points * scale)` pixel geometry, PIL's
2622/// BICUBIC ≙ CatmullRom. Unchanged when the geometry already matches.
2623#[cfg(feature = "ocr-prep")]
2624fn rescale(img: RgbImage, w_pt: f32, h_pt: f32, to: f32) -> RgbImage {
2625 let tw = (w_pt * to).round().max(1.0) as u32;
2626 let th = (h_pt * to).round().max(1.0) as u32;
2627 if (tw, th) == img.dimensions() {
2628 return img;
2629 }
2630 image::imageops::resize(&img, tw, th, image::imageops::FilterType::CatmullRom)
2631}
2632
2633/// Encode `img` as a PNG [`PictureImage`] rendered at `scale` px/pt.
2634#[cfg(feature = "ocr-prep")]
2635fn png_image(img: &RgbImage, scale: f32) -> Option<PictureImage> {
2636 let mut buf = std::io::Cursor::new(Vec::new());
2637 img.write_to(&mut buf, image::ImageFormat::Png).ok()?;
2638 Some(PictureImage {
2639 mimetype: "image/png".into(),
2640 width: img.width(),
2641 height: img.height(),
2642 data: buf.into_inner(),
2643 dpi: PictureImage::dpi_for_scale(scale),
2644 })
2645}
2646
2647/// The whole page render as docling's `PageItem.image` (#520): at `scale`
2648/// px/pt (`None` = the render's own), `None` when the page has no bitmap.
2649#[cfg(feature = "ocr-prep")]
2650pub fn page_image(page: &PdfPage, scale: Option<f32>) -> Option<PictureImage> {
2651 if page.image.width() == 0 || page.image.height() == 0 || page.scale <= 0.0 {
2652 return None;
2653 }
2654 let scale = scale.unwrap_or(page.scale);
2655 let img = rescale(page.image.clone(), page.width, page.height, scale);
2656 png_image(&img, scale)
2657}
2658
2659/// Crop a layout region from the rendered page image and encode it as PNG (the
2660/// figure bytes docling stores on a `PictureItem`). Region coordinates are page
2661/// points; the image is rendered at `page.scale` and resampled to `scale`
2662/// px/pt when one is given (docling's `images_scale`, #520). The image's `dpi`
2663/// is 72·scale (#519).
2664#[cfg(feature = "ocr-prep")]
2665fn crop_region(page: &PdfPage, region: &Region, scale: Option<f32>) -> Option<PictureImage> {
2666 let s = page.scale;
2667 let (iw, ih) = (page.image.width(), page.image.height());
2668 let x = (region.l * s).max(0.0) as u32;
2669 let y = (region.t * s).max(0.0) as u32;
2670 if x >= iw || y >= ih {
2671 return None;
2672 }
2673 let w = (((region.r - region.l) * s) as u32).min(iw - x);
2674 let h = (((region.b - region.t) * s) as u32).min(ih - y);
2675 if w == 0 || h == 0 {
2676 return None;
2677 }
2678 let sub = image::imageops::crop_imm(&page.image, x, y, w, h).to_image();
2679 match scale {
2680 Some(to) if (to - s).abs() > f32::EPSILON => {
2681 // The crop's own point extent (the pixel box, back in points), so
2682 // the resampled geometry is `round(points * scale)`.
2683 let img = rescale(sub, w as f32 / s, h as f32 / s, to);
2684 png_image(&img, to)
2685 }
2686 _ => png_image(&sub, s),
2687 }
2688}
2689
2690/// For each `picture` region, find the `caption` region closest below it (and
2691/// horizontally overlapping); docling pairs them and emits the caption first.
2692/// Each caption is claimed by at most one picture.
2693fn pair_captions(regions: &[Region]) -> Vec<Option<usize>> {
2694 let mut pairs = vec![None; regions.len()];
2695 let mut taken = vec![false; regions.len()];
2696 for (pi, p) in regions.iter().enumerate() {
2697 if p.label != "picture" {
2698 continue;
2699 }
2700 let mut best: Option<(usize, f32)> = None;
2701 for (ci, c) in regions.iter().enumerate() {
2702 if c.label != "caption" || taken[ci] {
2703 continue;
2704 }
2705 let line_h = (c.b - c.t).abs().max(1.0);
2706 let gap = c.t - p.b; // caption sits below the picture
2707 let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
2708 if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
2709 let dist = gap.abs();
2710 if best.is_none_or(|(_, bd)| dist < bd) {
2711 best = Some((ci, dist));
2712 }
2713 }
2714 }
2715 if let Some((ci, _)) = best {
2716 pairs[pi] = Some(ci);
2717 taken[ci] = true;
2718 }
2719 }
2720 pairs
2721}
2722
2723/// Pair each `code` region with the `caption` region just **above** it (a
2724/// `Listing N:` label). docling renders the code block first, then its caption,
2725/// so the caption is consumed from its own (earlier) reading-order slot and
2726/// re-emitted after the code.
2727fn pair_code_captions(regions: &[Region]) -> Vec<Option<usize>> {
2728 let mut pairs = vec![None; regions.len()];
2729 let mut taken = vec![false; regions.len()];
2730 for (pi, p) in regions.iter().enumerate() {
2731 if p.label != "code" {
2732 continue;
2733 }
2734 let mut best: Option<(usize, f32)> = None;
2735 for (ci, c) in regions.iter().enumerate() {
2736 if c.label != "caption" || taken[ci] {
2737 continue;
2738 }
2739 let line_h = (c.b - c.t).abs().max(1.0);
2740 let gap = p.t - c.b; // caption sits above the code
2741 let h_overlap = (p.r.min(c.r) - p.l.max(c.l)).max(0.0);
2742 if gap > -line_h && gap < line_h * 3.0 && h_overlap > 0.0 {
2743 let dist = gap.abs();
2744 if best.is_none_or(|(_, bd)| dist < bd) {
2745 best = Some((ci, dist));
2746 }
2747 }
2748 }
2749 if let Some((ci, _)) = best {
2750 pairs[pi] = Some(ci);
2751 taken[ci] = true;
2752 }
2753 }
2754 pairs
2755}
2756
2757/// Pair each `table`/`document_index` region with its `caption` (#265) the way
2758/// docling's `ReadingOrderPredictor._find_to_captions` does: by **reading-order
2759/// adjacency**, not geometry. A caption claims the media element
2760/// (table/picture/code) immediately next to it in the ordered region sequence,
2761/// and only when exactly one side holds one — a caption sandwiched between two
2762/// media elements stays unattached, and a text paragraph between caption and
2763/// table breaks the bond. This is what lets a flush-left "Table 3: …" label
2764/// bind a centered grid it doesn't horizontally overlap, while a caption in
2765/// the neighbouring column of a two-column page — geometrically close — never
2766/// pairs across the gutter. Runs after the picture and code pairings (the
2767/// picture/code arms of the same upstream matcher), so a caption they claimed
2768/// stays claimed. docling attaches these as `TableItem.captions` refs; the
2769/// paired caption is consumed from its own reading-order slot and rides on the
2770/// table node instead.
2771fn pair_table_captions(regions: &[Region], taken: &mut [bool]) -> Vec<Option<usize>> {
2772 let is_media = |label: &str| is_table_like(label) || matches!(label, "picture" | "code");
2773 let mut pairs: Vec<Option<usize>> = vec![None; regions.len()];
2774 for ci in 0..regions.len() {
2775 if regions[ci].label != "caption" || taken[ci] {
2776 continue;
2777 }
2778 // Furniture (headers/footers, form chrome) is not part of docling's
2779 // body-element sequence, so it neither bonds nor blocks.
2780 let prev = regions[..ci].iter().rposition(|r| !is_skipped(r.label));
2781 let next = regions[ci + 1..]
2782 .iter()
2783 .position(|r| !is_skipped(r.label))
2784 .map(|off| ci + 1 + off);
2785 let prev_media = prev.is_some_and(|j| is_media(regions[j].label));
2786 let next_media = next.is_some_and(|j| is_media(regions[j].label));
2787 let target = match (prev_media, next_media) {
2788 (true, false) => prev,
2789 (false, true) => next,
2790 // Ambiguous (media on both sides) or no media at all: leave the
2791 // caption in its own reading-order slot, as docling does.
2792 _ => None,
2793 };
2794 if let Some(ti) = target {
2795 // A first claim wins (a table with captions above *and* below
2796 // keeps the earlier one — docling's nearest-first tiebreak).
2797 if is_table_like(regions[ti].label) && pairs[ti].is_none() {
2798 pairs[ti] = Some(ci);
2799 taken[ci] = true;
2800 }
2801 }
2802 }
2803 pairs
2804}
2805
2806/// Assemble one page from its (already overlap-resolved) layout regions and
2807/// text cells.
2808/// Normalize a layout region (page points, top-left origin) to DocLang's 0–511
2809/// location grid: `clamp(round(512 · coord / page_dim), 0, 511)`, per axis,
2810/// order `[x0, y0, x1, y1]`. Mirrors docling_core's
2811/// `_doclang_utils._create_location_tokens_for_bbox` (resolution 512) so the
2812/// emitted `<location>` tokens line up with the Python groundtruth. Our heron
2813/// cluster boxes match docling's to within ~1 grid unit; the residual (mainly
2814/// the aspect-ratio-stretch vs letterbox preprocessing difference) is absorbed
2815/// by the conformance harness's geometry tolerance.
2816fn norm_loc(region: &Region, page_w: f32, page_h: f32) -> [u16; 4] {
2817 let q = |v: f32, dim: f32| -> u16 {
2818 if dim <= 0.0 {
2819 return 0;
2820 }
2821 let g = (512.0 * (v as f64) / (dim as f64)).round() as i64;
2822 g.clamp(0, 511) as u16
2823 };
2824 [
2825 q(region.l, page_w),
2826 q(region.t, page_h),
2827 q(region.r, page_w),
2828 q(region.b, page_h),
2829 ]
2830}
2831
2832/// Wrap a node in its layout provenance so the DocLang serializer emits the four
2833/// `<location>` tokens as the element's head (Markdown/JSON render `inner`
2834/// unchanged).
2835fn located(loc: [u16; 4], inner: Node) -> Node {
2836 Node::Located {
2837 location: loc,
2838 inner: Box::new(inner),
2839 }
2840}
2841
2842/// Stamp the real 1-based page number onto a page's leading marker (see
2843/// [`assemble_page`], which emits it with `page_no: 0` because only the
2844/// document-level collector knows the true index — `--pages` windows shift it).
2845pub fn stamp_page_no(nodes: &mut [Node], page_no: usize) {
2846 if let Some(Node::PageInfo { page_no: p, .. }) = nodes.first_mut() {
2847 *p = page_no;
2848 }
2849}
2850
2851/// A dense table grid plus its first-class cells (#240): `rows` is the text
2852/// grid every serializer renders (spans replicate their anchor's text);
2853/// `cells` are the docling-parity per-cell records (text, page-point bbox,
2854/// span rectangle, OTSL header roles). Produced by the TableFormer paths
2855/// (`tf_core`); lives in this always-compiled module so the pure-text (wasm
2856/// `pdf-text`) build sees the type.
2857#[derive(Clone, Debug)]
2858pub struct TableGrid {
2859 pub rows: Vec<Vec<String>>,
2860 pub cells: Vec<docling_core::TableCell>,
2861}
2862
2863/// docling's `_RICH_CELL_PICTURE_COVERAGE_THRESHOLD`.
2864const RICH_CELL_PICTURE_COVERAGE: f32 = 0.8;
2865
2866/// docling `ReadingOrderModel._match_table_pictures` (#3906, 2.118.1): every
2867/// picture ≥ 80 % inside a TableFormer-structured table on the page is matched
2868/// to the cell covering it, and returned per table as `cell index → pictures`.
2869/// A picture that pairs with a caption stays a standalone figure (upstream
2870/// would nest it and lose the caption; keeping the caption is the better
2871/// failure). Tables without first-class cells (geometric fallback) have no cell
2872/// boxes to match against and nest nothing.
2873fn match_table_pictures(
2874 regions: &[Region],
2875 table_rows: &[Option<TableGrid>],
2876 caption_for: &[Option<usize>],
2877) -> std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> {
2878 let mut out: std::collections::HashMap<usize, Vec<(usize, Vec<usize>)>> =
2879 std::collections::HashMap::new();
2880 for (p, pic) in regions.iter().enumerate() {
2881 if pic.label != "picture" || caption_for.get(p).is_some_and(Option::is_some) {
2882 continue;
2883 }
2884 let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2885 let mut best: Option<(f32, usize, usize)> = None; // (coverage, table, cell)
2886 for (t, tbl) in regions.iter().enumerate() {
2887 if !is_table_like(tbl.label) {
2888 continue;
2889 }
2890 let Some(grid) = table_rows.get(t).and_then(Option::as_ref) else {
2891 continue;
2892 };
2893 if inter(pic, tbl.l, tbl.t, tbl.r, tbl.b) / pa < RICH_CELL_PICTURE_COVERAGE {
2894 continue;
2895 }
2896 if let Some((cov, cell)) = match_picture_to_cell(pic, &grid.cells) {
2897 if best.is_none_or(|(b, _, _)| cov > b) {
2898 best = Some((cov, t, cell));
2899 }
2900 }
2901 }
2902 if let Some((_, t, cell)) = best {
2903 let entry = out.entry(t).or_default();
2904 match entry.iter_mut().find(|(c, _)| *c == cell) {
2905 Some((_, pics)) => pics.push(p),
2906 None => entry.push((cell, vec![p])),
2907 }
2908 }
2909 }
2910 out
2911}
2912
2913/// docling `_match_picture_to_table_cell`: among the cells covering ≥ 80 % of
2914/// the picture, prefer the one at the picture's inferred grid position (the
2915/// row / column whose median cell center is nearest the picture's center —
2916/// cell boxes can overlap across logical rows and columns), else the best
2917/// coverage. Returns `(coverage, cell index)`.
2918fn match_picture_to_cell(pic: &Region, cells: &[docling_core::TableCell]) -> Option<(f32, usize)> {
2919 let pa = area(pic.l, pic.t, pic.r, pic.b).max(1.0);
2920 let cover = |b: &[f32; 4]| inter(pic, b[0], b[1], b[2], b[3]) / pa;
2921 let eligible: Vec<(f32, usize)> = cells
2922 .iter()
2923 .enumerate()
2924 .filter_map(|(i, c)| {
2925 let b = c.bbox.as_ref()?;
2926 let cov = cover(b);
2927 (cov >= RICH_CELL_PICTURE_COVERAGE).then_some((cov, i))
2928 })
2929 .collect();
2930 if eligible.is_empty() {
2931 return None;
2932 }
2933 let mut row_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2934 let mut col_centers: std::collections::BTreeMap<usize, Vec<f32>> = Default::default();
2935 for c in cells {
2936 let Some(b) = c.bbox.as_ref() else { continue };
2937 for r in c.start_row..c.start_row + c.row_span {
2938 row_centers.entry(r).or_default().push((b[1] + b[3]) / 2.0);
2939 }
2940 for k in c.start_col..c.start_col + c.col_span {
2941 col_centers.entry(k).or_default().push((b[0] + b[2]) / 2.0);
2942 }
2943 }
2944 let median = |v: &mut Vec<f32>| -> f32 {
2945 v.sort_by(f32::total_cmp);
2946 let n = v.len();
2947 if n % 2 == 1 {
2948 v[n / 2]
2949 } else {
2950 (v[n / 2 - 1] + v[n / 2]) / 2.0
2951 }
2952 };
2953 let (px, py) = ((pic.l + pic.r) / 2.0, (pic.t + pic.b) / 2.0);
2954 let nearest = |centers: &mut std::collections::BTreeMap<usize, Vec<f32>>, target: f32| {
2955 centers
2956 .iter_mut()
2957 .map(|(&i, v)| (i, (median(v) - target).abs()))
2958 .min_by(|a, b| a.1.total_cmp(&b.1))
2959 .map(|(i, _)| i)
2960 };
2961 let row = nearest(&mut row_centers, py);
2962 let col = nearest(&mut col_centers, px);
2963 let logical: Vec<(f32, usize)> = eligible
2964 .iter()
2965 .copied()
2966 .filter(|&(_, i)| {
2967 let c = &cells[i];
2968 row.is_some_and(|r| c.start_row <= r && r < c.start_row + c.row_span)
2969 && col.is_some_and(|k| c.start_col <= k && k < c.start_col + c.col_span)
2970 })
2971 .collect();
2972 let pool = if logical.is_empty() {
2973 &eligible
2974 } else {
2975 &logical
2976 };
2977 // Python's `max` over `(coverage, cell_index, cell)` tuples: highest
2978 // coverage, ties to the higher index.
2979 pool.iter()
2980 .copied()
2981 .max_by(|a, b| a.0.total_cmp(&b.0).then(a.1.cmp(&b.1)))
2982}
2983
2984/// The DocLang structure overlay derived from first-class cells: span
2985/// continuations (`lcel`/`ucel`/`xcel`) and per-cell header roles, so the
2986/// PDF path's DCLX carries real spans instead of a flat grid.
2987fn structure_from_cells(
2988 cells: &[docling_core::TableCell],
2989 nrows: usize,
2990 ncols: usize,
2991) -> docling_core::TableStructure {
2992 let grid = || vec![vec![false; ncols]; nrows];
2993 let mut col_cont = grid();
2994 let mut row_cont = grid();
2995 let mut row_header = grid();
2996 let mut col_header = grid();
2997 for c in cells {
2998 for r in c.start_row..(c.start_row + c.row_span).min(nrows) {
2999 for k in c.start_col..(c.start_col + c.col_span).min(ncols) {
3000 col_cont[r][k] = k > c.start_col;
3001 row_cont[r][k] = r > c.start_row;
3002 row_header[r][k] = c.row_header;
3003 col_header[r][k] = c.column_header;
3004 }
3005 }
3006 }
3007 docling_core::TableStructure {
3008 header_row: Vec::new(),
3009 col_continuation: col_cont,
3010 row_continuation: row_cont,
3011 row_header,
3012 col_header,
3013 }
3014}
3015
3016pub fn assemble_page(
3017 page: &PdfPage,
3018 mut regions: Vec<Region>,
3019 table_rows: &[Option<TableGrid>],
3020 enrichments: &[Option<Enrichment>],
3021 // Picture-crop scale in px/pt (docling's `images_scale`, #520); `None`
3022 // keeps the page render's own scale.
3023 picture_scale: Option<f32>,
3024) -> (Vec<Node>, Vec<(String, String)>) {
3025 // Without pixels (the text-layer-only wasm build) no picture is cropped.
3026 #[cfg(not(feature = "ocr-prep"))]
3027 let _ = picture_scale;
3028 let mut nodes: Vec<Node> = Vec::new();
3029 // Every page opens with an invisible page marker carrying its size in
3030 // points — what the JSON export needs to build docling's `pages` map and
3031 // denormalize the 0–511 `<location>` grid into point bboxes (#171). The
3032 // page *number* is stamped by the document-level collector (which knows
3033 // the real 1-based index, `--pages` windows included); every serializer
3034 // except JSON skips the marker, so Markdown/DocLang stay byte-identical.
3035 nodes.push(Node::PageInfo {
3036 page_no: 0,
3037 width: page.width,
3038 height: page.height,
3039 });
3040 // Recover this page's hyperlinks (anchor-precise pairs for strict
3041 // Markdown; whole-item docling-parity links are baked below and their
3042 // pairs dropped from this list so strict output doesn't double-wrap).
3043 let mut links = resolve_link_anchors(page);
3044 // Lines with a drawn checkbox square in front become checkbox items (#609).
3045 split_checkbox_lines(&mut regions, &page.cells, &page.checkboxes);
3046 // Pair each region with its precomputed TableFormer grid and enrichment
3047 // (indexed by original order) and order by reading order together, so they
3048 // stay aligned.
3049 // A picture's children (docling's `_set_cluster_children`: the regulars
3050 // > 80 % inside it) are not page elements — they leave the reading order
3051 // here and ride with their picture, to be written under it in the JSON.
3052 let parents = picture_parents(®ions);
3053 let mut kids: Vec<Vec<Region>> = vec![Vec::new(); regions.len()];
3054 let mut top: Vec<(usize, Region)> = Vec::with_capacity(regions.len());
3055 for (i, (r, parent)) in regions.into_iter().zip(parents).enumerate() {
3056 match parent {
3057 Some(p) => kids[p].push(r),
3058 None => top.push((i, r)),
3059 }
3060 }
3061 // docling's assembly order of the regions — what its reading-order
3062 // predictor knows as `cid` (#424) — before they are shuffled.
3063 let top_regions: Vec<Region> = top.iter().map(|(_, r)| r.clone()).collect();
3064 let cids = cluster_cids(&top_regions, &page.cells);
3065 type RegionItem = (Region, Option<TableGrid>, Option<Enrichment>, Vec<Region>);
3066 let mut items: Vec<RegionItem> = top
3067 .into_iter()
3068 .map(|(i, r)| {
3069 (
3070 r,
3071 table_rows.get(i).cloned().flatten(),
3072 enrichments.get(i).cloned().flatten(),
3073 std::mem::take(&mut kids[i]),
3074 )
3075 })
3076 .collect();
3077 order_with_containers(&mut items, &cids, page.width, page.height, |it| &it.0);
3078 // Float a margin page number to the front of reading order (docling parity:
3079 // right_to_left_02's bottom `11` is its first item). Stable, so everything
3080 // else keeps its order; no-op on pages without such a region.
3081 let page_h = page.height;
3082 items.sort_by_key(|(r, _, _, _)| !is_page_number(r, &page.cells, page_h));
3083 let table_rows: Vec<Option<TableGrid>> = items.iter().map(|(_, t, _, _)| t.clone()).collect();
3084 let enrichments: Vec<Option<Enrichment>> = items.iter().map(|(_, _, e, _)| e.clone()).collect();
3085 let mut picture_children: Vec<Vec<Region>> = items
3086 .iter_mut()
3087 .map(|it| std::mem::take(&mut it.3))
3088 .collect();
3089 let regions: Vec<Region> = items.into_iter().map(|(r, _, _, _)| r).collect();
3090 // Children in docling's `_sort_clusters(mode="id")` order: first source
3091 // cell, then top, then left.
3092 for kids in picture_children.iter_mut().filter(|k| k.len() > 1) {
3093 let rank = cluster_cids(kids, &page.cells);
3094 let mut ranked: Vec<(usize, Region)> = rank.into_iter().zip(kids.drain(..)).collect();
3095 ranked.sort_by_key(|(k, _)| *k);
3096 kids.extend(ranked.into_iter().map(|(_, r)| r));
3097 }
3098 // docling emits a figure's caption *before* the image marker. Pair each
3099 // picture with the caption region nearest below it and consume that caption,
3100 // so it isn't also emitted in its own (lower) reading-order position.
3101 let caption_for = pair_captions(®ions);
3102 let code_caption_for = pair_code_captions(®ions);
3103 let mut consumed = vec![false; regions.len()];
3104 for ci in caption_for.iter().flatten() {
3105 consumed[*ci] = true;
3106 }
3107 for ci in code_caption_for.iter().flatten() {
3108 consumed[*ci] = true;
3109 }
3110 // Table captions (#265) claim from what the picture/code pairings left.
3111 let mut caption_taken = consumed.clone();
3112 let table_caption_for = pair_table_captions(®ions, &mut caption_taken);
3113 for ci in table_caption_for.iter().flatten() {
3114 consumed[*ci] = true;
3115 }
3116 // Pictures inside a table become rich-cell content (docling#3906, 2.118.1):
3117 // the picture is nested in the cell it covers and not emitted standalone.
3118 let rich_cell_pictures = match_table_pictures(®ions, &table_rows, &caption_for);
3119 for (_, pics) in rich_cell_pictures.values().flatten() {
3120 for &p in pics {
3121 consumed[p] = true;
3122 }
3123 }
3124 // A code block's language label (`XML`, `C#`, …) is chrome, not content — the
3125 // detector emits it as its own region above the code; consume it.
3126 for (i, is_label) in code_language_labels(®ions, &page.cells)
3127 .into_iter()
3128 .enumerate()
3129 {
3130 if is_label {
3131 consumed[i] = true;
3132 }
3133 }
3134
3135 // docling `ReadingOrderPredictor.predict_merges`: join a text fragment with a
3136 // following text fragment strictly to its right (an author column that wraps
3137 // into the next, a paragraph continuing in the next column) into one block —
3138 // the intra-page half of docling's reading-order merges (cross-page/vertical
3139 // continuations stay with [`merge_continuations`]). Already-consumed regions
3140 // (paired captions, code labels) are excluded.
3141 // Exclusive docling cell assignment: computed once for the ordered region
3142 // list and reused for every serialization below, so a cell can never render
3143 // in two regions. The picture children take part (docling assigns cells to
3144 // every regular cluster before it nests any); their texts are split off.
3145 let with_children: Vec<Region> = regions
3146 .iter()
3147 .chain(picture_children.iter().flatten())
3148 .cloned()
3149 .collect();
3150 let mut region_texts: Vec<String> = region_texts_exclusive(&with_children, &page.cells);
3151 let mut kid_texts = region_texts.split_off(regions.len()).into_iter();
3152 let child_texts: Vec<Vec<String>> = picture_children
3153 .iter()
3154 .map(|k| kid_texts.by_ref().take(k.len()).collect())
3155 .collect();
3156 let is_text: Vec<bool> = regions
3157 .iter()
3158 .enumerate()
3159 .map(|(i, r)| r.label == "text" && !consumed[i])
3160 .collect();
3161 let is_skip: Vec<bool> = regions
3162 .iter()
3163 .enumerate()
3164 .map(|(i, r)| {
3165 consumed[i]
3166 || matches!(
3167 r.label,
3168 "page_header" | "page_footer" | "table" | "picture" | "caption" | "footnote"
3169 )
3170 })
3171 .collect();
3172 let boxes: Vec<(f32, f32, f32, f32)> = regions.iter().map(|r| (r.l, r.t, r.r, r.b)).collect();
3173 if docling_core::env::flag("DOCLING_RS_DEBUG_MERGES") {
3174 for (i, r) in regions.iter().enumerate() {
3175 eprintln!(
3176 "MRG {i:2} {} text={} skip={} [{:.0},{:.0},{:.0},{:.0}] {:?}",
3177 r.label,
3178 is_text[i],
3179 is_skip[i],
3180 r.l,
3181 r.t,
3182 r.r,
3183 r.b,
3184 region_texts[i].chars().take(40).collect::<String>()
3185 );
3186 }
3187 }
3188 let mut merge_suffix: Vec<String> = vec![String::new(); regions.len()];
3189 for (head, children) in
3190 crate::reading_order::predict_merges(&boxes, ®ion_texts, &is_text, &is_skip)
3191 .into_iter()
3192 .enumerate()
3193 {
3194 for c in children {
3195 let t = region_texts[c].trim();
3196 if !t.is_empty() {
3197 merge_suffix[head].push(' ');
3198 merge_suffix[head].push_str(t);
3199 }
3200 consumed[c] = true;
3201 }
3202 }
3203
3204 for (i, region) in regions.iter().enumerate() {
3205 if consumed[i] {
3206 continue;
3207 }
3208 // Page headers/footers: docling emits them as furniture blocks
3209 // (`<page_header>`/`<page_footer>` with a layer + location + text) at
3210 // their reading-order position, not as body — emit them, don't skip.
3211 if matches!(region.label, "page_header" | "page_footer") {
3212 let text = region_texts[i].clone();
3213 if !text.is_empty() {
3214 nodes.push(Node::PageFurniture {
3215 footer: region.label == "page_footer",
3216 location: norm_loc(region, page.width, page_h),
3217 text: md_escape(&text),
3218 });
3219 }
3220 continue;
3221 }
3222 if is_skipped(region.label) {
3223 continue;
3224 }
3225 // Layout provenance for this region, normalized to docling's 0–511 grid.
3226 let loc = norm_loc(region, page.width, page_h);
3227 if region.label == "picture" {
3228 // The figure pixels are cropped from the page render for image export.
3229 // Captions are prose: markdown-escaped like a paragraph (the JSON
3230 // export unescapes back to the raw text, matching docling).
3231 let caption = caption_for[i]
3232 .map(|ci| md_escape(®ion_texts[ci]))
3233 .filter(|t| !t.is_empty());
3234 // The caption's own region box (#609): docling gives every caption
3235 // the cluster it came from as its `prov`, not the picture's.
3236 let caption_location = caption_for[i]
3237 .filter(|_| caption.is_some())
3238 .map(|ci| norm_loc(®ions[ci], page.width, page_h));
3239 let classification = match &enrichments[i] {
3240 Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
3241 _ => None,
3242 };
3243 // Without the page render (text-layer-only build) a picture keeps
3244 // its caption/classification but carries no cropped pixels.
3245 #[cfg(feature = "ocr-prep")]
3246 let image =
3247 crate::timing::timed("crop_region", || crop_region(page, region, picture_scale));
3248 #[cfg(not(feature = "ocr-prep"))]
3249 let image: Option<PictureImage> = None;
3250 nodes.push(located(
3251 loc,
3252 Node::Picture {
3253 caption,
3254 caption_href: None,
3255 image,
3256 classification,
3257 // docling's layout pipeline parents a figure's caption to
3258 // the picture itself (#390) — the one backend that does.
3259 caption_parent: CaptionParent::Item,
3260 caption_location,
3261 },
3262 ));
3263 let children: Vec<Node> = picture_children[i]
3264 .iter()
3265 .zip(&child_texts[i])
3266 .filter_map(|(r, text)| {
3267 picture_child_node(r, text, norm_loc(r, page.width, page_h))
3268 })
3269 .collect();
3270 if !children.is_empty() {
3271 nodes.push(Node::PictureChildren(children));
3272 }
3273 continue;
3274 }
3275 let mut text = region_texts[i].clone();
3276 text.push_str(&merge_suffix[i]);
3277 if text.is_empty() {
3278 continue;
3279 }
3280 match region.label {
3281 // docling assembles checkboxes as TEXT_ELEM items (the region's
3282 // cells are the option label, e.g. right_to_left_03's بلی/خير)
3283 // and its Markdown serializer renders them as task-list lines
3284 // (`- [x] …`) — mirrored by [`Node::CheckboxItem`].
3285 // Located like every other text item, so the JSON item carries
3286 // its page and box (#609) — a chunk of checkboxes has a page.
3287 // The label without a leading ballot-box glyph: the item's state
3288 // already says what `☐` / `☒` drew (#609).
3289 "checkbox_selected" | "checkbox_unselected" => nodes.push(located(
3290 loc,
3291 Node::CheckboxItem {
3292 checked: region.label == "checkbox_selected",
3293 text: md_escape(strip_checkbox_glyph(&text)),
3294 },
3295 )),
3296 // docling renders both the document title and section headers as
3297 // `##` (it never emits a top-level `#` for PDFs), so match that.
3298 "title" | "section_header" => nodes.push(located(
3299 loc,
3300 Node::Heading {
3301 level: 2,
3302 text: md_escape(&text),
3303 },
3304 )),
3305 // docling's `ListItemMarkerProcessor.process_list_item` runs on
3306 // every PDF list item: a leading bullet glyph or enumeration marker
3307 // followed by whitespace is split off into the item's `marker`, and
3308 // docling-core's Markdown then prints `- text` for a bullet, `N. text`
3309 // for an `N.` marker and `- a) text` for any other marker holding a
3310 // letter or digit (see [`list_item_node`]). The symbol-font bullets
3311 // docling-parse filters out of its cells are stripped first.
3312 "list_item" => nodes.push(list_item_node(&text, loc, false)),
3313 // TableFormer structure (cells + spans, text matched from word cells)
3314 // when available; otherwise geometric grid reconstruction; finally a
3315 // single cell.
3316 "table" | "document_index" => {
3317 // TableFormer grids carry first-class cells (#240: text +
3318 // page-point bbox + span rectangle + OTSL header roles) into
3319 // the public model, and the DocLang structure overlay derives
3320 // from them so DCLX emits real span/header tokens. The
3321 // geometric fallback has no per-cell records.
3322 let (mut rows, cells, structure) = match table_rows[i].clone() {
3323 Some(grid) => {
3324 let nrows = grid.rows.len();
3325 let ncols = grid.rows.first().map_or(0, Vec::len);
3326 let structure = structure_from_cells(&grid.cells, nrows, ncols);
3327 (grid.rows, Some(grid.cells), Some(structure))
3328 }
3329 None => {
3330 let rows = reconstruct_table(region, &page.cells);
3331 let rows = if rows.iter().any(|r| r.len() > 1) {
3332 rows
3333 } else {
3334 vec![vec![text.clone()]]
3335 };
3336 (rows, None, None)
3337 }
3338 };
3339 // The paired caption (#265) rides on the table — docling's
3340 // TableItem.captions ref; Markdown prints it above the grid,
3341 // the JSON export emits the $ref, DocLang the <caption>.
3342 let caption = table_caption_for[i]
3343 .map(|ci| md_escape(®ion_texts[ci]))
3344 .filter(|t| !t.is_empty());
3345 // Its own region box becomes the caption item's `prov` (#609).
3346 let caption_location = table_caption_for[i]
3347 .filter(|_| caption.is_some())
3348 .map(|ci| norm_loc(®ions[ci], page.width, page_h));
3349 // Rich cells (docling#3906): the covering cell's blocks are its
3350 // text followed by the nested picture(s). docling's Markdown
3351 // renders a `RichTableCell` through the serializer — the
3352 // group's children joined by blank lines, newlines flattened
3353 // to spaces — so the flat `rows` text becomes
3354 // `text <!-- image -->`; the first-class `cells` (the JSON
3355 // `table_cells` / `grid`) keep the plain text, as upstream.
3356 let mut cell_blocks: Option<Vec<Vec<Vec<Node>>>> = None;
3357 if let (Some(by_cell), Some(fc)) = (rich_cell_pictures.get(&i), cells.as_ref()) {
3358 let nrows = rows.len();
3359 let ncols = rows.iter().map(Vec::len).max().unwrap_or(0);
3360 let mut blocks = vec![vec![Vec::<Node>::new(); ncols]; nrows];
3361 for (cell_idx, pics) in by_cell {
3362 let cell = &fc[*cell_idx];
3363 let (r, c) = (cell.start_row, cell.start_col);
3364 if r >= nrows || c >= ncols {
3365 continue;
3366 }
3367 let mut parts: Vec<String> = Vec::new();
3368 let mut cell_nodes: Vec<Node> = Vec::new();
3369 if !cell.text.trim().is_empty() {
3370 parts.push(cell.text.clone());
3371 cell_nodes.push(Node::Paragraph {
3372 text: cell.text.clone(),
3373 });
3374 }
3375 for &p in pics {
3376 parts.push("<!-- image -->".to_string());
3377 let classification = match &enrichments[p] {
3378 Some(Enrichment::PictureClasses(classes)) => Some(classes.clone()),
3379 _ => None,
3380 };
3381 #[cfg(feature = "ocr-prep")]
3382 let image = crop_region(page, ®ions[p], picture_scale);
3383 #[cfg(not(feature = "ocr-prep"))]
3384 let image: Option<PictureImage> = None;
3385 cell_nodes.push(located(
3386 norm_loc(®ions[p], page.width, page_h),
3387 Node::Picture {
3388 caption: None,
3389 caption_href: None,
3390 image,
3391 classification,
3392 caption_parent: Default::default(),
3393 caption_location: None,
3394 },
3395 ));
3396 }
3397 let rendered = parts.join(" ");
3398 for row in rows.iter_mut().skip(r).take(cell.row_span) {
3399 for slot in row.iter_mut().skip(c).take(cell.col_span) {
3400 *slot = rendered.clone();
3401 }
3402 }
3403 blocks[r][c] = cell_nodes;
3404 }
3405 cell_blocks = Some(blocks);
3406 }
3407 nodes.push(located(
3408 loc,
3409 Node::Table(Table {
3410 rows,
3411 location: None,
3412 structure,
3413 cell_blocks,
3414 cells,
3415 caption,
3416 // As for pictures: the caption is the table's child.
3417 caption_parent: CaptionParent::Item,
3418 caption_location,
3419 }),
3420 ));
3421 }
3422 // With formula enrichment the CodeFormula model decodes the region
3423 // to LaTeX; otherwise docling emits a placeholder comment rather
3424 // than the (garbled) raw glyph text.
3425 "formula" => match &enrichments[i] {
3426 Some(Enrichment::Formula { latex }) => nodes.push(Node::Formula {
3427 latex: latex.clone(),
3428 orig: text.clone(),
3429 location: Some(loc),
3430 }),
3431 _ => nodes.push(Node::Paragraph {
3432 text: "<!-- formula-not-decoded -->".into(),
3433 }),
3434 },
3435 // Code blocks: use the space-glyph-only grouping (monospace keeps its
3436 // source spacing) and emit a fenced block, preserving the line breaks
3437 // and indentation of the source (unlike prose, which reflows). pdfium
3438 // still inserts spaces around tight punctuation (`console .log`,
3439 // `add (3 , 5)`); tighten them to match docling-parse's source spacing.
3440 "code" => {
3441 // `code_region_text` preserves line breaks/indentation and tightens
3442 // each line itself; the fallback prose `text` is tightened here.
3443 let code = code_region_text(region, &page.code_cells);
3444 let code = if code.is_empty() {
3445 tighten_code_punct(&text)
3446 } else {
3447 code
3448 };
3449 // With code enrichment the CodeFormula model rewrites the block
3450 // (and names its language); `orig` keeps the raw extraction in
3451 // docling's shape — its parser has no line-preserving code
3452 // path, so its `orig` is the same code with the lines joined
3453 // by single spaces (indentation collapsed).
3454 // docling's parser has no line-preserving code path — its code
3455 // items carry the lines joined by single spaces. That flat
3456 // form is what every byte-conformance surface serializes
3457 // (legacy Markdown, JSON, DocLang); the line-preserving
3458 // extraction rides in `pretty` for strict Markdown only.
3459 let flat = code
3460 .lines()
3461 .map(str::trim)
3462 .filter(|l| !l.is_empty())
3463 .collect::<Vec<_>>()
3464 .join(" ");
3465 let node = match &enrichments[i] {
3466 Some(Enrichment::Code {
3467 language,
3468 text: enriched,
3469 }) => Node::Code {
3470 language: language.clone(),
3471 text: enriched.clone(),
3472 orig: Some(flat),
3473 pretty: None,
3474 },
3475 _ => Node::Code {
3476 language: None,
3477 text: flat,
3478 orig: None,
3479 pretty: Some(code),
3480 },
3481 };
3482 nodes.push(located(loc, node));
3483 // docling emits the `Listing N:` caption after the code block.
3484 if let Some(ci) = code_caption_for[i] {
3485 let cap = md_escape(®ion_texts[ci]);
3486 if !cap.is_empty() {
3487 // With its own region box, like every caption (#609).
3488 nodes.push(located(
3489 norm_loc(®ions[ci], page.width, page_h),
3490 Node::Paragraph { text: cap },
3491 ));
3492 }
3493 }
3494 }
3495 // text, caption, footnote → paragraph
3496 _ => {
3497 // docling parity (`PageAssembleModel._match_hyperlink`): when
3498 // link annotations cover ≥ half of the region's box, the
3499 // hyperlink attaches to the item and the legacy Markdown
3500 // serializer wraps its full text — 2206.01062's footnote URLs
3501 // render as `[1 https://…](https://…)`. Sparse in-paragraph
3502 // citation links stay below the 0.5 coverage threshold and
3503 // remain plain text, exactly like docling.
3504 //
3505 // Scope: **footnote regions only.** Upstream's page_assemble
3506 // matches every TEXT_ELEM label, but published docling
3507 // observably carries the hyperlink into the document only for
3508 // footnote items — in both committed groundtruth generations
3509 // (docling-JSON and Markdown, independent runs) the fully
3510 // covered plain-text DOI line of 2206.01062 page 1 has
3511 // `hyperlink: None` while the equally covered footnotes carry
3512 // theirs. The corpus is the conformance reference, so match
3513 // the observed behavior; widen the label set if a future
3514 // groundtruth refresh starts linking plain text too.
3515 let escaped = md_escape(&text);
3516 let hyperlink = (region.label == "footnote")
3517 .then(|| region_hyperlink(region, &page.links))
3518 .flatten();
3519 if let Some(uri) = &hyperlink {
3520 // The strict-mode anchor pairs this item covers are
3521 // superseded by the whole-item link.
3522 links.retain(|(anchor, href)| {
3523 !(href == uri && region_texts[i].contains(anchor.as_str()))
3524 });
3525 }
3526 // A footnote keeps docling's label and carries its link as
3527 // the item's `hyperlink` (the JSON's `text` stays the raw
3528 // footnote, not Markdown); every other serializer renders it
3529 // as the `[text](uri)` paragraph it was before.
3530 let node = if region.label == "footnote" {
3531 Node::LabeledText {
3532 label: "footnote".into(),
3533 text: escaped,
3534 href: hyperlink,
3535 }
3536 } else {
3537 Node::Paragraph { text: escaped }
3538 };
3539 nodes.push(located(loc, node))
3540 }
3541 }
3542 }
3543 // A `/Rotate`-normalized scanned page (see `pdfium_backend`) was assembled
3544 // in upright space; rotate the finished geometry back so locations and the
3545 // page size are display-space, like docling and every viewer report them.
3546 if page.rotation != 0 {
3547 rotate_nodes_to_display(&mut nodes, page.rotation);
3548 }
3549 (nodes, links)
3550}
3551
3552/// One child of a picture as docling's `ReadingOrderModel._add_child_elements`
3553/// writes it under the `PictureItem`: a heading for a `section_header` /
3554/// `title` (upstream remaps title to section header), a list item for a
3555/// `list_item`, a furniture-layer text for a page header/footer, a caption,
3556/// otherwise a text item. `None` for a child that claimed no text.
3557fn picture_child_node(region: &Region, text: &str, loc: [u16; 4]) -> Option<Node> {
3558 if text.is_empty() {
3559 return None;
3560 }
3561 Some(match region.label {
3562 "title" | "section_header" => located(
3563 loc,
3564 Node::Heading {
3565 level: 2,
3566 text: md_escape(text),
3567 },
3568 ),
3569 // docling-core's `add_list_item` under a non-list parent opens a
3570 // list group per item, so every child item starts its own list;
3571 // `_add_child_elements` runs the marker processor on it too.
3572 "list_item" => list_item_node(text, loc, true),
3573 "page_header" | "page_footer" => Node::PageFurniture {
3574 footer: region.label == "page_footer",
3575 location: loc,
3576 text: md_escape(text),
3577 },
3578 "caption" => located(
3579 loc,
3580 Node::Caption {
3581 text: md_escape(text),
3582 href: None,
3583 },
3584 ),
3585 _ => located(
3586 loc,
3587 Node::Paragraph {
3588 text: md_escape(text),
3589 },
3590 ),
3591 })
3592}
3593
3594/// Rotate one 0–511 location bbox 90° clockwise on the grid (top-left origin):
3595/// `(x, y) → (511 - y, x)`.
3596fn rot_loc_cw(l: [u16; 4]) -> [u16; 4] {
3597 [511 - l[3], l[0], 511 - l[1], l[2]]
3598}
3599
3600/// Map upright-space geometry back to display space for a page whose `/Rotate`
3601/// was normalized away before inference: every `<location>` rotates `rot`°
3602/// clockwise on the 0–511 grid (the grid is per-axis normalized, so no page
3603/// dims are needed), and the `PageInfo` size returns to the display box. Node
3604/// text and order are untouched — reading order was decided upright, which is
3605/// the whole point.
3606fn rotate_nodes_to_display(nodes: &mut [Node], rot: u16) {
3607 let quarter_turns = (rot / 90) as usize;
3608 let rot_loc = |l: &mut [u16; 4]| {
3609 for _ in 0..quarter_turns {
3610 *l = rot_loc_cw(*l);
3611 }
3612 };
3613 fn walk(node: &mut Node, rot_loc: &impl Fn(&mut [u16; 4]), swap_dims: bool) {
3614 match node {
3615 Node::PageInfo { width, height, .. } => {
3616 if swap_dims {
3617 std::mem::swap(width, height);
3618 }
3619 }
3620 Node::Located { location, inner } => {
3621 rot_loc(location);
3622 walk(inner, rot_loc, swap_dims);
3623 }
3624 Node::Furniture { inner, .. } => walk(inner, rot_loc, swap_dims),
3625 Node::Group { children, .. } | Node::PictureChildren(children) => {
3626 for c in children {
3627 walk(c, rot_loc, swap_dims);
3628 }
3629 }
3630 Node::ListItem { location, .. }
3631 | Node::Formula { location, .. }
3632 | Node::Chart { location, .. } => {
3633 if let Some(l) = location {
3634 rot_loc(l);
3635 }
3636 }
3637 Node::PageFurniture { location, .. } => rot_loc(location),
3638 Node::Table(t) => {
3639 if let Some(l) = &mut t.location {
3640 rot_loc(l);
3641 }
3642 }
3643 _ => {}
3644 }
3645 }
3646 let swap_dims = quarter_turns % 2 == 1;
3647 for node in nodes {
3648 walk(node, &rot_loc, swap_dims);
3649 }
3650}
3651
3652/// Merge paragraph fragments split across a column or page break. docling joins a
3653/// paragraph whose previous fragment ends mid-sentence (a letter, not sentence
3654/// punctuation) with a lowercase continuation: `…definition of` + `lists in…` →
3655/// `…definition of lists in…`. The fragments are consecutive paragraphs, or
3656/// separated only by figure(s) the text wraps around: a column whose body flows
3657/// past a figure resumes below it (`…The wing type that is` ⟶[figure]⟶ `the most
3658/// common…`), and docling emits the whole paragraph before the figure. A heading,
3659/// table, or list between them ends the paragraph (no merge).
3660/// A paragraph that is really a figure/table caption (`Fig. 1. …`, `Table 2 …`).
3661/// Used to skip an unpaired caption when stitching a paragraph that wraps around
3662/// a figure.
3663fn looks_like_caption(text: &str) -> bool {
3664 let head: String = text.trim_start().chars().take(14).collect();
3665 (head.starts_with("Fig") || head.starts_with("Table"))
3666 && head.contains(|c: char| c.is_ascii_digit())
3667}
3668
3669/// A paragraph fragment is "open" — i.e. it might continue into the next
3670/// paragraph — when it ends mid-word (a letter) or with a wrap hyphen/dash.
3671/// docling joins `vocab-` + `ulary` → `vocab- ulary`.
3672fn paragraph_is_open(text: &str) -> bool {
3673 // docling's merge head test (`.+([a-z,\-\u00AD])\s*`): at least two chars,
3674 // ending in an ASCII lowercase letter, a comma, a hyphen, or a soft
3675 // hyphen. The comma matters: 2206's "…In phase four," resumes across the
3676 // page break. Uppercase/non-Latin endings do not merge, exactly as
3677 // upstream (the dash family is already `-` here — clean_text normalized).
3678 let t = text.trim_end();
3679 t.chars().count() >= 2
3680 && t.chars()
3681 .next_back()
3682 .is_some_and(|c| matches!(c, 'a'..='z' | ',' | '-' | '\u{ad}'))
3683}
3684
3685/// The paragraph text inside a node, looking through a [`Node::Located`]
3686/// provenance wrapper (PDF body paragraphs are wrapped since they carry a
3687/// `<location>`). Returns `None` for non-paragraph nodes.
3688fn as_paragraph(n: &Node) -> Option<&str> {
3689 match n {
3690 Node::Paragraph { text } => Some(text),
3691 Node::Located { inner, .. } => match inner.as_ref() {
3692 Node::Paragraph { text } => Some(text),
3693 _ => None,
3694 },
3695 _ => None,
3696 }
3697}
3698
3699/// Whether a node is a picture, looking through a [`Node::Located`] wrapper.
3700fn is_picture_node(n: &Node) -> bool {
3701 match n {
3702 Node::Picture { .. } => true,
3703 Node::Located { inner, .. } => matches!(inner.as_ref(), Node::Picture { .. }),
3704 _ => false,
3705 }
3706}
3707
3708/// A node a forward paragraph merge looks straight past: a figure or *table*
3709/// the text wraps around, or a page header/footer that falls between the two
3710/// fragments of a paragraph continuing across a page break (docling's merge
3711/// skip-labels: page_header, page_footer, table, picture, caption, footnote —
3712/// 2206's "…In phase four," resumes after a full caption+table+figure block).
3713fn is_merge_trailer(n: &Node) -> bool {
3714 is_picture_node(n)
3715 || matches!(
3716 n,
3717 Node::PageFurniture { .. }
3718 | Node::PageInfo { .. }
3719 | Node::Table(_)
3720 | Node::PictureChildren(_)
3721 )
3722 || matches!(n, Node::Located { inner, .. } if matches!(inner.as_ref(), Node::Table(_)))
3723 || is_footnote_node(n)
3724 || as_paragraph(n).is_some_and(looks_like_caption)
3725}
3726
3727/// Whether a node is a footnote item (a [`Node::LabeledText`] labelled
3728/// `footnote`), looking through a [`Node::Located`] wrapper — one of
3729/// docling's merge skip-labels: a footnote between the two halves of a
3730/// paragraph is looked past, never merged into.
3731fn is_footnote_node(n: &Node) -> bool {
3732 let n = match n {
3733 Node::Located { inner, .. } => inner.as_ref(),
3734 other => other,
3735 };
3736 matches!(n, Node::LabeledText { label, .. } if label == "footnote")
3737}
3738
3739/// Rebuild node `i` as a paragraph with `text`, preserving its `<location>`
3740/// wrapper (and thus provenance) if it had one.
3741fn reparagraph(node: &Node, text: String) -> Node {
3742 match node {
3743 Node::Located { location, .. } => located(*location, Node::Paragraph { text }),
3744 _ => Node::Paragraph { text },
3745 }
3746}
3747
3748pub(crate) fn merge_continuations(nodes: &mut Vec<Node>) {
3749 let mut i = 0;
3750 while i + 1 < nodes.len() {
3751 let Some(a) = as_paragraph(&nodes[i]) else {
3752 i += 1;
3753 continue;
3754 };
3755 // A figure/table caption is a self-contained unit; body text resuming
3756 // after a figure is the continuation case, not the caption itself. Never
3757 // stitch *from* a caption — otherwise a caption that ends in a lone glyph
3758 // (`Fig. 5. … PubTabNet. μ`) would swallow a following stray figure label
3759 // (a standalone `μ`) into `… μ μ`.
3760 if looks_like_caption(a) {
3761 i += 1;
3762 continue;
3763 }
3764 if !paragraph_is_open(a) {
3765 i += 1;
3766 continue;
3767 }
3768 // The continuation is the next paragraph, looking past any figures the
3769 // text wraps around — and a figure/table caption that was emitted as its
3770 // own paragraph (an above-the-figure caption that didn't pair), since the
3771 // body text resumes after the whole figure+caption block.
3772 let mut j = i + 1;
3773 while nodes.get(j).is_some_and(is_merge_trailer) {
3774 j += 1;
3775 }
3776 // docling's continuation regex allows either case, but its merge runs
3777 // over the pre-assembly element stream; at node level an uppercase
3778 // start is overwhelmingly a new sentence/heading fragment (allowing it
3779 // swallowed 2305's formula blocks and redp's chapter openers), so the
3780 // continuation stays lowercase-start here.
3781 let cont = nodes.get(j).and_then(as_paragraph).is_some_and(|b| {
3782 b.trim_start()
3783 .chars()
3784 .next()
3785 .is_some_and(char::is_lowercase)
3786 });
3787 if cont {
3788 let a = as_paragraph(&nodes[i]).unwrap().trim_end().to_string();
3789 let b = as_paragraph(&nodes[j]).unwrap().trim_start().to_string();
3790 // A soft hyphen -- or a hard hyphen followed by a lowercase
3791 // continuation (guaranteed lowercase by the `cont` gate above) --
3792 // is a word split across the break: strip it and join without a
3793 // space, docling#3888 ("vocab-" + "ulary" -> "vocabulary");
3794 // docling's older serializer kept the artifact ("vocab- ulary").
3795 // Everything else joins with the space, as before.
3796 let merged = match a.strip_suffix('\u{ad}').or_else(|| a.strip_suffix('-')) {
3797 Some(stem) => format!("{stem}{b}"),
3798 None => format!("{a} {b}"),
3799 };
3800 // Keep node i's provenance wrapper; docling's merged paragraph keeps
3801 // the first fragment's geometry as its primary location.
3802 nodes[i] = reparagraph(&nodes[i], merged);
3803 nodes.remove(j);
3804 // Re-check i: the merged paragraph may continue further.
3805 } else {
3806 i += 1;
3807 }
3808 }
3809}
3810
3811/// How many leading nodes of `nodes` are safe to flush now — i.e. cannot be
3812/// rewritten by a future [`merge_continuations`] once more pages are appended.
3813///
3814/// A forward merge can only start from an "open" paragraph (ends mid-word) and
3815/// only reaches across trailing pictures and figure/table captions. So we scan
3816/// from the end past those skippable trailers: if the first non-skippable node is
3817/// an open paragraph, it (and the trailers after it) must be held; anything else —
3818/// a closed paragraph, a heading, a table, a list — blocks any forward merge, so
3819/// the whole buffer is safe to flush.
3820fn hold_start(nodes: &[Node]) -> usize {
3821 for k in (0..nodes.len()).rev() {
3822 // Skippable trailers (figures, page furniture, captions): a forward merge
3823 // looks straight past them.
3824 if is_merge_trailer(&nodes[k]) {
3825 continue;
3826 }
3827 match as_paragraph(&nodes[k]) {
3828 // An open body paragraph might still pull a continuation off the next
3829 // page — hold from here to the end.
3830 Some(text) if paragraph_is_open(text) => return k,
3831 // A closed paragraph, heading, table, list, etc. ends the paragraph:
3832 // nothing after it can merge backwards across it. Flush everything.
3833 _ => return nodes.len(),
3834 }
3835 }
3836 // Only skippable trailers (or empty) and no open paragraph to anchor a merge.
3837 nodes.len()
3838}
3839
3840/// Streaming counterpart of [`merge_continuations`]: feed per-page node batches in
3841/// document order and get back the prefix that is final (its cross-page merges are
3842/// resolved and no future page can change it), holding back only the small tail
3843/// that might still merge into the next page. Concatenating every flushed batch
3844/// (then [`finish`](Self::finish)) yields exactly the same nodes as running
3845/// [`merge_continuations`] once over the whole document.
3846pub(crate) struct StreamAssembler {
3847 pending: Vec<Node>,
3848}
3849
3850impl StreamAssembler {
3851 pub(crate) fn new() -> Self {
3852 Self {
3853 pending: Vec::new(),
3854 }
3855 }
3856
3857 /// Append one page's nodes, resolve merges within the buffer, and return the
3858 /// now-final prefix to emit (possibly empty).
3859 pub(crate) fn push(&mut self, mut nodes: Vec<Node>) -> Vec<Node> {
3860 self.pending.append(&mut nodes);
3861 merge_continuations(&mut self.pending);
3862 let cut = hold_start(&self.pending);
3863 let tail = self.pending.split_off(cut);
3864 std::mem::replace(&mut self.pending, tail)
3865 }
3866
3867 /// Flush whatever is left after the last page (the held tail is final once no
3868 /// more pages can follow).
3869 pub(crate) fn finish(self) -> Vec<Node> {
3870 self.pending
3871 }
3872}
3873
3874#[cfg(test)]
3875mod tests {
3876 use super::{cells_text, clean_text, merge_overlapping_regulars, reclaim_heading_body_footers};
3877
3878 /// docling drops a picture covering > 90 % of the page (its labels then
3879 /// read out as text); a dominant-but-not-full figure and any other label
3880 /// stay whatever their size.
3881 #[test]
3882 fn full_page_pictures_are_dropped_like_docling() {
3883 use super::drop_full_page_pictures;
3884 use crate::layout::Region;
3885 let region = |label: &'static str, l, t, r, b| Region {
3886 label,
3887 score: 0.99,
3888 l,
3889 t,
3890 r,
3891 b,
3892 };
3893 let mut regions = vec![
3894 region("picture", 0.0, 0.5, 478.9, 241.8),
3895 region("picture", 10.0, 10.0, 400.0, 200.0),
3896 region("table", 0.0, 0.0, 480.0, 243.0),
3897 region("text", 5.0, 5.0, 100.0, 20.0),
3898 ];
3899 drop_full_page_pictures(&mut regions, 480.75, 243.75);
3900 let labels: Vec<_> = regions.iter().map(|r| (r.label, r.l)).collect();
3901 assert_eq!(
3902 labels,
3903 vec![("picture", 10.0), ("table", 0.0), ("text", 5.0)]
3904 );
3905 }
3906 use super::{code_region_text, merge_continuations, resolve_link_anchors, StreamAssembler};
3907 use crate::layout::Region;
3908 use crate::pdfium_backend::{LinkAnnot, PdfPage, TextCell};
3909 use docling_core::Node;
3910
3911 /// The int8-layout guard's coverage metric: cells under detections count,
3912 /// cells outside don't, whitespace cells are ignored, and a cell-less page
3913 /// reads as fully covered (nothing to rescue).
3914 #[test]
3915 fn layout_cell_coverage_counts_claimed_text_cells() {
3916 let cell = |text: &str, l: f32, t: f32| TextCell {
3917 text: text.into(),
3918 l,
3919 t,
3920 r: l + 40.0,
3921 b: t + 10.0,
3922 };
3923 let region = Region {
3924 label: "text",
3925 score: 0.9,
3926 l: 0.0,
3927 t: 0.0,
3928 r: 100.0,
3929 b: 50.0,
3930 };
3931 let cells = vec![
3932 cell("inside", 10.0, 10.0),
3933 cell("also inside", 10.0, 30.0),
3934 cell("outside", 10.0, 200.0),
3935 cell(" ", 10.0, 210.0), // whitespace: not counted at all
3936 ];
3937 let cov = super::layout_cell_coverage(std::slice::from_ref(®ion), &cells);
3938 assert!((cov - 2.0 / 3.0).abs() < 1e-6, "got {cov}");
3939 assert_eq!(super::layout_cell_coverage(&[], &[]), 1.0);
3940 assert_eq!(super::layout_cell_coverage(&[], &cells), 0.0);
3941 }
3942
3943 /// #165: a picture no longer claims cells at 0.2 intersection-over-self.
3944 /// A line straddling the figure border (≤80 % contained) becomes an orphan
3945 /// region and is emitted as page text — before the fix its cells were
3946 /// silently erased. A line fully inside the picture is the picture's child
3947 /// (docling's `_set_cluster_children`): it survives the containment drop,
3948 /// leaves the page's reading order, and is written only under the picture
3949 /// in the JSON — never in the Markdown, like docling's picture serializer.
3950 #[test]
3951 fn border_straddlers_are_page_text_picture_interior_is_a_picture_child() {
3952 let pic = Region {
3953 label: "picture",
3954 score: 0.9,
3955 l: 0.0,
3956 t: 0.0,
3957 r: 100.0,
3958 b: 100.0,
3959 };
3960 // ~35 % of this cell overlaps the picture (l=90..120 of 0..100): above
3961 // the old 0.2 claim (was swallowed), below full containment (survives).
3962 let straddler = TextCell {
3963 text: "axis label".into(),
3964 l: 90.0,
3965 t: 40.0,
3966 r: 120.0,
3967 b: 48.0,
3968 };
3969 let interior = TextCell {
3970 text: "in-figure callout".into(),
3971 l: 10.0,
3972 t: 10.0,
3973 r: 60.0,
3974 b: 18.0,
3975 };
3976 let cells = vec![straddler, interior];
3977 let mut regions = vec![pic];
3978 super::add_orphan_regions(&mut regions, &cells);
3979 super::drop_contained_regulars(&mut regions);
3980 assert_eq!(
3981 regions.iter().filter(|r| r.label == "text").count(),
3982 2,
3983 "both unclaimed lines become orphans, and a picture swallows neither"
3984 );
3985 let parents = super::picture_parents(®ions);
3986 let parent_of = |l: f32| {
3987 regions
3988 .iter()
3989 .zip(&parents)
3990 .find(|(r, _)| r.label == "text" && r.l == l)
3991 .and_then(|(_, p)| *p)
3992 };
3993 assert_eq!(
3994 parent_of(10.0),
3995 Some(0),
3996 "the callout is the picture's child"
3997 );
3998 assert_eq!(parent_of(90.0), None, "the straddler is a page element");
3999
4000 let page = PdfPage::from_cells(200.0, 200.0, 2.0, cells);
4001 let n = regions.len();
4002 let (nodes, _) = super::assemble_page(&page, regions, &vec![None; n], &vec![None; n], None);
4003 let children: Vec<&Node> = nodes
4004 .iter()
4005 .filter_map(|n| match n {
4006 Node::PictureChildren(c) => Some(c),
4007 _ => None,
4008 })
4009 .flatten()
4010 .collect();
4011 assert!(
4012 matches!(children.as_slice(), [Node::Located { inner, .. }]
4013 if matches!(inner.as_ref(), Node::Paragraph { text } if text == "in-figure callout")),
4014 "{children:?}"
4015 );
4016 let mut doc = docling_core::DoclingDocument::new("t");
4017 doc.nodes = nodes;
4018 let md = doc.export_to_markdown();
4019 assert!(md.contains("axis label"), "{md}");
4020 assert!(!md.contains("in-figure callout"), "{md}");
4021 let json = doc.export_to_json_value();
4022 let pic = &json["pictures"][0];
4023 let child = pic["children"][0]["$ref"].as_str().expect("a child ref");
4024 let idx: usize = child.rsplit('/').next().unwrap().parse().unwrap();
4025 assert_eq!(json["texts"][idx]["text"], "in-figure callout");
4026 assert_eq!(json["texts"][idx]["parent"]["$ref"], "#/pictures/0");
4027 assert_eq!(json["texts"][idx]["content_layer"], "body");
4028 }
4029
4030 /// docling#3906's concern, pinned on our side: a picture detected fully
4031 /// inside a table region must survive the containment drop (upstream now
4032 /// attaches it to the table's cell; we keep it as a body sibling — either
4033 /// way it must not vanish). The text region inside the same table is the
4034 /// control: regulars are the ones the drop swallows.
4035 #[test]
4036 fn picture_inside_a_table_region_survives_the_containment_drop() {
4037 let mut regions = vec![
4038 region("table", 0.9, 0.0, 0.0, 200.0, 200.0),
4039 region("picture", 0.9, 20.0, 20.0, 120.0, 120.0),
4040 region("text", 0.9, 20.0, 140.0, 180.0, 180.0),
4041 ];
4042 super::drop_contained_regulars(&mut regions);
4043 let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
4044 assert_eq!(
4045 labels,
4046 ["table", "picture"],
4047 "the in-table picture stays; the in-table regular is the special's child"
4048 );
4049 }
4050
4051 /// Table–caption pairing (#265) is reading-order adjacency, docling's
4052 /// `_find_to_captions`: a caption binds the table directly next to it in
4053 /// the region sequence — above-caption and below-caption both work, and
4054 /// geometry is irrelevant (a same-page caption in the other column of a
4055 /// two-column layout is *not* adjacent, however close its box is). A
4056 /// caption with media on both sides, or separated from the table by a
4057 /// text paragraph, stays unattached.
4058 #[test]
4059 fn table_captions_pair_by_reading_order_adjacency() {
4060 // caption → table (above-caption), then table → caption (below-caption),
4061 // then a caption fenced off by a paragraph, then one between two tables.
4062 let regions = vec![
4063 region("text", 0.9, 0.0, 0.0, 100.0, 10.0), // 0 body text
4064 region("caption", 0.9, 0.0, 12.0, 60.0, 20.0), // 1 above-caption
4065 region("table", 0.9, 20.0, 22.0, 90.0, 60.0), // 2 ← pairs with 1
4066 region("table", 0.9, 0.0, 70.0, 100.0, 110.0), // 3 ← pairs with 4
4067 region("caption", 0.9, 0.0, 112.0, 60.0, 120.0), // 4 below-caption
4068 region("text", 0.9, 0.0, 130.0, 100.0, 140.0), // 5 body text
4069 region("caption", 0.9, 0.0, 142.0, 60.0, 150.0), // 6 fenced by 5/7
4070 region("text", 0.9, 0.0, 152.0, 100.0, 162.0), // 7 body text
4071 region("table", 0.9, 0.0, 170.0, 100.0, 200.0), // 8 unpaired
4072 region("caption", 0.9, 0.0, 202.0, 60.0, 210.0), // 9 ambiguous
4073 region("table", 0.9, 0.0, 212.0, 100.0, 240.0), // 10 unpaired
4074 ];
4075 let mut taken = vec![false; regions.len()];
4076 let pairs = super::pair_table_captions(®ions, &mut taken);
4077 assert_eq!(pairs[2], Some(1), "caption directly above its table pairs");
4078 assert_eq!(pairs[3], Some(4), "caption directly below its table pairs");
4079 assert_eq!(
4080 pairs[8], None,
4081 "a text paragraph between caption and table breaks the bond"
4082 );
4083 assert_eq!(
4084 pairs[10], None,
4085 "a caption between two tables is ambiguous and stays loose"
4086 );
4087 assert!(taken[1] && taken[4] && !taken[6] && !taken[9]);
4088 }
4089
4090 /// A colored terms-and-conditions panel detected as `picture` demotes into
4091 /// per-paragraph `text` regions (the blank line between C.7 and C.8 splits
4092 /// them); a chart whose only text is a few narrow axis labels keeps its
4093 /// crop untouched.
4094 #[test]
4095 fn text_panels_demote_to_paragraphs_but_charts_keep_their_crop() {
4096 let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
4097 text: text.to_string(),
4098 l,
4099 t,
4100 r,
4101 b,
4102 };
4103 let panel = Region {
4104 label: "picture",
4105 score: 0.9,
4106 l: 0.0,
4107 t: 0.0,
4108 r: 100.0,
4109 b: 100.0,
4110 };
4111 // Three tight lines, a blank-line gap, two more: two paragraphs.
4112 let cells = vec![
4113 cell(
4114 "C.7. Wenn Sie diesen Vertrag widerrufen,",
4115 5.0,
4116 10.0,
4117 95.0,
4118 18.0,
4119 ),
4120 cell(
4121 "haben wir Ihnen alle Zahlungen, die wir",
4122 5.0,
4123 20.0,
4124 95.0,
4125 28.0,
4126 ),
4127 cell(
4128 "von Ihnen erhalten haben, zurückzuzahlen.",
4129 5.0,
4130 30.0,
4131 90.0,
4132 38.0,
4133 ),
4134 cell(
4135 "C.8. Wir können die Rückzahlung verweigern,",
4136 5.0,
4137 52.0,
4138 95.0,
4139 60.0,
4140 ),
4141 cell(
4142 "bis wir die Waren wieder zurückerhalten haben.",
4143 5.0,
4144 62.0,
4145 92.0,
4146 70.0,
4147 ),
4148 ];
4149 let mut regions = vec![panel.clone()];
4150 super::recover_text_panels(&mut regions, &cells);
4151 assert_eq!(
4152 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
4153 ["text", "text"],
4154 "dense panel must demote into one text region per paragraph"
4155 );
4156 assert!(regions[0].b < regions[1].t, "paragraphs split at the gap");
4157 // Sparse narrow labels (a chart): picture survives.
4158 let labels = vec![
4159 cell("0", 5.0, 90.0, 8.0, 95.0),
4160 cell("50", 5.0, 50.0, 10.0, 55.0),
4161 cell("100", 5.0, 10.0, 12.0, 15.0),
4162 cell("t, s", 45.0, 96.0, 55.0, 100.0),
4163 ];
4164 let mut regions = vec![panel];
4165 super::recover_text_panels(&mut regions, &labels);
4166 assert_eq!(
4167 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
4168 ["picture"]
4169 );
4170 }
4171
4172 /// An uncaptioned chart on a scanned page whose title, axis labels, and
4173 /// OCR boxes over the plot area are dense and wide enough to pass the
4174 /// coverage/width gates still keeps its crop: its line heights are ragged
4175 /// (title face vs tick labels vs bar-area OCR), failing the uniform-leading
4176 /// gate — a real text panel is set with constant leading (#173).
4177 #[test]
4178 fn dense_titled_chart_keeps_its_crop() {
4179 let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
4180 text: text.to_string(),
4181 l,
4182 t,
4183 r,
4184 b,
4185 };
4186 let chart = Region {
4187 label: "picture",
4188 score: 0.9,
4189 l: 0.0,
4190 t: 0.0,
4191 r: 100.0,
4192 b: 100.0,
4193 };
4194 // Five wide lines at wildly different heights: a 12-pt title, 20-pt OCR
4195 // boxes over the bars, 4–5-pt tick/axis labels. Coverage and median
4196 // width both clear the panel thresholds.
4197 let cells = vec![
4198 cell("Underground Water Storage", 10.0, 5.0, 90.0, 17.0),
4199 cell("aquifer recharge zone", 15.0, 30.0, 75.0, 50.0),
4200 cell("confined | unconfined | perched", 12.0, 55.0, 80.0, 59.0),
4201 cell("saturated thickness", 8.0, 70.0, 60.0, 90.0),
4202 cell("distance from well, km", 20.0, 92.0, 85.0, 97.0),
4203 ];
4204 let mut regions = vec![chart];
4205 super::recover_text_panels(&mut regions, &cells);
4206 assert_eq!(
4207 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
4208 ["picture"],
4209 "ragged line heights mark a figure, not a text panel"
4210 );
4211 }
4212
4213 /// docling serializes a cluster's cells in docling-parse index order
4214 /// (`_sort_cells`) and joins them with `PageAssembleModel.sanitize_text`:
4215 /// a space after every line except one ending in `-`, which either fuses a
4216 /// wrapped word (alnum on both sides — dash dropped) or glues verbatim (a
4217 /// bare `-` cell: `[0000` `-` `0002` → `[0000 -0002`, the 2305 ORCID line;
4218 /// `-` + `"C" cell -` + `a new table cell` → `-"C" cell a new table cell`,
4219 /// its OTSL list). Verified against the corpus: pure index order beats any
4220 /// geometric re-sort (normal_4pages' heading numerals paint after their
4221 /// text and belong last: `## 들어가며 1`).
4222 #[test]
4223 fn cells_join_in_index_order_with_sanitize_text_rules() {
4224 let cell = |text: &str, l: f32, t: f32, r: f32, b: f32| TextCell {
4225 text: text.to_string(),
4226 l,
4227 t,
4228 r,
4229 b,
4230 };
4231 let region = Region {
4232 label: "text",
4233 score: 1.0,
4234 l: 0.0,
4235 t: 95.0,
4236 r: 200.0,
4237 b: 130.0,
4238 };
4239 // ORCID superscript: a bare dash cell is a *detached* dash — kept, and
4240 // since docling#4052 (2.122) it joins with the ordinary space on both
4241 // sides (`[0000 -0002 -6960]` before that fix).
4242 let orcid = vec![
4243 cell("[0000", 10.0, 100.0, 30.0, 110.0),
4244 cell("−", 30.0, 100.0, 34.0, 110.0),
4245 cell("0002", 34.0, 100.0, 50.0, 110.0),
4246 cell("−", 50.0, 100.0, 54.0, 110.0),
4247 cell("6960]", 54.0, 100.0, 70.0, 110.0),
4248 ];
4249 assert_eq!(super::region_text(®ion, &orcid), "[0000 - 0002 - 6960]");
4250 // Wrapped word: dash dropped, lines fused (both boundary words alnum).
4251 let wrapped = vec![
4252 cell("platforms-", 10.0, 100.0, 60.0, 110.0),
4253 cell("reflects the design", 10.0, 112.0, 90.0, 122.0),
4254 ];
4255 assert_eq!(
4256 super::region_text(®ion, &wrapped),
4257 "platformsreflects the design"
4258 );
4259 // Dash-ending lines that are *detached* dashes (a bare bullet cell, a
4260 // `cell -` separator): the dash stays and the lines join with a space
4261 // — docling#4052; before it they glued (`-"C" cell a new table cell`,
4262 // 2305's OTSL list bullets).
4263 let otsl = vec![
4264 cell("–", 10.0, 100.0, 14.0, 110.0),
4265 cell("\"C\" cell -", 16.0, 100.0, 60.0, 110.0),
4266 cell("a new table cell", 10.0, 112.0, 80.0, 122.0),
4267 ];
4268 assert_eq!(
4269 super::region_text(®ion, &otsl),
4270 "- \"C\" cell - a new table cell"
4271 );
4272 // Index order is authoritative — no geometric re-sort.
4273 let numeral = vec![
4274 cell("들어가며", 30.0, 100.0, 80.0, 110.0),
4275 cell("1", 10.0, 98.0, 25.0, 112.0), // big numeral painted last
4276 ];
4277 assert_eq!(super::region_text(®ion, &numeral), "들어가며 1");
4278 }
4279
4280 /// The geometric-reliability gate, on the two shapes it has to tell apart.
4281 #[test]
4282 fn geometric_reliability_rejects_split_column_grids() {
4283 let g = |rows: &[&[&str]]| -> Vec<Vec<String>> {
4284 rows.iter()
4285 .map(|r| r.iter().map(|c| c.to_string()).collect())
4286 .collect()
4287 };
4288 // A genuine grid: dense, every column carrying entries. Nothing for
4289 // TableFormer to improve, so geometry is used as-is.
4290 assert!(super::geometric_table_is_reliable(&g(&[
4291 &["Datum", "Leistung", "Anzahl", "Kosten"],
4292 &["04.07", "Internet", "1", "40.30"],
4293 &["04.07", "Telefon", "2", "8.06"],
4294 ])));
4295 // The left-edge split artefact (the shape a scanned invoice produced):
4296 // one real label column plus values scattered across three sparse ones.
4297 assert!(!super::geometric_table_is_reliable(&g(&[
4298 &["www.magenta.at/faq", "", "", ""],
4299 &["Serviceteam", "", "", ""],
4300 &["Telefon", "0676/2000", "", ""],
4301 &["Kundennummer", "", "", "1.21699482"],
4302 &["Rechnungsnummer", "", "922769430725", ""],
4303 &["Rechnungsdatum", "", "", "04.07.2025"],
4304 ])));
4305 // A column only one row ever uses is a split artefact even when the
4306 // grid is otherwise dense.
4307 assert!(!super::geometric_table_is_reliable(&g(&[
4308 &["a", "b", ""],
4309 &["c", "d", ""],
4310 &["e", "f", "g"],
4311 ])));
4312 // Degenerate shapes are never vouched for — TableFormer may recover
4313 // structure a collapsed reconstruction lost.
4314 assert!(!super::geometric_table_is_reliable(&g(&[&[
4315 "only one column"
4316 ]])));
4317 assert!(!super::geometric_table_is_reliable(&[]));
4318 }
4319
4320 /// A `picture` region is cropped out of the rendered page, whatever built
4321 /// that page. The browser pipeline (#157) has no pdfium but does hand over
4322 /// the rasterized bitmap through `from_cells_with_image`, so it must get
4323 /// the same figure bytes the native path does — that is what makes
4324 /// `images = "embedded"` inline real pixels instead of a placeholder.
4325 #[cfg(feature = "ocr-prep")]
4326 #[test]
4327 fn picture_regions_are_cropped_from_a_host_supplied_page_image() {
4328 let mut img = image::RgbImage::new(200, 200);
4329 // Paint the figure area so the crop is distinguishable from the page.
4330 for y in 100..160 {
4331 for x in 20..120 {
4332 img.put_pixel(x, y, image::Rgb([255, 0, 0]));
4333 }
4334 }
4335 // scale 2.0: the region is in page points, the bitmap in pixels.
4336 let page = PdfPage::from_cells_with_image(100.0, 100.0, 2.0, Vec::new(), img);
4337 let region = Region {
4338 label: "picture",
4339 score: 0.9,
4340 l: 10.0,
4341 t: 50.0,
4342 r: 60.0,
4343 b: 80.0,
4344 };
4345 let (nodes, _) = super::assemble_page(&page, vec![region], &[None], &[None], None);
4346 // Layout-derived nodes carry provenance, so the picture arrives wrapped.
4347 let image = nodes
4348 .iter()
4349 .find_map(|n| match n {
4350 Node::Located { inner, .. } => match &**inner {
4351 Node::Picture { image, .. } => image.as_ref(),
4352 _ => None,
4353 },
4354 Node::Picture { image, .. } => image.as_ref(),
4355 _ => None,
4356 })
4357 .expect("a picture node with cropped pixels");
4358 assert_eq!(image.mimetype, "image/png");
4359 assert_eq!((image.width, image.height), (100, 60), "region × scale");
4360 assert!(!image.data.is_empty(), "PNG bytes were encoded");
4361 }
4362
4363 #[test]
4364 fn link_anchors_split_a_shared_word_cell_between_adjacent_links() {
4365 // A common header layout: one text run holds several pipe-separated
4366 // labels, each carrying its own link annotation. Every link must get
4367 // its own label as the anchor (and the "|" separators must belong to
4368 // none), not the whole run.
4369 let annot = |l: f32, r: f32, uri: &str| LinkAnnot {
4370 l,
4371 t: 100.0,
4372 r,
4373 b: 114.0,
4374 uri: uri.into(),
4375 };
4376 let page = PdfPage {
4377 width: 600.0,
4378 height: 800.0,
4379 scale: 2.0,
4380 cells: Vec::new(),
4381 code_cells: Vec::new(),
4382 checkboxes: Vec::new(),
4383 // "LinkedIn | GitHub | Credly" = 26 chars over x 100..360.
4384 word_cells: vec![cell(
4385 "LinkedIn | GitHub | Credly",
4386 100.0,
4387 100.0,
4388 360.0,
4389 114.0,
4390 )],
4391 image: image::RgbImage::new(1, 1),
4392 image_layout: None,
4393 links: vec![
4394 annot(100.0, 180.0, "https://l"),
4395 annot(200.0, 260.0, "https://g"),
4396 annot(290.0, 360.0, "https://c"),
4397 ],
4398 rotation: 0,
4399 };
4400 assert_eq!(
4401 resolve_link_anchors(&page),
4402 vec![
4403 ("LinkedIn".to_string(), "https://l".to_string()),
4404 ("GitHub".to_string(), "https://g".to_string()),
4405 ("Credly".to_string(), "https://c".to_string()),
4406 ]
4407 );
4408 }
4409
4410 /// A one-line code cell at `[l, r] × [t, b]` (top-left coords).
4411 fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
4412 TextCell {
4413 text: text.into(),
4414 l,
4415 t,
4416 r,
4417 b,
4418 }
4419 }
4420
4421 /// OCR-path grouping (docling's `_remove_overlapping_clusters("regular")`):
4422 /// the low-score paragraph box RT-DETR draws over its own high-score line
4423 /// boxes collapses to one region — the group's union, with the survivor's
4424 /// label and score — so region-scoped OCR reads each line once. Regions
4425 /// that merely sit near each other, and specials, are untouched.
4426 #[test]
4427 fn merge_overlapping_regulars_collapses_a_block_over_its_lines() {
4428 let mut regions = vec![
4429 region("text", 0.84, 60.0, 186.0, 270.0, 198.0),
4430 region("text", 0.80, 60.0, 160.0, 294.0, 172.0),
4431 region("text", 0.79, 59.0, 107.0, 272.0, 119.0),
4432 // The paragraph box, lower score, containing all three lines.
4433 region("text", 0.52, 59.0, 107.0, 295.0, 200.0),
4434 // Elsewhere on the page: stays as is.
4435 region("section_header", 0.77, 60.0, 71.0, 253.0, 86.0),
4436 // A picture the block overlaps is not a regular — never grouped.
4437 region("picture", 0.9, 50.0, 100.0, 300.0, 210.0),
4438 ];
4439 merge_overlapping_regulars(&mut regions);
4440 assert_eq!(regions.len(), 3, "{regions:?}");
4441 let block = regions
4442 .iter()
4443 .find(|r| r.label == "text")
4444 .expect("one text");
4445 // docling keeps the largest passing candidate unless a rival is both
4446 // comparable in size and > 0.05 more confident; the 16× larger block
4447 // passes, and a smaller line never replaces a larger current best.
4448 // Either way the survivor spans the whole group.
4449 assert_eq!(
4450 (block.l, block.t, block.r, block.b),
4451 (59.0, 107.0, 295.0, 200.0)
4452 );
4453 assert!(regions.iter().any(|r| r.label == "section_header"));
4454 assert!(regions.iter().any(|r| r.label == "picture"));
4455 }
4456
4457 /// The pairwise rules, each in the arrangement where it decides the
4458 /// outcome: docling seeds the survivor with the group's first passing
4459 /// cluster and a later one replaces it only when larger *and* within
4460 /// 0.05 confidence, so a rule that merely lets a cluster pass matters
4461 /// exactly when that cluster comes first — a same-sized list item ahead
4462 /// of a far more confident text box, a code box ahead of the text it
4463 /// contains. Without the rule either would be rejected outright (similar
4464 /// size, rival > 0.05 more confident) and the text box would win.
4465 #[test]
4466 fn merge_overlapping_regulars_follows_the_preference_rules() {
4467 let mut regions = vec![
4468 region("list_item", 0.6, 0.0, 0.0, 102.0, 20.0),
4469 region("text", 0.9, 0.0, 0.0, 100.0, 20.0),
4470 ];
4471 merge_overlapping_regulars(&mut regions);
4472 assert_eq!(regions.len(), 1);
4473 assert_eq!(regions[0].label, "list_item");
4474
4475 let mut regions = vec![
4476 region("code", 0.6, 0.0, 0.0, 100.0, 100.0),
4477 region("text", 0.9, 2.0, 2.0, 98.0, 98.0),
4478 ];
4479 merge_overlapping_regulars(&mut regions);
4480 assert_eq!(regions.len(), 1);
4481 assert_eq!(regions[0].label, "code");
4482
4483 // No rule applies: a near-identical rival that is > 0.05 more
4484 // confident rejects the candidate whatever the order.
4485 for order in [[0.9, 0.6], [0.6, 0.9]] {
4486 let mut regions = vec![
4487 region("text", order[0], 0.0, 0.0, 100.0, 20.0),
4488 region("text", order[1], 0.0, 0.0, 105.0, 21.0),
4489 ];
4490 merge_overlapping_regulars(&mut regions);
4491 assert_eq!(regions.len(), 1);
4492 assert_eq!(regions[0].score, 0.9, "the confident twin wins");
4493 assert_eq!(
4494 (regions[0].r, regions[0].b),
4495 (105.0, 21.0),
4496 "on the union box"
4497 );
4498 }
4499
4500 // Side by side (no containment, IoU 0): nothing to merge.
4501 let mut regions = vec![
4502 region("text", 0.9, 0.0, 0.0, 100.0, 20.0),
4503 region("text", 0.9, 0.0, 22.0, 100.0, 42.0),
4504 ];
4505 merge_overlapping_regulars(&mut regions);
4506 assert_eq!(regions.len(), 2);
4507 }
4508
4509 #[test]
4510 fn footer_under_a_body_less_heading_is_its_text() {
4511 // The reporting CV's last page: `## Languages` at t=773.7..783.4 and
4512 // its one-line body at 798.5..808.7, labelled page_footer (A4, 595 pt).
4513 let mut regions = vec![
4514 region("list_item", 0.9, 66.0, 754.0, 353.0, 765.0),
4515 region("section_header", 0.94, 43.0, 773.7, 89.5, 783.4),
4516 region("page_footer", 0.88, 43.0, 798.5, 527.8, 808.7),
4517 ];
4518 reclaim_heading_body_footers(&mut regions, 595.28);
4519 assert_eq!(regions[2].label, "text");
4520 assert_eq!(regions[1].label, "section_header");
4521
4522 // A heading with its own paragraph and a running footer below: kept.
4523 let mut regions = vec![
4524 region("section_header", 0.9, 43.0, 700.0, 120.0, 710.0),
4525 region("text", 0.9, 43.0, 714.0, 520.0, 780.0),
4526 region("page_footer", 0.9, 43.0, 798.0, 520.0, 808.0),
4527 ];
4528 reclaim_heading_body_footers(&mut regions, 595.28);
4529 assert_eq!(regions[2].label, "page_footer");
4530
4531 // A page number under a trailing heading is too narrow to be a body.
4532 let mut regions = vec![
4533 region("section_header", 0.9, 43.0, 773.0, 120.0, 783.0),
4534 region("page_footer", 0.9, 280.0, 798.0, 300.0, 808.0),
4535 ];
4536 reclaim_heading_body_footers(&mut regions, 595.28);
4537 assert_eq!(regions[1].label, "page_footer");
4538
4539 // Too far below the heading (a real footer after a heading that ends
4540 // the page): kept.
4541 let mut regions = vec![
4542 region("section_header", 0.9, 43.0, 700.0, 120.0, 710.0),
4543 region("page_footer", 0.9, 43.0, 798.0, 520.0, 808.0),
4544 ];
4545 reclaim_heading_body_footers(&mut regions, 595.28);
4546 assert_eq!(regions[1].label, "page_footer");
4547 }
4548
4549 fn region(label: &'static str, score: f32, l: f32, t: f32, r: f32, b: f32) -> Region {
4550 Region {
4551 label,
4552 score,
4553 l,
4554 t,
4555 r,
4556 b,
4557 }
4558 }
4559
4560 #[test]
4561 fn resolve_collapses_nested_code_keeping_the_larger_box() {
4562 // A tight high-score `code` box and a taller lower-score near-duplicate that
4563 // contains it must collapse to one — the *larger* box, so every cell stays
4564 // covered and nothing leaks out as orphan text.
4565 let tight = region("code", 0.95, 78.0, 292.0, 300.0, 330.0);
4566 let wide = region("code", 0.66, 63.0, 260.0, 320.0, 346.0);
4567 let kept = super::resolve(vec![tight, wide]);
4568 assert_eq!(kept.len(), 1, "nested code boxes must collapse to one");
4569 assert!(
4570 kept[0].l == 63.0 && kept[0].b == 346.0,
4571 "the larger containing box is kept"
4572 );
4573 }
4574
4575 #[test]
4576 fn resolve_keeps_distinct_and_differently_typed_regions() {
4577 // A text box fully inside a lower-score *table* must NOT be collapsed (the
4578 // code dedup is code-only), and two separate code blocks stay separate.
4579 let text = region("text", 0.95, 90.0, 210.0, 200.0, 230.0);
4580 let table = region("table", 0.60, 80.0, 200.0, 400.0, 500.0);
4581 assert_eq!(super::resolve(vec![text, table]).len(), 2);
4582
4583 let code_a = region("code", 0.9, 78.0, 100.0, 300.0, 140.0);
4584 let code_b = region("code", 0.9, 78.0, 300.0, 300.0, 360.0); // far below, no overlap
4585 assert_eq!(super::resolve(vec![code_a, code_b]).len(), 2);
4586 }
4587
4588 /// A two-column glossary page came out as three column
4589 /// tables *and* one low-score whole-page table over them. docling's wrapper
4590 /// `_remove_overlapping_clusters` keeps one table per overlapping group
4591 /// (here the whole-page one: > 2× every rival's area and ≤ 0.2 less
4592 /// confident than the running best); `greedy` alone kept all four and
4593 /// emitted every cell twice.
4594 #[test]
4595 fn resolve_keeps_one_table_per_nested_group() {
4596 let kept = super::resolve(vec![
4597 region("table", 0.71, 26.0, 203.0, 183.0, 558.0),
4598 region("table", 0.67, 26.0, 55.0, 183.0, 196.0),
4599 region("table", 0.66, 196.0, 56.0, 354.0, 561.0),
4600 region("table", 0.53, 25.0, 53.0, 354.0, 561.0),
4601 ]);
4602 assert_eq!(kept.len(), 1, "one survivor per overlapping group");
4603 assert_eq!((kept[0].l, kept[0].b), (25.0, 561.0));
4604 // Side-by-side tables that don't overlap stay separate.
4605 let kept = super::resolve(vec![
4606 region("table", 0.9, 26.0, 55.0, 183.0, 558.0),
4607 region("table", 0.9, 196.0, 56.0, 354.0, 561.0),
4608 ]);
4609 assert_eq!(kept.len(), 2);
4610 }
4611
4612 /// A dense data table detected as both picture (0.80) and table (0.62) on one box.
4613 /// The picture is ≥ 0.1 more confident, so `_handle_cross_type_overlaps`
4614 /// keeps both, and the dense table text passed the text-panel gates: the
4615 /// demoted paragraph repeated every cell the table grid renders. A
4616 /// paragraph > 80 % inside a surviving table is the table's child and is
4617 /// not emitted; a panel with no table under it still demotes.
4618 #[test]
4619 fn text_panel_over_a_table_does_not_repeat_its_cells() {
4620 let lines = |t0: f32| -> Vec<TextCell> {
4621 (0..4)
4622 .map(|i| {
4623 let t = t0 + 10.0 * i as f32;
4624 cell("1 2 3 4 5 6 7 8 9 10 11 12", 5.0, t, 95.0, t + 8.0)
4625 })
4626 .collect()
4627 };
4628 let mut cells = lines(0.0);
4629 cells.extend(lines(200.0));
4630 let mut regions = vec![
4631 region("picture", 0.80, 0.0, 0.0, 100.0, 45.0),
4632 region("table", 0.62, 0.0, 0.0, 100.0, 45.0),
4633 region("picture", 0.80, 0.0, 200.0, 100.0, 245.0),
4634 ];
4635 super::recover_text_panels(&mut regions, &cells);
4636 assert_eq!(
4637 regions.iter().map(|r| r.label).collect::<Vec<_>>(),
4638 ["table", "text"]
4639 );
4640 assert_eq!(regions[1].t, 200.0, "the table-free panel still demotes");
4641 }
4642
4643 /// docling's `ListItemMarkerProcessor` + docling-core's Markdown list rules
4644 /// (see `list_item_node`): a bullet glyph or `-` is split off and the
4645 /// serializer's own `-` printed (2305's OTSL list reads `- "C" cell`, not
4646 /// `- - "C" cell`); `N.` is an ordered item; any other marker with a letter
4647 /// or digit stays in the text behind the bullet; no marker → plain bullet.
4648 #[test]
4649 fn list_item_markers_split_like_docling() {
4650 let text_of = |n: &Node| match n {
4651 Node::ListItem {
4652 ordered,
4653 number,
4654 text,
4655 marker,
4656 ..
4657 } => (*ordered, *number, text.clone(), marker.clone()),
4658 other => panic!("{other:?}"),
4659 };
4660 let loc = [0, 0, 100, 10];
4661 assert_eq!(
4662 text_of(&super::list_item_node(
4663 "- \"C\" cell - a new table cell",
4664 loc,
4665 false
4666 )),
4667 (
4668 false,
4669 0,
4670 "\"C\" cell - a new table cell".into(),
4671 Some("-".into())
4672 )
4673 );
4674 assert_eq!(
4675 text_of(&super::list_item_node("• Bullet text", loc, false)),
4676 (false, 0, "Bullet text".into(), Some("•".into()))
4677 );
4678 assert_eq!(
4679 text_of(&super::list_item_node("3. Third step", loc, false)),
4680 (true, 3, "Third step".into(), Some("3.".into()))
4681 );
4682 assert_eq!(
4683 text_of(&super::list_item_node("a) Option", loc, false)),
4684 (false, 0, "a) Option".into(), Some("a)".into()))
4685 );
4686 assert_eq!(
4687 text_of(&super::list_item_node("1.2 Nested outline", loc, false)),
4688 (false, 0, "1.2 Nested outline".into(), Some("1.2".into()))
4689 );
4690 // 2203's `3.a. If all IOU scores…`: a compound marker is not an item
4691 // number — docling prints `- 3.a. If all…`.
4692 assert_eq!(
4693 text_of(&super::list_item_node("3.a. If all IOU scores", loc, false)),
4694 (
4695 false,
4696 0,
4697 "3.a. If all IOU scores".into(),
4698 Some("3.a.".into())
4699 )
4700 );
4701 // A glued symbol-font bullet is stripped, a spaced one is the marker.
4702 assert_eq!(
4703 text_of(&super::list_item_node("•Glued", loc, false)),
4704 (false, 0, "Glued".into(), Some("·".into()))
4705 );
4706 // No whitespace after the glyph → not a marker (docling's `\s` is required).
4707 assert_eq!(
4708 text_of(&super::list_item_node("-5 degrees", loc, false)),
4709 (false, 0, "-5 degrees".into(), Some("·".into()))
4710 );
4711 // The remaining numbered shapes, first-wins like docling's list.
4712 for (input, marker, body) in [
4713 ("1.2.3. Deep", "1.2.3.", "Deep"),
4714 ("9a) Nine-a", "9a)", "Nine-a"),
4715 ("(3.a) Paren", "(3.a)", "Paren"),
4716 ("12) Twelve", "12)", "Twelve"),
4717 ("(4) Four", "(4)", "Four"),
4718 ("[7] Seven", "[7]", "Seven"),
4719 ("iv. Roman", "iv.", "Roman"),
4720 ("IX. Roman", "IX.", "Roman"),
4721 ("b. Letter", "b.", "Letter"),
4722 ("B) Letter", "B)", "Letter"),
4723 ] {
4724 assert_eq!(
4725 super::split_list_marker(input),
4726 Some((marker, body, true)),
4727 "{input}"
4728 );
4729 }
4730 // A `1.2.` whose optional dot would eat the separator backtracks like
4731 // Python's regex; a marker with nothing after the whitespace is none.
4732 assert_eq!(
4733 super::split_list_marker("1.2.\tx"),
4734 Some(("1.2.", "x", true))
4735 );
4736 assert_eq!(super::split_list_marker("1. "), None);
4737 assert_eq!(super::split_list_marker("• "), None);
4738 assert_eq!(
4739 text_of(&super::list_item_node("Plain item", loc, false)),
4740 (false, 0, "Plain item".into(), Some("·".into()))
4741 );
4742 }
4743
4744 #[test]
4745 fn code_language_label_above_code_is_detected() {
4746 // A bare "XML" token directly above a code box is a language label; a real
4747 // heading above the same code is not; a language word with no code below is
4748 // left alone.
4749 let label = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
4750 let code = region("code", 0.7, 77.0, 552.0, 290.0, 640.0);
4751 let heading = region("section_header", 0.9, 76.0, 500.0, 260.0, 512.0);
4752 let cells = vec![
4753 cell("XML", 78.0, 541.0, 94.0, 548.0), // inside `label`
4754 cell("Overview", 78.0, 501.0, 250.0, 511.0), // inside `heading`
4755 ];
4756 let drop = super::code_language_labels(&[label, code, heading], &cells);
4757 assert_eq!(drop, vec![true, false, false], "only the label is consumed");
4758
4759 // Same label with no code region present → not consumed.
4760 let label2 = region("section_header", 0.9, 76.0, 540.0, 96.0, 549.0);
4761 let only = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
4762 assert_eq!(super::code_language_labels(&[label2], &only), vec![false]);
4763
4764 // A label swallowed into the top of a wider code box (negative gap) is still
4765 // recognized.
4766 let inside_lbl = region("text", 0.9, 76.0, 540.0, 96.0, 549.0);
4767 let wide_code = region("code", 0.7, 63.0, 531.0, 320.0, 654.0);
4768 let cells2 = vec![cell("XML", 78.0, 541.0, 94.0, 548.0)];
4769 assert_eq!(
4770 super::code_language_labels(&[inside_lbl, wide_code], &cells2),
4771 vec![true, false]
4772 );
4773
4774 assert!(super::is_code_language("XML") && super::is_code_language("c#"));
4775 assert!(!super::is_code_language("Configure") && !super::is_code_language("XML schema"));
4776 }
4777
4778 #[test]
4779 fn code_region_text_keeps_lines_and_indentation() {
4780 // Three source lines; each glyph is 6 units wide (width / chars = 6), so the
4781 // `int X;` line indented to x=22 is (22-10)/6 = 2 spaces in.
4782 let region = Region {
4783 label: "code",
4784 score: 1.0,
4785 l: 0.0,
4786 t: -5.0,
4787 r: 100.0,
4788 b: 40.0,
4789 };
4790 let cells = vec![
4791 cell("struct P {", 10.0, 0.0, 70.0, 10.0),
4792 cell("int X;", 22.0, 12.0, 58.0, 22.0),
4793 cell("}", 10.0, 24.0, 16.0, 34.0),
4794 ];
4795 assert_eq!(code_region_text(®ion, &cells), "struct P {\n int X;\n}");
4796 }
4797
4798 #[test]
4799 fn code_region_text_tightens_punctuation_without_eating_indentation() {
4800 // A fluent `.Foo()` line at x=22 (2 chars in). Per-line tightening must not
4801 // consume the leading indent space by matching " ." across it.
4802 let region = Region {
4803 label: "code",
4804 score: 1.0,
4805 l: 0.0,
4806 t: -5.0,
4807 r: 100.0,
4808 b: 40.0,
4809 };
4810 let cells = vec![
4811 cell("builder", 10.0, 0.0, 52.0, 10.0),
4812 // pdfium spaced the call: ".Foo (x)" tightens to ".Foo(x)", still 2-indented.
4813 cell(".Foo (x)", 22.0, 12.0, 70.0, 22.0),
4814 ];
4815 assert_eq!(code_region_text(®ion, &cells), "builder\n .Foo(x)");
4816 }
4817
4818 #[test]
4819 fn code_region_text_orders_out_of_order_cells_and_ignores_blank_lines() {
4820 let region = Region {
4821 label: "code",
4822 score: 1.0,
4823 l: 0.0,
4824 t: -5.0,
4825 r: 100.0,
4826 b: 60.0,
4827 };
4828 // Fed bottom-up and with a whitespace-only cell; output is top-down, no blank.
4829 let cells = vec![
4830 cell("b();", 10.0, 24.0, 34.0, 34.0),
4831 cell(" ", 10.0, 12.0, 20.0, 22.0),
4832 cell("a();", 10.0, 0.0, 34.0, 10.0),
4833 ];
4834 assert_eq!(code_region_text(®ion, &cells), "a();\nb();");
4835 // No code cells → empty, so the caller falls back to the prose text.
4836 assert_eq!(code_region_text(®ion, &[]), "");
4837 }
4838
4839 fn para(text: &str) -> Node {
4840 Node::Paragraph { text: text.into() }
4841 }
4842
4843 /// Run a node sequence through [`StreamAssembler`] with the given page splits
4844 /// and assert the flushed result equals one-shot [`merge_continuations`].
4845 fn assert_stream_eq(nodes: &[Node], splits: &[usize]) {
4846 let mut want = nodes.to_vec();
4847 merge_continuations(&mut want);
4848
4849 let mut asm = StreamAssembler::new();
4850 let mut got = Vec::new();
4851 let mut start = 0;
4852 for &end in splits {
4853 got.extend(asm.push(nodes[start..end].to_vec()));
4854 start = end;
4855 }
4856 got.extend(asm.push(nodes[start..].to_vec()));
4857 got.extend(asm.finish());
4858 assert_eq!(got, want, "stream assembly diverged (splits={splits:?})");
4859 }
4860
4861 #[test]
4862 fn stream_assembler_matches_merge_continuations() {
4863 // Open fragment + lowercase continuation split across a page boundary.
4864 let cross = [para("the definition of"), para("lists in scope")];
4865 assert_stream_eq(&cross, &[1]);
4866 assert_stream_eq(&cross, &[]);
4867
4868 // Continuation that wraps around a figure (+ its caption) on the boundary.
4869 let wrap = [
4870 para("the wing type that is"),
4871 Node::Picture {
4872 caption: None,
4873 caption_href: None,
4874 image: None,
4875 classification: None,
4876 caption_parent: Default::default(),
4877 caption_location: None,
4878 },
4879 para("Fig. 1. a diagram"),
4880 para("the most common kind"),
4881 ];
4882 for splits in [&[][..], &[1][..], &[2][..], &[3][..], &[1, 3][..]] {
4883 assert_stream_eq(&wrap, splits);
4884 }
4885
4886 // A heading between fragments blocks the merge (must still flush correctly).
4887 let blocked = [
4888 para("ends mid word and"),
4889 Node::Heading {
4890 level: 2,
4891 text: "New Section".into(),
4892 },
4893 para("more body here"),
4894 ];
4895 for splits in [&[][..], &[1][..], &[2][..]] {
4896 assert_stream_eq(&blocked, splits);
4897 }
4898
4899 // A chain across three pages: each page is one open lowercase fragment.
4900 let chain = [
4901 para("alpha beta"),
4902 para("gamma delta"),
4903 para("epsilon zeta"),
4904 ];
4905 assert_stream_eq(&chain, &[1, 2]);
4906 }
4907
4908 #[test]
4909 fn clean_text_dehyphenates_and_normalizes_typography() {
4910 // U+0002 line-wrap hyphen + the join space → merged word (like docling).
4911 assert_eq!(clean_text("com\u{2} pact"), "compact");
4912 assert_eq!(clean_text("end-to\u{2} end deep"), "end-toend deep");
4913 // A stray wrap hyphen (no following join) is dropped.
4914 assert_eq!(clean_text("word\u{2}"), "word");
4915 // Typographic punctuation → ASCII: every curly quote becomes `'`
4916 // (docling-parse's sanitizer table), a literal `"` stays.
4917 assert_eq!(
4918 clean_text("Graph\u{2019}s \u{201c}x\u{201d} \"y\""),
4919 "Graph's 'x' \"y\""
4920 );
4921 assert_eq!(clean_text("a\u{2026}"), "a...");
4922 // The docling-parse sanitizer's internal spacing is preserved as
4923 // placed; line breaks/tabs normalize to a space, ends trim.
4924 assert_eq!(clean_text("a b\nc"), "a b c");
4925 }
4926
4927 /// docling#4064: a form's children are emitted together where the form
4928 /// sits in the top-level order, not interleaved with surrounding text.
4929 #[test]
4930 fn form_children_stay_together_in_reading_order() {
4931 let reg = |label: &'static str, l: f32, t: f32, r: f32, b: f32| Region {
4932 label,
4933 score: 0.9,
4934 l,
4935 t,
4936 r,
4937 b,
4938 };
4939 // Page: intro text, then a form spanning the left column with two
4940 // fields and a table inside, while a right-column paragraph sits
4941 // level with the form's first field (it would otherwise be read
4942 // between the form's children).
4943 let mut items = vec![
4944 reg("text", 50.0, 50.0, 550.0, 70.0), // 0 intro
4945 reg("form", 50.0, 100.0, 300.0, 400.0), // 1 container
4946 reg("text", 60.0, 110.0, 290.0, 130.0), // 2 field A (child)
4947 reg("text", 320.0, 110.0, 550.0, 130.0), // 3 right column paragraph
4948 reg("table", 60.0, 150.0, 290.0, 300.0), // 4 table (child)
4949 reg("text", 60.0, 320.0, 290.0, 340.0), // 5 field B (child)
4950 reg("text", 50.0, 450.0, 550.0, 470.0), // 6 outro
4951 ];
4952 let cids = super::cluster_cids(&items, &[]);
4953 super::order_with_containers(&mut items, &cids, 600.0, 800.0, |r| r);
4954 let order: Vec<(&str, f32)> = items.iter().map(|r| (r.label, r.t)).collect();
4955 // The form block (container, then its children top-down) is one unit.
4956 let form_pos = order.iter().position(|(l, _)| *l == "form").unwrap();
4957 assert_eq!(
4958 &order[form_pos..form_pos + 4],
4959 &[
4960 ("form", 100.0),
4961 ("text", 110.0),
4962 ("table", 150.0),
4963 ("text", 320.0)
4964 ]
4965 );
4966 assert_eq!(order[0], ("text", 50.0));
4967 assert_eq!(order[order.len() - 1], ("text", 450.0));
4968 // Without a container the plain order interleaves by geometry.
4969 let mut flat: Vec<Region> = items
4970 .iter()
4971 .filter(|r| r.label != "form")
4972 .cloned()
4973 .collect();
4974 let cids = super::cluster_cids(&flat, &[]);
4975 super::order_regions(&mut flat, &cids, 600.0, 800.0, |r| r);
4976 assert_ne!(
4977 flat.iter().map(|r| r.t).collect::<Vec<_>>(),
4978 order
4979 .iter()
4980 .filter(|(l, _)| *l != "form")
4981 .map(|(_, t)| *t)
4982 .collect::<Vec<_>>()
4983 );
4984 }
4985
4986 /// docling#3906: a picture inside a table lands in the covering cell,
4987 /// chosen by the picture's inferred grid position when cell boxes overlap.
4988 #[test]
4989 fn picture_matches_the_cell_at_its_grid_position() {
4990 let cell = |r: usize, c: usize, bbox: [f32; 4]| docling_core::TableCell {
4991 text: format!("r{r}c{c}"),
4992 bbox: Some(bbox),
4993 start_row: r,
4994 start_col: c,
4995 row_span: 1,
4996 col_span: 1,
4997 column_header: false,
4998 row_header: false,
4999 row_section: false,
5000 };
5001 // 2×2 grid; the (1,0) cell box is generous and also covers the picture.
5002 let cells = vec![
5003 cell(0, 0, [0.0, 0.0, 100.0, 50.0]),
5004 cell(0, 1, [100.0, 0.0, 200.0, 50.0]),
5005 cell(1, 0, [0.0, 50.0, 100.0, 100.0]),
5006 cell(1, 1, [100.0, 50.0, 200.0, 100.0]),
5007 ];
5008 let pic = Region {
5009 label: "picture",
5010 score: 0.9,
5011 l: 110.0,
5012 t: 60.0,
5013 r: 190.0,
5014 b: 95.0,
5015 };
5016 assert_eq!(super::match_picture_to_cell(&pic, &cells), Some((1.0, 3)));
5017 // A picture only half inside any cell is not nested.
5018 let straddling = Region {
5019 label: "picture",
5020 score: 0.9,
5021 l: 60.0,
5022 t: 60.0,
5023 r: 160.0,
5024 b: 95.0,
5025 };
5026 assert_eq!(super::match_picture_to_cell(&straddling, &cells), None);
5027 }
5028
5029 /// docling#4052 (2.122): a line-final dash fuses the wrapped word only
5030 /// when attached to it; a detached dash is a literal and the lines join
5031 /// with a space.
5032 #[test]
5033 fn line_final_hyphen_fuses_only_when_attached_to_a_word() {
5034 let line = |text: &str, t: f32| TextCell {
5035 text: text.to_string(),
5036 l: 0.0,
5037 t,
5038 r: 100.0,
5039 b: t + 10.0,
5040 };
5041 // `algo-` / `rithms`: attached hyphen, alnum on both sides → fused.
5042 assert_eq!(
5043 cells_text(vec![&line("algo-", 0.0), &line("rithms", 12.0)]),
5044 "algorithms"
5045 );
5046 // `pp. 545-` / `561`: attached, digits count as alnum → `545561` (upstream).
5047 assert_eq!(
5048 cells_text(vec![&line("pp. 545-", 0.0), &line("561", 12.0)]),
5049 "pp. 545561"
5050 );
5051 // A dash after whitespace — a separator or a lone `-` cell — is kept and
5052 // the lines take the ordinary joining space.
5053 assert_eq!(
5054 cells_text(vec![&line("range -", 0.0), &line("wide", 12.0)]),
5055 "range - wide"
5056 );
5057 assert_eq!(
5058 cells_text(vec![&line("-", 0.0), &line("item", 12.0)]),
5059 "- item"
5060 );
5061 // Attached but the next line opens with no word (`x-` / `...`): dash
5062 // kept and, as before, no separating space.
5063 assert_eq!(
5064 cells_text(vec![&line("x-", 0.0), &line("...", 12.0)]),
5065 "x-..."
5066 );
5067 }
5068
5069 #[test]
5070 fn lam_alef_only_swaps_a_genuinely_reversed_ligature() {
5071 // A mid-word `alef-variant + lam` is pdfium's reversed lam-alef ligature and
5072 // is swapped back to logical `lam + alef-variant` (`ب أ ل` → `ب ل أ`).
5073 assert_eq!(
5074 clean_text("\u{0628}\u{0623}\u{0644}"),
5075 "\u{0628}\u{0644}\u{0623}"
5076 );
5077 // But when the alef-variant is *already* preceded by a lam it is the logical
5078 // ligature `لآ`; the following lam is the next syllable's letter and must not
5079 // move. `التعلم الآلي` must stay `الآلي`, not become `اللآي`.
5080 assert_eq!(
5081 clean_text("\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"),
5082 "\u{0627}\u{0644}\u{0622}\u{0644}\u{064a}"
5083 );
5084 }
5085
5086 /// The #419 page, in points: three layout boxes over one paragraph, two of
5087 /// them ending partway through a line. The sliced lines miss the 0.2 claim
5088 /// and become orphans; the third model box starts above the second orphan,
5089 /// so unfitted the reading order emits that box first and strands the line.
5090 fn sliced_paragraph() -> (Vec<Region>, Vec<TextCell>) {
5091 let line = |text: &str, t: f32, r: f32| cell(text, 60.0, t, r, t + 11.0);
5092 let cells = vec![
5093 line("The mission of this series is to improve", 135.0, 458.0),
5094 line("The books in this series are technical,", 147.0, 458.0),
5095 line("substantial. The authors are", 159.0, 458.0),
5096 line("highly experienced craftsmen and", 171.5, 458.0), // sliced: 1.5/11 under box A
5097 line("actually works in practice, as opposed", 185.0, 458.0),
5098 line("about what the author has done, not", 197.0, 458.0),
5099 line("about programming, there will be lots", 210.5, 458.0), // sliced: 1.5/11 under box B
5100 line("will be lots of case studies from real", 223.0, 206.0), // C's line
5101 ];
5102 let regions = vec![
5103 region("text", 0.9, 60.0, 132.0, 458.0, 173.0), // A: three lines + a sliver of the 4th
5104 region("text", 0.9, 60.0, 184.0, 458.0, 212.0), // B: two lines + a sliver of the 7th
5105 region("text", 0.9, 60.0, 216.0, 206.0, 227.0), // C: last line, box opening 5.5pt too early
5106 ];
5107 (regions, cells)
5108 }
5109
5110 fn ordered_texts(regions: &[Region], cells: &[TextCell]) -> Vec<String> {
5111 let mut items: Vec<Region> = regions.to_vec();
5112 let cids = super::cluster_cids(&items, cells);
5113 super::order_regions(&mut items, &cids, 500.0, 700.0, |r| r);
5114 super::region_texts_exclusive(&items, cells)
5115 .into_iter()
5116 .map(|t| t.chars().take(9).collect())
5117 .collect()
5118 }
5119
5120 /// #419: fitted to its cells, a model box that cut a line in half no longer
5121 /// overlaps the orphan that line became, so the orphan orders where it
5122 /// reads; unfitted, the same page strands the line after the paragraph.
5123 #[test]
5124 fn fitting_boxes_to_cells_puts_a_sliced_line_back_in_order() {
5125 let (mut regions, cells) = sliced_paragraph();
5126 super::add_orphan_regions(&mut regions, &cells);
5127 assert_eq!(regions.len(), 5, "two orphan lines");
5128 // The defect, for the record: C (top 216) is not strictly below the
5129 // orphan at 210.5–221.5, so the graph orders C first.
5130 assert_eq!(
5131 ordered_texts(®ions, &cells).last().map(String::as_str),
5132 Some("about pro")
5133 );
5134
5135 super::fit_regions_to_cells(&mut regions, &cells);
5136 assert_eq!(regions.len(), 5);
5137 // A ends on its last claimed line, C starts on its only one.
5138 assert_eq!((regions[0].t, regions[0].b), (135.0, 170.0));
5139 assert_eq!((regions[2].t, regions[2].b), (223.0, 234.0));
5140 assert_eq!(
5141 ordered_texts(®ions, &cells),
5142 [
5143 "The missi",
5144 "highly ex",
5145 "actually ",
5146 "about pro",
5147 "will be l"
5148 ]
5149 );
5150 }
5151
5152 /// An orphan the fitted paragraph box surrounds (a short middle line the
5153 /// narrow model box missed while claiming the lines around it) is folded
5154 /// into the paragraph; an empty regular box goes away, a formula stays, a
5155 /// picture is never refitted, and a page with no cells is left untouched.
5156 #[test]
5157 fn fitting_folds_surrounded_orphans_and_drops_empty_regulars() {
5158 let wide = |text: &str, t: f32| cell(text, 60.0, t, 400.0, t + 11.0);
5159 let cells = vec![
5160 wide("first line of the paragraph", 100.0),
5161 cell("stray", 250.0, 112.0, 400.0, 123.0), // clear of the narrow box
5162 wide("third line of the paragraph", 124.0),
5163 ];
5164 let mut regions = vec![
5165 // Narrow box: claims the wide lines at 0.41, misses the short one.
5166 region("text", 0.9, 60.0, 98.0, 200.0, 136.0),
5167 region("section_header", 0.8, 60.0, 300.0, 200.0, 320.0), // no cells
5168 region("formula", 0.8, 60.0, 340.0, 200.0, 360.0), // no cells, kept
5169 region("picture", 0.8, 0.0, 400.0, 500.0, 600.0),
5170 ];
5171 super::add_orphan_regions(&mut regions, &cells);
5172 assert_eq!(regions.len(), 5, "the short line became an orphan");
5173 super::fit_regions_to_cells(&mut regions, &cells);
5174 let labels: Vec<&str> = regions.iter().map(|r| r.label).collect();
5175 assert_eq!(labels, ["text", "formula", "picture"]);
5176 let para = ®ions[0];
5177 assert_eq!(
5178 (para.l, para.t, para.r, para.b),
5179 (60.0, 100.0, 400.0, 135.0)
5180 );
5181 assert_eq!(
5182 super::region_texts_exclusive(®ions, &cells)[0],
5183 "first line of the paragraph stray third line of the paragraph"
5184 );
5185 assert_eq!(
5186 (regions[2].t, regions[2].b),
5187 (400.0, 600.0),
5188 "picture untouched"
5189 );
5190
5191 let mut untouched = vec![region("text", 0.9, 0.0, 0.0, 10.0, 10.0)];
5192 super::fit_regions_to_cells(&mut untouched, &[]);
5193 assert_eq!(untouched.len(), 1, "no cells yet: nothing dropped");
5194 }
5195
5196 /// #609: one `text` region over a checklist whose lines each have a drawn
5197 /// square in front splits into a checkbox region per boxed line — the
5198 /// wrapped second line of an option stays with it, the intro line above
5199 /// the first square keeps the region's label, and a ticked square gives
5200 /// `checkbox_selected`. Without squares nothing changes.
5201 #[test]
5202 fn checklist_lines_with_squares_split_into_checkbox_regions() {
5203 use crate::checkbox::CheckBox;
5204 let cells = vec![
5205 cell("Pick any:", 100.0, 90.0, 160.0, 100.0),
5206 cell("First option", 120.0, 110.0, 180.0, 120.0),
5207 cell("Second option that", 120.0, 129.0, 200.0, 139.0),
5208 cell("wraps onto a line", 120.0, 141.0, 196.0, 151.0),
5209 cell("Third option", 120.0, 160.0, 182.0, 170.0),
5210 ];
5211 let sq = |t: f32, checked| CheckBox {
5212 l: 100.0,
5213 t,
5214 r: 112.0,
5215 b: t + 12.0,
5216 checked,
5217 };
5218 let boxes = [sq(109.0, false), sq(128.0, true), sq(159.0, false)];
5219 let block = || vec![region("text", 0.6, 100.0, 90.0, 200.0, 170.0)];
5220
5221 let mut regions = block();
5222 super::split_checkbox_lines(&mut regions, &cells, &boxes);
5223 let texts = super::region_texts_exclusive(®ions, &cells);
5224 let got: Vec<(&str, &str)> = regions
5225 .iter()
5226 .zip(&texts)
5227 .map(|(r, t)| (r.label, t.as_str()))
5228 .collect();
5229 assert_eq!(
5230 got,
5231 [
5232 ("text", "Pick any:"),
5233 ("checkbox_unselected", "First option"),
5234 ("checkbox_selected", "Second option that wraps onto a line"),
5235 ("checkbox_unselected", "Third option"),
5236 ]
5237 );
5238 // A checkbox region spans its square and its line(s).
5239 assert_eq!((regions[2].l, regions[2].t), (100.0, 128.0));
5240 assert_eq!((regions[2].r, regions[2].b), (200.0, 151.0));
5241
5242 let shape = |rs: &[Region]| -> Vec<(&'static str, [f32; 4])> {
5243 rs.iter().map(|r| (r.label, [r.l, r.t, r.r, r.b])).collect()
5244 };
5245 let mut untouched = block();
5246 super::split_checkbox_lines(&mut untouched, &cells, &[]);
5247 assert_eq!(shape(&untouched), shape(&block()));
5248 // A square far left of the text (a margin mark), or one the text
5249 // sits inside (a comb field), does not make a checkbox.
5250 for far in [
5251 CheckBox {
5252 l: 40.0,
5253 r: 52.0,
5254 ..sq(109.0, false)
5255 },
5256 CheckBox {
5257 l: 115.0,
5258 r: 127.0,
5259 ..sq(109.0, false)
5260 },
5261 ] {
5262 let mut regions = block();
5263 super::split_checkbox_lines(&mut regions, &cells, &[far]);
5264 assert_eq!(shape(®ions), shape(&block()));
5265 }
5266 }
5267
5268 /// #609: a ballot-box glyph opening a line marks a checkbox item like a
5269 /// drawn square — `☐` unchecked, `☒` checked, the glyph stripped from the
5270 /// label at emission — while a line holding two boxes (`☐ Yes ☐ No`)
5271 /// stays text of its own.
5272 #[test]
5273 fn checklist_lines_opening_with_a_ballot_box_split_into_checkbox_regions() {
5274 let cells = vec![
5275 cell("Options:", 100.0, 90.0, 150.0, 100.0),
5276 cell("\u{2610} Tea", 100.0, 110.0, 140.0, 120.0),
5277 cell("\u{2612} Coffee", 100.0, 130.0, 150.0, 140.0),
5278 cell("\u{2610} Yes \u{2610} No", 100.0, 150.0, 170.0, 160.0),
5279 ];
5280 let mut regions = vec![region("text", 0.6, 100.0, 90.0, 170.0, 160.0)];
5281 super::split_checkbox_lines(&mut regions, &cells, &[]);
5282 let texts = super::region_texts_exclusive(®ions, &cells);
5283 let got: Vec<(&str, &str)> = regions
5284 .iter()
5285 .zip(&texts)
5286 .map(|(r, t)| match r.label {
5287 "text" => (r.label, t.as_str()),
5288 _ => (r.label, super::strip_checkbox_glyph(t)),
5289 })
5290 .collect();
5291 assert_eq!(
5292 got,
5293 [
5294 ("text", "Options:"),
5295 ("checkbox_unselected", "Tea"),
5296 ("checkbox_selected", "Coffee"),
5297 // Two boxes on one line: its own text, the item closed.
5298 ("text", "\u{2610} Yes \u{2610} No"),
5299 ]
5300 );
5301 assert_eq!(super::strip_checkbox_glyph(" \u{2611} Done"), "Done");
5302 assert_eq!(super::strip_checkbox_glyph("No box"), "No box");
5303 }
5304}