xberg 1.0.6

High-performance document intelligence library for Rust. Extract text, metadata, and structured data from PDFs, Office documents, images, and 98+ formats and 306 programming languages via tree-sitter code intelligence with async/sync APIs.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
//! Utilities for splitting and analyzing PDF paragraphs.

use super::types::PdfParagraph;

/// Maximum baseline-to-baseline gap, as a multiple of the larger paragraph's
/// dominant font size, permitted when merging a continuation paragraph.
///
/// A wrapped continuation line sits roughly one line-height (leading is
/// typically 1.0–1.6× the font size) below its predecessor. A gap several times
/// larger means the two paragraphs come from spatially distinct regions — e.g. a
/// recipient block near the top of an invoice and a legal footer near the
/// bottom — and must never be joined into one logical paragraph, which would
/// associate their text (and mangle the merged block's bounding box). See #1350.
const MAX_CONTINUATION_LINE_GAP_MULTIPLE: f32 = 3.0;

/// Merge consecutive body-text paragraphs that are continuations of the same logical paragraph.
///
/// Two consecutive paragraphs are merged if:
/// - Both are body text (no heading_level, not is_list_item)
/// - The first paragraph doesn't end with sentence-ending punctuation
/// - Font sizes are within 2pt of each other
/// - Their baselines are within [`MAX_CONTINUATION_LINE_GAP_MULTIPLE`] line-heights
pub(super) fn merge_continuation_paragraphs(paragraphs: &mut Vec<PdfParagraph>) {
    if paragraphs.len() < 2 {
        return;
    }

    let old = std::mem::take(paragraphs);
    let mut iter = old.into_iter();
    let mut current = iter.next().unwrap();

    for next in iter {
        let both_body = current.heading_level.is_none()
            && next.heading_level.is_none()
            && !current.is_list_item
            && !next.is_list_item
            && !current.is_code_block
            && !next.is_code_block
            && !current.is_formula
            && !next.is_formula;
        let fonts_compatible = (current.dominant_font_size - next.dominant_font_size).abs() < 2.0;
        // Never merge across a bold-state boundary. A bold run following non-bold
        // prose (or vice versa) is a formatting break — an emphasized heading or a
        // list item's bold lead-in — not a wrapped continuation. Absorbing it would
        // bury the heading as inline bold before it can be classified. This mirrors
        // the `bold_change` paragraph break in the heuristic line grouper. ~keep
        let bold_compatible = current.is_bold == next.is_bold;
        let continuation_signal = !ends_with_sentence_terminator(&current) || starts_with_lowercase_continuation(&next);
        let same_region = current.layout_region_path == next.layout_region_path;
        let vertical_gap_compatible = baselines_within_continuation_gap(&current, &next);
        let should_merge = both_body
            && fonts_compatible
            && bold_compatible
            && continuation_signal
            && same_region
            && vertical_gap_compatible;

        if should_merge {
            current.text.clear();
            current.block_bbox = union_block_bbox(current.block_bbox, next.block_bbox);
            current.lines.extend(next.lines);
        } else {
            paragraphs.push(current);
            current = next;
        }
    }

    paragraphs.push(current);
}

/// Whether `next`'s first baseline is close enough below `current`'s last
/// baseline to be a wrapped continuation rather than a spatially distant block.
///
/// Returns `true` when either paragraph lacks per-line geometry (`baseline_y`
/// unset on the structure-tree path), preserving the prior behavior for inputs
/// where a vertical distance cannot be computed.
fn baselines_within_continuation_gap(current: &PdfParagraph, next: &PdfParagraph) -> bool {
    let (Some(current_last), Some(next_first)) = (current.lines.last(), next.lines.first()) else {
        return true;
    };
    // No geometry available (e.g. synthesized paragraphs): fall back to allowing
    // the merge, gated by the other continuation signals.
    if current_last.baseline_y == 0.0 || next_first.baseline_y == 0.0 {
        return true;
    }
    let gap = (current_last.baseline_y - next_first.baseline_y).abs();
    let line_height = current.dominant_font_size.max(next.dominant_font_size).max(1.0);
    gap <= line_height * MAX_CONTINUATION_LINE_GAP_MULTIPLE
}

/// Union of two optional block bounding boxes in `(left, bottom, right, top)`
/// PDF-coordinate form, so a merged paragraph's box spans all of its text.
fn union_block_bbox(
    current: Option<(f32, f32, f32, f32)>,
    next: Option<(f32, f32, f32, f32)>,
) -> Option<(f32, f32, f32, f32)> {
    match (current, next) {
        (Some((cl, cb, cr, ct)), Some((nl, nb, nr, nt))) => Some((cl.min(nl), cb.min(nb), cr.max(nr), ct.max(nt))),
        (Some(bbox), None) | (None, Some(bbox)) => Some(bbox),
        (None, None) => None,
    }
}

/// Check if a paragraph starts with a lowercase letter, indicating it's a
/// continuation of a previous sentence split across paragraph boundaries.
fn starts_with_lowercase_continuation(para: &PdfParagraph) -> bool {
    let first_text = para
        .lines
        .first()
        .and_then(|l| l.segments.first())
        .map(|s| s.text.trim_start())
        .unwrap_or("");
    first_text.chars().next().is_some_and(|c| c.is_lowercase())
}

/// Check if a paragraph's last line ends with sentence-terminating punctuation.
///
/// Used by merge_continuation_paragraphs to determine if two consecutive
/// paragraphs should be merged. Supports ASCII and CJK sentence terminators.
fn ends_with_sentence_terminator(para: &PdfParagraph) -> bool {
    let last_text = para
        .lines
        .last()
        .and_then(|l| l.segments.last())
        .map(|s| s.text.trim_end())
        .unwrap_or("");
    matches!(
        last_text.chars().last(),
        Some('.' | '?' | '!' | ':' | ';' | '\u{3002}' | '\u{FF1F}' | '\u{FF01}')
    )
}

/// Split paragraphs that contain embedded bullet characters (e.g. `•`) into separate list items.
///
/// Structure tree pages sometimes merge all text into one block with inline bullets.
/// This splits "text before • item1 • item2" into separate paragraphs.
pub(super) fn split_embedded_list_items(paragraphs: &mut Vec<PdfParagraph>) {
    let old = std::mem::take(paragraphs);
    for para in old {
        if para.heading_level.is_some() || para.is_list_item || para.is_code_block || para.is_formula {
            paragraphs.push(para);
            continue;
        }

        let full_text: String = para
            .lines
            .iter()
            .flat_map(|l| l.segments.iter())
            .map(|s| s.text.as_str())
            .collect::<Vec<_>>()
            .join(" ");

        let bullet_count = full_text.matches(['\u{2022}', '\u{00B7}']).count();
        if bullet_count < 2 {
            paragraphs.push(para);
            continue;
        }

        let font_size = para.dominant_font_size;
        let is_bold = para.is_bold;

        let parts: Vec<&str> = full_text.split(['\u{2022}', '\u{00B7}']).collect();
        let before = parts[0].trim().trim_end_matches('\u{00C2}').trim();
        if !before.is_empty() {
            paragraphs.push(text_to_paragraph(before, font_size, is_bold, false));
        }
        for part in &parts[1..] {
            let item_text = part
                .trim()
                .trim_start_matches('\u{00C2}')
                .trim_end_matches('\u{00C2}')
                .trim();
            if !item_text.is_empty() {
                paragraphs.push(text_to_paragraph(item_text, font_size, is_bold, true));
            }
        }
    }
}

/// Create a simple paragraph from text.
fn text_to_paragraph(text: &str, font_size: f32, is_bold: bool, is_list_item: bool) -> PdfParagraph {
    use crate::pdf::hierarchy::SegmentData;

    let segments: Vec<SegmentData> = text
        .split_whitespace()
        .map(|w| SegmentData {
            text: w.to_string(),
            x: 0.0,
            y: 0.0,
            width: 0.0,
            height: 0.0,
            font_size,
            is_bold,
            is_italic: false,
            is_monospace: false,
            baseline_y: 0.0,
            assigned_role: None,
        })
        .collect();

    let line = super::types::PdfLine {
        segments,
        baseline_y: 0.0,
        dominant_font_size: font_size,
        is_bold,
        is_monospace: false,
    };

    let lines = vec![line];
    let word_count = PdfParagraph::compute_word_count("", &lines);
    PdfParagraph {
        text: String::new(),
        lines,
        dominant_font_size: font_size,
        heading_level: None,
        is_bold,
        is_list_item,
        is_code_block: false,
        is_formula: false,
        is_page_furniture: false,
        layout_class: None,
        layout_region_path: None,
        caption_for: None,
        block_bbox: None,
        word_count,
    }
}

#[cfg(test)]
mod tests {
    use super::*;

    fn make_body_paragraph(text: &str, font_size: f32) -> PdfParagraph {
        use crate::pdf::hierarchy::SegmentData;

        let segments = vec![SegmentData {
            text: text.to_string(),
            x: 0.0,
            y: 700.0,
            width: 200.0,
            height: font_size,
            font_size,
            is_bold: false,
            is_italic: false,
            is_monospace: false,
            baseline_y: 700.0,
            assigned_role: None,
        }];

        let lines = vec![super::super::types::PdfLine {
            segments,
            baseline_y: 700.0,
            dominant_font_size: font_size,
            is_bold: false,
            is_monospace: false,
        }];
        let word_count = PdfParagraph::compute_word_count("", &lines);
        PdfParagraph {
            text: String::new(),
            lines,
            dominant_font_size: font_size,
            heading_level: None,
            is_bold: false,
            is_list_item: false,
            is_code_block: false,
            is_formula: false,
            is_page_furniture: false,
            layout_class: None,
            layout_region_path: None,
            caption_for: None,
            block_bbox: None,
            word_count,
        }
    }

    fn make_body_paragraph_at(text: &str, font_size: f32, baseline_y: f32) -> PdfParagraph {
        let mut para = make_body_paragraph(text, font_size);
        para.lines[0].baseline_y = baseline_y;
        if let Some(segment) = para.lines[0].segments.first_mut() {
            segment.baseline_y = baseline_y;
            segment.y = baseline_y;
        }
        para
    }

    #[test]
    fn test_no_merge_across_distant_regions() {
        // A buyer tax ID in the upper recipient block and a seller tax ID in the
        // page footer are ~590pt apart. Neither ends with a sentence terminator,
        // so the older heuristic would merge them; the vertical-gap guard must
        // keep them separate. Regression for #1350.
        let mut paragraphs = vec![
            make_body_paragraph_at("Buyer tax ID SYNTH-BUYER-TAX-359370919", 8.0, 185.5),
            make_body_paragraph_at("Seller tax ID SYNTH-SELLER-TAX-815876165", 8.0, 775.0),
        ];
        merge_continuation_paragraphs(&mut paragraphs);
        assert_eq!(
            paragraphs.len(),
            2,
            "paragraphs from spatially distant regions must not merge"
        );
    }

    #[test]
    fn test_merge_adjacent_lines_within_gap() {
        // Genuine wrapped continuation one line-height apart still merges.
        let mut paragraphs = vec![
            make_body_paragraph_at("The committee reviewed the annual", 11.0, 712.0),
            make_body_paragraph_at("report and approved the budget", 11.0, 698.0),
        ];
        merge_continuation_paragraphs(&mut paragraphs);
        assert_eq!(paragraphs.len(), 1, "adjacent continuation lines should merge");
    }

    #[test]
    fn test_merge_unions_block_bbox() {
        let mut upper = make_body_paragraph_at("first line without terminator", 11.0, 712.0);
        upper.block_bbox = Some((60.0, 705.0, 260.0, 720.0));
        let mut lower = make_body_paragraph_at("second line continues", 11.0, 698.0);
        lower.block_bbox = Some((60.0, 691.0, 300.0, 706.0));
        let mut paragraphs = vec![upper, lower];
        merge_continuation_paragraphs(&mut paragraphs);
        assert_eq!(paragraphs.len(), 1, "adjacent lines should merge");
        assert_eq!(
            paragraphs[0].block_bbox,
            Some((60.0, 691.0, 300.0, 720.0)),
            "merged block bbox must span both source boxes"
        );
    }

    #[test]
    fn test_merge_lowercase_continuation() {
        let mut paragraphs = vec![
            make_body_paragraph("The regulation requires.", 12.0),
            make_body_paragraph("and all operators must comply", 12.0),
        ];
        merge_continuation_paragraphs(&mut paragraphs);
        assert_eq!(paragraphs.len(), 1, "lowercase continuation should be merged");
    }

    #[test]
    fn test_no_merge_different_font_sizes() {
        let mut paragraphs = vec![
            make_body_paragraph("First paragraph", 12.0),
            make_body_paragraph("second paragraph", 16.0),
        ];
        merge_continuation_paragraphs(&mut paragraphs);
        assert_eq!(paragraphs.len(), 2, "different font sizes should prevent merge");
    }

    #[test]
    fn test_merge_no_terminator() {
        let mut paragraphs = vec![
            make_body_paragraph("The regulation requires", 12.0),
            make_body_paragraph("All operators must comply", 12.0),
        ];
        merge_continuation_paragraphs(&mut paragraphs);
        assert_eq!(paragraphs.len(), 1, "unterminated paragraph should merge with next");
    }

    #[test]
    fn test_no_merge_terminated_uppercase() {
        let mut paragraphs = vec![
            make_body_paragraph("The regulation requires compliance.", 12.0),
            make_body_paragraph("All operators must comply", 12.0),
        ];
        merge_continuation_paragraphs(&mut paragraphs);
        assert_eq!(
            paragraphs.len(),
            2,
            "terminated paragraph + uppercase start should not merge"
        );
    }

    #[test]
    fn test_no_merge_across_bold_boundary() {
        // A bold header following unterminated body prose must not be absorbed as
        // a continuation — it should survive as its own paragraph for classification. ~keep
        let body = make_body_paragraph(
            "here is also available other sources of this Manual MetcalUser Guide",
            12.0,
        );
        let mut header = make_body_paragraph("Impaired Glucose Tolerance And Impaired Fasting Glucose ...", 12.0);
        header.is_bold = true;
        let mut paragraphs = vec![body, header];
        merge_continuation_paragraphs(&mut paragraphs);
        assert_eq!(paragraphs.len(), 2, "bold header must not merge into non-bold prose");
        assert!(paragraphs[1].is_bold, "the bold header paragraph must be preserved");
    }

    #[test]
    fn test_starts_with_lowercase_continuation_fn() {
        let para_lower = make_body_paragraph("and furthermore", 12.0);
        assert!(starts_with_lowercase_continuation(&para_lower));

        let para_upper = make_body_paragraph("Furthermore", 12.0);
        assert!(!starts_with_lowercase_continuation(&para_upper));
    }

    #[test]
    fn test_merge_clears_precomputed_text_on_heuristic_path() {
        let mut p1 = make_body_paragraph("een indicative", 12.0);
        p1.text = "een indicative".to_string();
        let mut p2 = make_body_paragraph("van toenemende merkbekendheid", 12.0);
        p2.text = "van toenemende merkbekendheid".to_string();
        let mut paragraphs = vec![p1, p2];
        merge_continuation_paragraphs(&mut paragraphs);
        assert_eq!(paragraphs.len(), 1, "lowercase continuation should merge");
        assert!(
            paragraphs[0].text.is_empty(),
            "merged paragraph must clear pre-computed text so assembly joins from segments"
        );
        assert_eq!(paragraphs[0].lines.len(), 2, "both lines must be present after merge");
    }

    #[test]
    fn test_merge_struct_tree_path_text_stays_empty() {
        let mut paragraphs = vec![
            make_body_paragraph("first sentence without terminator", 12.0),
            make_body_paragraph("second continues here", 12.0),
        ];
        assert!(paragraphs[0].text.is_empty());
        merge_continuation_paragraphs(&mut paragraphs);
        assert_eq!(paragraphs.len(), 1);
        assert!(paragraphs[0].text.is_empty());
    }
}