use super::types::PdfParagraph;
const MAX_CONTINUATION_LINE_GAP_MULTIPLE: f32 = 3.0;
pub(super) fn merge_continuation_paragraphs(paragraphs: &mut Vec<PdfParagraph>) {
if paragraphs.len() < 2 {
return;
}
let old = std::mem::take(paragraphs);
let mut iter = old.into_iter();
let mut current = iter.next().unwrap();
for next in iter {
let both_body = current.heading_level.is_none()
&& next.heading_level.is_none()
&& !current.is_list_item
&& !next.is_list_item
&& !current.is_code_block
&& !next.is_code_block
&& !current.is_formula
&& !next.is_formula;
let fonts_compatible = (current.dominant_font_size - next.dominant_font_size).abs() < 2.0;
let bold_compatible = current.is_bold == next.is_bold;
let continuation_signal = !ends_with_sentence_terminator(¤t) || starts_with_lowercase_continuation(&next);
let same_region = current.layout_region_path == next.layout_region_path;
let same_rotation = paragraphs_share_rotation(¤t, &next);
let vertical_gap_compatible = baselines_within_continuation_gap(¤t, &next);
let next_starts_section = starts_numbered_section(&next);
let current_starts_section = starts_numbered_section(¤t);
let boundary_is_heading_wrap = current
.lines
.last()
.and_then(|line| line.segments.last())
.zip(next.lines.first().and_then(|line| line.segments.first()))
.is_some_and(|(prev_segment, next_segment)| {
super::pipeline::heading_wraps_onto(prev_segment, next_segment)
});
let next_right_edge = next
.lines
.iter()
.filter_map(|line| line.segments.last())
.map(|segment| segment.upright_advance_extent().1)
.filter(|edge| edge.is_finite())
.fold(f32::NEG_INFINITY, f32::max);
let heading_fills_column = current
.lines
.last()
.and_then(|line| line.segments.last())
.is_some_and(|prev_segment| super::pipeline::heading_fills_column(prev_segment, next_right_edge));
let heading_wrap_exempt =
boundary_is_heading_wrap || (starts_with_lowercase_continuation(&next) && heading_fills_column);
let should_merge = both_body
&& fonts_compatible
&& bold_compatible
&& continuation_signal
&& same_region
&& same_rotation
&& vertical_gap_compatible
&& !next_starts_section
&& (!current_starts_section || heading_wrap_exempt);
if should_merge {
current.text.clear();
current.block_bbox = union_block_bbox(current.block_bbox, next.block_bbox);
current.lines.extend(next.lines);
} else {
paragraphs.push(current);
current = next;
}
}
paragraphs.push(current);
}
fn paragraphs_share_rotation(current: &PdfParagraph, next: &PdfParagraph) -> bool {
let current_rotation = current.lines.last().and_then(|line| line.segments.last());
let next_rotation = next.lines.first().and_then(|line| line.segments.first());
match (current_rotation, next_rotation) {
(Some(current), Some(next)) => current.has_same_rotation(next),
_ => true,
}
}
fn baselines_within_continuation_gap(current: &PdfParagraph, next: &PdfParagraph) -> bool {
let (Some(current_last), Some(next_first)) = (current.lines.last(), next.lines.first()) else {
return true;
};
if current_last.baseline_y == 0.0 || next_first.baseline_y == 0.0 {
return true;
}
let gap = (current_last.baseline_y - next_first.baseline_y).abs();
let line_height = current.dominant_font_size.max(next.dominant_font_size).max(1.0);
gap <= line_height * MAX_CONTINUATION_LINE_GAP_MULTIPLE
}
fn union_block_bbox(
current: Option<(f32, f32, f32, f32)>,
next: Option<(f32, f32, f32, f32)>,
) -> Option<(f32, f32, f32, f32)> {
match (current, next) {
(Some((cl, cb, cr, ct)), Some((nl, nb, nr, nt))) => Some((cl.min(nl), cb.min(nb), cr.max(nr), ct.max(nt))),
(Some(bbox), None) | (None, Some(bbox)) => Some(bbox),
(None, None) => None,
}
}
fn starts_numbered_section(para: &PdfParagraph) -> bool {
let Some(first_line) = para.lines.first() else {
return false;
};
let joined = first_line
.segments
.iter()
.map(|segment| segment.text.trim())
.filter(|text| !text.is_empty())
.collect::<Vec<_>>()
.join(" ");
super::classify::is_numbered_section_heading(&joined)
}
fn starts_with_lowercase_continuation(para: &PdfParagraph) -> bool {
let first_text = para
.lines
.first()
.and_then(|l| l.segments.first())
.map(|s| s.text.trim_start())
.unwrap_or("");
first_text.chars().next().is_some_and(|c| c.is_lowercase())
}
fn ends_with_sentence_terminator(para: &PdfParagraph) -> bool {
let last_text = para
.lines
.last()
.and_then(|l| l.segments.last())
.map(|s| s.text.trim_end())
.unwrap_or("");
matches!(
last_text.chars().last(),
Some('.' | '?' | '!' | ':' | ';' | '\u{3002}' | '\u{FF1F}' | '\u{FF01}')
)
}
pub(super) fn split_embedded_list_items(paragraphs: &mut Vec<PdfParagraph>) {
let old = std::mem::take(paragraphs);
for para in old {
if para.heading_level.is_some() || para.is_list_item || para.is_code_block || para.is_formula {
paragraphs.push(para);
continue;
}
let full_text: String = para
.lines
.iter()
.flat_map(|l| l.segments.iter())
.map(|s| s.text.as_str())
.collect::<Vec<_>>()
.join(" ");
let bullet_count = full_text.matches(['\u{2022}', '\u{00B7}']).count();
if bullet_count < 2 {
paragraphs.push(para);
continue;
}
let font_size = para.dominant_font_size;
let is_bold = para.is_bold;
let parts: Vec<&str> = full_text.split(['\u{2022}', '\u{00B7}']).collect();
let before = parts[0].trim().trim_end_matches('\u{00C2}').trim();
if !before.is_empty() {
paragraphs.push(text_to_paragraph(before, font_size, is_bold, false));
}
for part in &parts[1..] {
let item_text = part
.trim()
.trim_start_matches('\u{00C2}')
.trim_end_matches('\u{00C2}')
.trim();
if !item_text.is_empty() {
paragraphs.push(text_to_paragraph(item_text, font_size, is_bold, true));
}
}
}
}
fn text_to_paragraph(text: &str, font_size: f32, is_bold: bool, is_list_item: bool) -> PdfParagraph {
use crate::pdf::hierarchy::SegmentData;
let segments: Vec<SegmentData> = text
.split_whitespace()
.map(|w| SegmentData {
text: w.to_string(),
x: 0.0,
y: 0.0,
width: 0.0,
height: 0.0,
font_size,
is_bold,
is_italic: false,
is_monospace: false,
baseline_y: 0.0,
rotation_degrees: 0.0,
assigned_role: None,
})
.collect();
let line = super::types::PdfLine {
segments,
baseline_y: 0.0,
dominant_font_size: font_size,
is_bold,
is_monospace: false,
};
let lines = vec![line];
let word_count = PdfParagraph::compute_word_count("", &lines);
PdfParagraph {
text: String::new(),
lines,
dominant_font_size: font_size,
heading_level: None,
is_bold,
is_list_item,
is_code_block: false,
is_formula: false,
is_page_furniture: false,
layout_class: None,
layout_region_path: None,
caption_for: None,
block_bbox: None,
word_count,
}
}
#[cfg(test)]
mod tests {
use super::*;
fn make_body_paragraph(text: &str, font_size: f32) -> PdfParagraph {
use crate::pdf::hierarchy::SegmentData;
let segments = vec![SegmentData {
text: text.to_string(),
x: 0.0,
y: 700.0,
width: 200.0,
height: font_size,
font_size,
is_bold: false,
is_italic: false,
is_monospace: false,
baseline_y: 700.0,
rotation_degrees: 0.0,
assigned_role: None,
}];
let lines = vec![super::super::types::PdfLine {
segments,
baseline_y: 700.0,
dominant_font_size: font_size,
is_bold: false,
is_monospace: false,
}];
let word_count = PdfParagraph::compute_word_count("", &lines);
PdfParagraph {
text: String::new(),
lines,
dominant_font_size: font_size,
heading_level: None,
is_bold: false,
is_list_item: false,
is_code_block: false,
is_formula: false,
is_page_furniture: false,
layout_class: None,
layout_region_path: None,
caption_for: None,
block_bbox: None,
word_count,
}
}
#[test]
fn should_not_merge_paragraphs_across_rotation_boundary() {
let mut rotated = make_body_paragraph("Engine oil need only meet the", 12.0);
rotated.lines[0].segments[0].rotation_degrees = 90.0;
let footer = make_body_paragraph("vehicle footer", 12.0);
let mut paragraphs = vec![rotated, footer];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(paragraphs.len(), 2);
}
fn make_body_paragraph_at(text: &str, font_size: f32, baseline_y: f32) -> PdfParagraph {
let mut para = make_body_paragraph(text, font_size);
para.lines[0].baseline_y = baseline_y;
if let Some(segment) = para.lines[0].segments.first_mut() {
segment.baseline_y = baseline_y;
segment.y = baseline_y;
}
para
}
#[test]
fn test_no_merge_across_distant_regions() {
let mut paragraphs = vec![
make_body_paragraph_at("Buyer tax ID SYNTH-BUYER-TAX-359370919", 8.0, 185.5),
make_body_paragraph_at("Seller tax ID SYNTH-SELLER-TAX-815876165", 8.0, 775.0),
];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(
paragraphs.len(),
2,
"paragraphs from spatially distant regions must not merge"
);
}
#[test]
fn test_merge_adjacent_lines_within_gap() {
let mut paragraphs = vec![
make_body_paragraph_at("The committee reviewed the annual", 11.0, 712.0),
make_body_paragraph_at("report and approved the budget", 11.0, 698.0),
];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(paragraphs.len(), 1, "adjacent continuation lines should merge");
}
#[test]
fn test_merge_unions_block_bbox() {
let mut upper = make_body_paragraph_at("first line without terminator", 11.0, 712.0);
upper.block_bbox = Some((60.0, 705.0, 260.0, 720.0));
let mut lower = make_body_paragraph_at("second line continues", 11.0, 698.0);
lower.block_bbox = Some((60.0, 691.0, 300.0, 706.0));
let mut paragraphs = vec![upper, lower];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(paragraphs.len(), 1, "adjacent lines should merge");
assert_eq!(
paragraphs[0].block_bbox,
Some((60.0, 691.0, 300.0, 720.0)),
"merged block bbox must span both source boxes"
);
}
#[test]
fn test_merge_lowercase_continuation() {
let mut paragraphs = vec![
make_body_paragraph("The regulation requires.", 12.0),
make_body_paragraph("and all operators must comply", 12.0),
];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(paragraphs.len(), 1, "lowercase continuation should be merged");
}
#[test]
fn test_no_merge_different_font_sizes() {
let mut paragraphs = vec![
make_body_paragraph("First paragraph", 12.0),
make_body_paragraph("second paragraph", 16.0),
];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(paragraphs.len(), 2, "different font sizes should prevent merge");
}
#[test]
fn test_merge_no_terminator() {
let mut paragraphs = vec![
make_body_paragraph("The regulation requires", 12.0),
make_body_paragraph("All operators must comply", 12.0),
];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(paragraphs.len(), 1, "unterminated paragraph should merge with next");
}
#[test]
fn test_no_merge_terminated_uppercase() {
let mut paragraphs = vec![
make_body_paragraph("The regulation requires compliance.", 12.0),
make_body_paragraph("All operators must comply", 12.0),
];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(
paragraphs.len(),
2,
"terminated paragraph + uppercase start should not merge"
);
}
#[test]
fn test_no_merge_across_bold_boundary() {
let body = make_body_paragraph(
"here is also available other sources of this Manual MetcalUser Guide",
12.0,
);
let mut header = make_body_paragraph("Impaired Glucose Tolerance And Impaired Fasting Glucose ...", 12.0);
header.is_bold = true;
let mut paragraphs = vec![body, header];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(paragraphs.len(), 2, "bold header must not merge into non-bold prose");
assert!(paragraphs[1].is_bold, "the bold header paragraph must be preserved");
}
fn first_line_text(para: &PdfParagraph) -> String {
para.lines
.first()
.map(|line| {
line.segments
.iter()
.map(|s| s.text.trim())
.collect::<Vec<_>>()
.join(" ")
})
.unwrap_or_default()
}
#[test]
fn should_not_merge_consecutive_numbered_section_headings() {
let mut paragraphs = vec![
make_body_paragraph_at("1.3 Gasinstallatie", 11.0, 700.0),
make_body_paragraph_at("1.4 Elektrische installatie", 11.0, 686.0),
make_body_paragraph_at("1.5 Waterinstallatie", 11.0, 672.0),
make_body_paragraph_at("1.6 Ventilatie", 11.0, 658.0),
];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(
paragraphs.len(),
4,
"each numbered subsection heading must remain a separate element"
);
assert_eq!(first_line_text(¶graphs[0]), "1.3 Gasinstallatie");
assert_eq!(first_line_text(¶graphs[1]), "1.4 Elektrische installatie");
assert_eq!(first_line_text(¶graphs[2]), "1.5 Waterinstallatie");
assert_eq!(first_line_text(¶graphs[3]), "1.6 Ventilatie");
}
#[test]
fn should_merge_prose_paragraph_starting_with_a_year() {
let mut paragraphs = vec![
make_body_paragraph_at("Het bestuur meldt", 11.0, 700.0),
make_body_paragraph_at("2024 was een druk jaar", 11.0, 686.0),
];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(
paragraphs.len(),
1,
"prose beginning with a bare year is not a section heading and must stay one paragraph"
);
assert_eq!(paragraphs[0].lines.len(), 2, "both prose lines must be present");
}
#[test]
fn should_not_merge_roman_or_allcaps_numbered_headings() {
let mut paragraphs = vec![
make_body_paragraph_at("III. Scope", 11.0, 700.0),
make_body_paragraph_at("IV. Results", 11.0, 686.0),
make_body_paragraph_at("5. CONCLUSIONS", 11.0, 672.0),
];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(
paragraphs.len(),
3,
"roman and ALL-CAPS numbered headings must not merge"
);
assert_eq!(first_line_text(¶graphs[0]), "III. Scope");
assert_eq!(first_line_text(¶graphs[1]), "IV. Results");
assert_eq!(first_line_text(¶graphs[2]), "5. CONCLUSIONS");
}
#[test]
fn should_still_merge_single_level_mixed_case_numbering() {
let mut paragraphs = vec![
make_body_paragraph_at("De procedure verloopt als volgt", 11.0, 700.0),
make_body_paragraph_at("1. Eerste stap in het proces", 11.0, 686.0),
];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(
paragraphs.len(),
1,
"'1. Eerste stap' is list-shaped, not a section heading; the guard must not fire"
);
}
#[test]
fn test_starts_with_lowercase_continuation_fn() {
let para_lower = make_body_paragraph("and furthermore", 12.0);
assert!(starts_with_lowercase_continuation(¶_lower));
let para_upper = make_body_paragraph("Furthermore", 12.0);
assert!(!starts_with_lowercase_continuation(¶_upper));
}
#[test]
fn test_merge_clears_precomputed_text_on_heuristic_path() {
let mut p1 = make_body_paragraph("een indicative", 12.0);
p1.text = "een indicative".to_string();
let mut p2 = make_body_paragraph("van toenemende merkbekendheid", 12.0);
p2.text = "van toenemende merkbekendheid".to_string();
let mut paragraphs = vec![p1, p2];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(paragraphs.len(), 1, "lowercase continuation should merge");
assert!(
paragraphs[0].text.is_empty(),
"merged paragraph must clear pre-computed text so assembly joins from segments"
);
assert_eq!(paragraphs[0].lines.len(), 2, "both lines must be present after merge");
}
#[test]
fn test_merge_struct_tree_path_text_stays_empty() {
let mut paragraphs = vec![
make_body_paragraph("first sentence without terminator", 12.0),
make_body_paragraph("second continues here", 12.0),
];
assert!(paragraphs[0].text.is_empty());
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(paragraphs.len(), 1);
assert!(paragraphs[0].text.is_empty());
}
fn wrapped_heading_paragraph(text: &str, x: f32, right_edge: f32, baseline_y: f32) -> PdfParagraph {
use crate::pdf::hierarchy::SegmentData;
let segments = vec![SegmentData {
text: text.to_string(),
x,
y: baseline_y,
width: right_edge - x,
height: 12.0,
font_size: 12.0,
is_bold: true,
is_italic: false,
is_monospace: false,
baseline_y,
rotation_degrees: 0.0,
assigned_role: None,
}];
let lines = vec![super::super::types::PdfLine {
segments,
baseline_y,
dominant_font_size: 12.0,
is_bold: true,
is_monospace: false,
}];
let word_count = PdfParagraph::compute_word_count("", &lines);
PdfParagraph {
text: String::new(),
lines,
dominant_font_size: 12.0,
heading_level: None,
is_bold: true,
is_list_item: false,
is_code_block: false,
is_formula: false,
is_page_furniture: false,
layout_class: None,
layout_region_path: None,
caption_for: None,
block_bbox: None,
word_count,
}
}
#[test]
fn numbered_heading_wrapping_onto_a_short_line_stays_one_paragraph() {
let mut paragraphs = vec![
wrapped_heading_paragraph("2.4 Aandachtspunten ten behoeve van de", 72.0, 307.5, 700.0),
wrapped_heading_paragraph("watertechnische installatie", 72.0, 248.8, 684.0),
];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(
paragraphs.len(),
1,
"GH#1605: a heading wrapping onto a short continuation must stay one paragraph; \
the continuation starts lowercase and carries no section number of its own"
);
}
#[test]
fn numbered_heading_wrapping_onto_a_long_line_stays_one_paragraph() {
let mut paragraphs = vec![
wrapped_heading_paragraph("2.4 Aandachtspunten ten behoeve van de", 72.0, 307.5, 700.0),
wrapped_heading_paragraph("watertechnische installatie en de meting", 72.0, 325.5, 684.0),
];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(
paragraphs.len(),
1,
"GH#1605 control: this case already merged and must keep merging"
);
}
#[test]
fn numbered_heading_followed_by_unrelated_capitalised_text_still_splits() {
let mut paragraphs = vec![
wrapped_heading_paragraph("1.1.1 Pictogrammen in het installatievoorschrift", 72.0, 374.2, 700.0),
wrapped_heading_paragraph("VOORZICHTIG / BELANGRIJK Procedures die schade", 72.0, 402.7, 684.0),
];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(
paragraphs.len(),
2,
"GH#1605 negative control: unrelated capitalised text after a heading must NOT be absorbed"
);
}
#[test]
fn numbered_heading_followed_by_wider_lowercase_body_still_splits() {
let mut paragraphs = vec![
wrapped_heading_paragraph("3.1.7 Innovatie/ontwikkelingen", 104.42, 230.0, 700.0),
wrapped_heading_paragraph(
"innovatie ontwikkelingen toekomstige verwachten gebied",
104.42,
500.0,
688.0,
),
];
merge_continuation_paragraphs(&mut paragraphs);
assert_eq!(
paragraphs.len(),
2,
"a complete heading must not absorb wider body prose merely because it opens lowercase"
);
}
}