mod helpers;
use xberg::core::config::OutputFormat;
use xberg::extraction::derive::derive_extraction_result;
use xberg::types::document_structure::{AnnotationKind, TextAnnotation};
use xberg::types::internal_builder::InternalDocumentBuilder;
fn build_rich_document() -> xberg::types::internal::InternalDocument {
let mut b = InternalDocumentBuilder::new("test");
b.push_heading(1, "Main Heading", None, None);
b.push_paragraph("This is a paragraph with some text.", vec![], None, None);
b.push_list(false);
b.push_list_item("First item", false, vec![], None, None);
b.push_list_item("Second item", false, vec![], None, None);
b.push_list_item("Third item", false, vec![], None, None);
b.end_list();
b.push_code("fn main() {\n println!(\"hello\");\n}", Some("rust"), None, None);
b.push_table_from_cells(
&[
vec!["Name".to_string(), "Value".to_string()],
vec!["alpha".to_string(), "1".to_string()],
vec!["beta".to_string(), "2".to_string()],
],
None,
None,
);
b.build()
}
fn derive(doc: xberg::types::internal::InternalDocument, format: OutputFormat) -> xberg::types::ExtractedDocument {
derive_extraction_result(doc, false, format)
}
fn effective_content(result: &xberg::types::ExtractedDocument) -> &str {
result.formatted_content.as_deref().unwrap_or(&result.content)
}
#[tokio::test]
async fn test_markdown_output_preserves_structure() {
let doc = build_rich_document();
let result = derive(doc, OutputFormat::Markdown);
let md = effective_content(&result);
assert!(
md.contains("# Main Heading"),
"Markdown should contain an ATX heading, got:\n{md}"
);
assert!(
md.contains("This is a paragraph"),
"Markdown should contain the paragraph text"
);
assert!(md.contains("First item"), "Markdown should contain list items");
assert!(md.contains("```"), "Markdown should contain a fenced code block");
assert!(md.contains("fn main()"), "Markdown code block should contain the code");
assert!(md.contains('|'), "Markdown should contain pipe-delimited table syntax");
assert!(md.contains("Name"), "Markdown table should contain header cells");
}
#[tokio::test]
async fn test_djot_output_preserves_structure() {
let doc = build_rich_document();
let result = derive(doc, OutputFormat::Djot);
let djot = effective_content(&result);
assert!(
djot.contains("# Main Heading"),
"Djot should contain a heading, got:\n{djot}"
);
assert!(
djot.contains("This is a paragraph"),
"Djot should contain the paragraph text"
);
assert!(djot.contains("fn main()"), "Djot should contain the code content");
}
#[tokio::test]
async fn test_html_output_preserves_structure() {
let doc = build_rich_document();
let result = derive(doc, OutputFormat::Html);
let html = effective_content(&result);
assert!(html.contains("<h1"), "HTML should contain an h1 tag, got:\n{html}");
assert!(html.contains("Main Heading"), "HTML h1 should contain the heading text");
assert!(html.contains("<p"), "HTML should contain paragraph tags");
assert!(html.contains("<li"), "HTML should contain list item tags");
assert!(
html.contains("<code") || html.contains("<pre"),
"HTML should contain code/pre tags"
);
assert!(
html.contains("<table") || html.contains("<th") || html.contains("<td"),
"HTML should contain table markup"
);
}
#[tokio::test]
async fn test_plain_text_output_has_no_formatting() {
let doc = build_rich_document();
let result = derive(doc, OutputFormat::Plain);
assert!(
result.formatted_content.is_none(),
"Plain format should not set formatted_content"
);
let plain = &result.content;
assert!(plain.contains("Main Heading"), "Plain text should contain heading text");
assert!(
plain.contains("This is a paragraph"),
"Plain text should contain paragraph text"
);
assert!(plain.contains("First item"), "Plain text should contain list items");
assert!(plain.contains("fn main()"), "Plain text should contain code content");
assert!(!plain.contains("<h1"), "Plain text should not contain HTML tags");
assert!(
!plain.lines().any(|l| l.starts_with("# ")),
"Plain text should not contain markdown heading syntax"
);
}
#[tokio::test]
async fn test_format_switching_consistency() {
let formats = [
OutputFormat::Plain,
OutputFormat::Markdown,
OutputFormat::Djot,
OutputFormat::Html,
];
let expected_words = [
"Main Heading",
"paragraph",
"First item",
"Second item",
"Third item",
"fn main()",
"alpha",
"beta",
];
for format in &formats {
let doc = build_rich_document();
let result = derive(doc, format.clone());
let text = effective_content(&result);
for word in &expected_words {
assert!(
text.contains(word),
"Format {format:?} should contain \"{word}\" but output was:\n{text}"
);
}
}
}
#[tokio::test]
async fn test_footnote_rendering_end_to_end() {
let mut b = InternalDocumentBuilder::new("test");
b.push_paragraph("Text with a footnote reference.", vec![], None, None);
b.push_footnote_ref("1", "fn1", None);
b.push_paragraph("More text after the reference.", vec![], None, None);
b.push_footnote_definition("This is the footnote content.", "fn1", None);
let doc = b.build();
let result = derive(doc, OutputFormat::Markdown);
let md = effective_content(&result);
assert!(
md.contains("[^") || md.contains("fn1") || md.contains("[1]"),
"Markdown should contain a footnote reference marker, got:\n{md}"
);
assert!(
md.contains("This is the footnote content"),
"Markdown should contain the footnote definition text, got:\n{md}"
);
}
#[tokio::test]
async fn test_annotation_rendering_end_to_end() {
let mut b = InternalDocumentBuilder::new("test");
let bold_text = "Hello bold world";
b.push_paragraph(
bold_text,
vec![TextAnnotation {
start: 6,
end: 10,
kind: AnnotationKind::Bold,
}],
None,
None,
);
let italic_text = "Some italic text";
b.push_paragraph(
italic_text,
vec![TextAnnotation {
start: 5,
end: 11,
kind: AnnotationKind::Italic,
}],
None,
None,
);
let link_text = "Click here for info";
b.push_paragraph(
link_text,
vec![TextAnnotation {
start: 6,
end: 10,
kind: AnnotationKind::Link {
url: "https://example.com".to_string(),
title: None,
},
}],
None,
None,
);
let doc = b.build();
let result = derive(doc, OutputFormat::Markdown);
let md = effective_content(&result);
assert!(
md.contains("**bold**"),
"Markdown should render bold annotation as **bold**, got:\n{md}"
);
assert!(
md.contains("*italic*"),
"Markdown should render italic annotation as *italic*, got:\n{md}"
);
assert!(
md.contains("[here](https://example.com)"),
"Markdown should render link annotation, got:\n{md}"
);
}
#[tokio::test]
async fn test_empty_document_handling() {
let formats = [
OutputFormat::Plain,
OutputFormat::Markdown,
OutputFormat::Djot,
OutputFormat::Html,
];
for format in &formats {
let b = InternalDocumentBuilder::new("test");
let doc = b.build();
let result = derive(doc, format.clone());
let text = effective_content(&result);
assert!(
text.len() < 100,
"Empty document in {format:?} should produce minimal output, got {} chars",
text.len()
);
}
}
#[tokio::test]
async fn test_large_document_renders_without_timeout() {
let mut b = InternalDocumentBuilder::new("test");
b.push_heading(1, "Large Document", None, None);
for i in 0..1000 {
b.push_paragraph(
&format!("Paragraph number {i} with some filler text to make it realistic."),
vec![],
None,
None,
);
}
let _doc = b.build();
let formats = [
OutputFormat::Plain,
OutputFormat::Markdown,
OutputFormat::Djot,
OutputFormat::Html,
];
for format in &formats {
let mut b2 = InternalDocumentBuilder::new("test");
b2.push_heading(1, "Large Document", None, None);
for i in 0..1000 {
b2.push_paragraph(
&format!("Paragraph number {i} with some filler text to make it realistic."),
vec![],
None,
None,
);
}
let doc2 = b2.build();
let result = derive(doc2, format.clone());
let text = effective_content(&result);
assert!(
text.contains("Paragraph number 0"),
"{format:?}: should contain first paragraph"
);
assert!(
text.contains("Paragraph number 999"),
"{format:?}: should contain last paragraph"
);
assert!(
text.len() > 50_000,
"{format:?}: 1000-paragraph document should produce substantial output, got {} bytes",
text.len()
);
}
}