use serde::Serialize;
use std::io::Write;
use xberg::{ExtractedDocument, ExtractionErrorItem, ProcessingWarning};
pub const WARNING_PREFIX: &str = "warning";
pub const ENVELOPE_HEADER: &str = "--- extraction envelope ---";
pub fn write_processing_warnings<W: Write>(warnings: &[ProcessingWarning], out: &mut W) -> std::io::Result<()> {
for warning in warnings {
writeln!(out, "{WARNING_PREFIX} [{}]: {}", warning.source, warning.message)?;
}
Ok(())
}
pub fn write_text_envelope<W: Write>(
document: &ExtractedDocument,
extraction_time_ms: f64,
out: &mut W,
) -> std::io::Result<()> {
write_processing_warnings(&document.processing_warnings, out)?;
writeln!(out, "{ENVELOPE_HEADER}")?;
writeln!(out, "mime type: {}", document.mime_type)?;
if document.counts.pages > 0 {
writeln!(out, "pages: {}", document.counts.pages)?;
}
if document.counts.tables > 0 {
writeln!(out, "tables: {}", document.counts.tables)?;
}
if document.counts.images > 0 {
writeln!(out, "images: {}", document.counts.images)?;
}
if let Some(languages) = document.detected_languages.as_ref().filter(|list| !list.is_empty()) {
writeln!(out, "languages: {}", languages.join(", "))?;
}
if let Some(score) = document.quality_score {
writeln!(out, "quality score: {score:.2}")?;
}
if let Some(chunks) = &document.chunks {
writeln!(out, "chunks: {}", chunks.len())?;
}
if let Some(entities) = &document.entities {
writeln!(out, "entities: {}", entities.len())?;
}
writeln!(out, "extraction time: {extraction_time_ms:.2} ms")
}
#[derive(Debug, Clone, Copy, Serialize)]
pub struct StageTimings {
pub process_init_ms: f64,
pub first_parse_ms: f64,
#[serde(skip_serializing_if = "Option::is_none")]
pub ort_session_and_inference_ms: Option<f64>,
}
#[derive(Debug, Serialize)]
pub struct ExtractEnvelope {
pub result: ExtractedDocument,
pub extraction_time_ms: f64,
pub peak_memory_bytes: u64,
#[serde(skip_serializing_if = "Option::is_none")]
pub stage_timings: Option<StageTimings>,
}
#[derive(Debug, Serialize)]
pub struct BatchEnvelope {
pub results: Vec<ExtractedDocument>,
pub total_ms: f64,
pub per_file_ms: Vec<Option<f64>>,
#[serde(skip_serializing_if = "Vec::is_empty")]
pub errors: Vec<ExtractionErrorItem>,
}
#[cfg(test)]
mod tests {
use super::*;
fn warning(source: &'static str, message: &'static str) -> ProcessingWarning {
ProcessingWarning {
source: source.into(),
message: message.into(),
}
}
fn rendered(document: &ExtractedDocument, extraction_time_ms: f64) -> String {
let mut buffer = Vec::new();
write_text_envelope(document, extraction_time_ms, &mut buffer).expect("writing to a Vec cannot fail");
String::from_utf8(buffer).expect("renderer emits UTF-8")
}
#[test]
fn write_processing_warnings_renders_source_and_message_for_each_warning() {
let warnings = vec![
warning("chunking", "chunk overlap exceeded chunk size"),
warning("language_detection", "model unavailable"),
];
let mut buffer = Vec::new();
write_processing_warnings(&warnings, &mut buffer).unwrap();
assert_eq!(
String::from_utf8(buffer).unwrap(),
"warning [chunking]: chunk overlap exceeded chunk size\n\
warning [language_detection]: model unavailable\n"
);
}
#[test]
fn write_processing_warnings_writes_nothing_when_there_are_no_warnings() {
let mut buffer = Vec::new();
write_processing_warnings(&[], &mut buffer).unwrap();
assert_eq!(String::from_utf8(buffer).unwrap(), "");
}
#[test]
fn write_text_envelope_includes_processing_warning_text() {
let mut document = ExtractedDocument::default();
document.mime_type = "text/plain".into();
document.processing_warnings = vec![warning("embedding", "backend not configured")];
assert_eq!(
rendered(&document, 12.5),
"warning [embedding]: backend not configured\n\
--- extraction envelope ---\n\
mime type: text/plain\n\
extraction time: 12.50 ms\n"
);
}
#[test]
fn write_text_envelope_reports_counts_languages_quality_chunks_and_entities() {
let mut document = ExtractedDocument::default();
document.mime_type = "application/pdf".into();
document.counts = xberg::DocumentCounts {
pages: 3,
tables: 2,
images: 1,
};
document.detected_languages = Some(vec!["en".to_string(), "de".to_string()]);
document.quality_score = Some(0.5);
document.chunks = Some(Vec::new());
document.entities = Some(Vec::new());
assert_eq!(
rendered(&document, 250.0),
"--- extraction envelope ---\n\
mime type: application/pdf\n\
pages: 3\n\
tables: 2\n\
images: 1\n\
languages: en, de\n\
quality score: 0.50\n\
chunks: 0\n\
entities: 0\n\
extraction time: 250.00 ms\n"
);
}
#[test]
fn write_text_envelope_omits_sections_that_carry_no_data() {
let mut document = ExtractedDocument::default();
document.mime_type = "text/plain".into();
document.detected_languages = Some(Vec::new());
let output = rendered(&document, 1.0);
assert_eq!(
output,
"--- extraction envelope ---\n\
mime type: text/plain\n\
extraction time: 1.00 ms\n"
);
assert!(!output.contains("pages:"), "zero page count must not be reported");
assert!(
!output.contains("languages:"),
"an empty language list must not be reported"
);
}
}