Skip to main content

mcd_core/
pdf.rs

1//! Simple PDF-to-MCD conversion.
2
3use std::io::{Cursor, Write};
4
5use zip::{CompressionMethod, ZipWriter, write::SimpleFileOptions};
6
7use crate::{
8    errors::{Diagnostic, McdError},
9    manifest::{AssetManifestEntry, ConformanceClaim, LayoutManifestEntry, Manifest, McdProfile},
10    package::MCD_MIMETYPE,
11};
12
13/// Options for converting a PDF into a minimal MCD package.
14#[derive(Debug, Clone, Default, PartialEq, Eq)]
15pub struct PdfConversionOptions {
16    /// Optional document title for the MCD manifest and Markdown heading.
17    pub title: Option<String>,
18    /// Optional original file name used for the embedded PDF asset path.
19    pub source_filename: Option<String>,
20}
21
22/// Convert PDF bytes into a minimal MCD archive.
23///
24/// The converter extracts text into `content/main.md` and embeds the original
25/// PDF as `assets/<source_filename>`. It does not attempt OCR, table recovery,
26/// or layout reconstruction.
27pub fn pdf_to_mcd_bytes(pdf: &[u8], options: PdfConversionOptions) -> crate::Result<Vec<u8>> {
28    if !pdf.starts_with(b"%PDF-") {
29        return Err(McdError::from_diagnostic(Diagnostic::error(
30            "pdf.signature.invalid",
31            "Input does not look like a PDF file.",
32        )));
33    }
34
35    let pages = extract_pdf_pages(pdf)?;
36    let asset_path = format!(
37        "assets/{}",
38        sanitize_pdf_filename(options.source_filename.as_deref())
39    );
40    let title = options
41        .title
42        .filter(|title| !title.trim().is_empty())
43        .unwrap_or_else(|| title_from_asset_path(&asset_path));
44    let markdown = markdown_from_pdf_pages(&title, &asset_path, &pages);
45    let manifest = Manifest {
46        format: "MCD".to_owned(),
47        version: "0.1".to_owned(),
48        profile: McdProfile::Core,
49        conformance: vec![ConformanceClaim::Core],
50        entrypoint: "content/main.md".to_owned(),
51        title: Some(title),
52        encoding: Some("utf-8".to_owned()),
53        tables: Vec::new(),
54        images: Vec::new(),
55        annotations: Vec::new(),
56        assets: vec![AssetManifestEntry {
57            id: Some("source-pdf".to_owned()),
58            path: asset_path.clone(),
59        }],
60        external_data: Vec::new(),
61        provenance: None,
62        layout: None::<LayoutManifestEntry>,
63    };
64
65    let manifest_json = serde_json::to_vec_pretty(&manifest)?;
66    let mut archive = ZipWriter::new(Cursor::new(Vec::new()));
67    let stored = SimpleFileOptions::default().compression_method(CompressionMethod::Stored);
68    let deflated = SimpleFileOptions::default().compression_method(CompressionMethod::Deflated);
69
70    archive.start_file("mimetype", stored)?;
71    archive.write_all(MCD_MIMETYPE.as_bytes())?;
72    archive.write_all(b"\n")?;
73    archive.start_file("manifest.json", deflated)?;
74    archive.write_all(&manifest_json)?;
75    archive.start_file("content/main.md", deflated)?;
76    archive.write_all(markdown.as_bytes())?;
77    archive.start_file(asset_path, deflated)?;
78    archive.write_all(pdf)?;
79
80    Ok(archive.finish()?.into_inner())
81}
82
83fn extract_pdf_pages(pdf: &[u8]) -> crate::Result<Vec<String>> {
84    #[cfg(not(target_arch = "wasm32"))]
85    {
86        match pdf_extract::extract_text_from_mem_by_pages(pdf) {
87            Ok(pages) => Ok(pages),
88            Err(err) => {
89                let fallback_pages = fallback_extract_literal_text(pdf);
90                if fallback_pages.is_empty() {
91                    Err(pdf_error(err))
92                } else {
93                    Ok(fallback_pages)
94                }
95            }
96        }
97    }
98    #[cfg(target_arch = "wasm32")]
99    {
100        Ok(fallback_extract_literal_text(pdf))
101    }
102}
103
104#[cfg(not(target_arch = "wasm32"))]
105fn pdf_error(err: pdf_extract::OutputError) -> McdError {
106    McdError::from_diagnostic(Diagnostic::error(
107        "pdf.text.extract.failed",
108        format!("Failed to extract text from PDF: {err}"),
109    ))
110}
111
112fn fallback_extract_literal_text(pdf: &[u8]) -> Vec<String> {
113    let mut strings = Vec::new();
114    let mut index = 0;
115    while index < pdf.len() {
116        if pdf[index] != b'(' {
117            index += 1;
118            continue;
119        }
120
121        if let Some((value, end)) = parse_pdf_literal_string(pdf, index + 1) {
122            if looks_like_text_showing_operator(pdf, end) && !value.trim().is_empty() {
123                strings.push(value);
124            }
125            index = end + 1;
126        } else {
127            index += 1;
128        }
129    }
130
131    if strings.is_empty() {
132        Vec::new()
133    } else {
134        vec![strings.join("\n")]
135    }
136}
137
138fn parse_pdf_literal_string(pdf: &[u8], mut index: usize) -> Option<(String, usize)> {
139    let mut value = Vec::new();
140    let mut depth = 1_u32;
141    while index < pdf.len() {
142        let byte = pdf[index];
143        match byte {
144            b'\\' => {
145                index += 1;
146                if index >= pdf.len() {
147                    return None;
148                }
149                match pdf[index] {
150                    b'n' => value.push(b'\n'),
151                    b'r' => value.push(b'\r'),
152                    b't' => value.push(b'\t'),
153                    b'b' => value.push(0x08),
154                    b'f' => value.push(0x0c),
155                    b'\n' => {}
156                    b'\r' => {
157                        if pdf.get(index + 1) == Some(&b'\n') {
158                            index += 1;
159                        }
160                    }
161                    escaped => value.push(escaped),
162                }
163            }
164            b'(' => {
165                depth += 1;
166                value.push(byte);
167            }
168            b')' => {
169                depth -= 1;
170                if depth == 0 {
171                    return Some((String::from_utf8_lossy(&value).into_owned(), index));
172                }
173                value.push(byte);
174            }
175            _ => value.push(byte),
176        }
177        index += 1;
178    }
179    None
180}
181
182fn looks_like_text_showing_operator(pdf: &[u8], end: usize) -> bool {
183    let tail_start = end.saturating_add(1);
184    let tail_end = tail_start.saturating_add(32).min(pdf.len());
185    let tail = &pdf[tail_start..tail_end];
186    let tail = trim_ascii_start(tail);
187    tail.starts_with(b"Tj")
188        || tail.starts_with(b"'")
189        || tail.starts_with(b"\"")
190        || tail.windows(2).take(16).any(|window| window == b"TJ")
191}
192
193fn trim_ascii_start(mut bytes: &[u8]) -> &[u8] {
194    while let Some((first, rest)) = bytes.split_first() {
195        if !first.is_ascii_whitespace() {
196            break;
197        }
198        bytes = rest;
199    }
200    bytes
201}
202
203fn markdown_from_pdf_pages(title: &str, asset_path: &str, pages: &[String]) -> String {
204    let mut markdown = String::new();
205    markdown.push_str("# ");
206    markdown.push_str(&escape_heading(title));
207    markdown.push_str("\n\n");
208    markdown.push_str("Source PDF asset: `");
209    markdown.push_str(asset_path);
210    markdown.push_str("`.\n");
211
212    if pages.is_empty() || pages.iter().all(|page| page.trim().is_empty()) {
213        markdown.push_str("\n_No extractable text was found in the PDF._\n");
214        return markdown;
215    }
216
217    for (index, page) in pages.iter().enumerate() {
218        let text = normalize_pdf_text(page);
219        if text.trim().is_empty() {
220            continue;
221        }
222        markdown.push_str("\n\n## Page ");
223        markdown.push_str(&(index + 1).to_string());
224        markdown.push_str("\n\n");
225        markdown.push_str(&text);
226    }
227    markdown.push('\n');
228    markdown
229}
230
231fn normalize_pdf_text(text: &str) -> String {
232    text.replace("\r\n", "\n")
233        .replace('\r', "\n")
234        .lines()
235        .map(str::trim_end)
236        .collect::<Vec<_>>()
237        .join("\n")
238        .trim()
239        .to_owned()
240}
241
242fn sanitize_pdf_filename(source_filename: Option<&str>) -> String {
243    let source_filename = source_filename
244        .and_then(|path| {
245            path.rsplit(['/', '\\'])
246                .find(|part| !part.trim().is_empty())
247        })
248        .unwrap_or("source.pdf");
249    let mut sanitized = source_filename
250        .chars()
251        .map(|character| {
252            if character.is_ascii_alphanumeric() || matches!(character, '.' | '-' | '_') {
253                character
254            } else {
255                '_'
256            }
257        })
258        .collect::<String>();
259
260    while sanitized.contains("..") {
261        sanitized = sanitized.replace("..", ".");
262    }
263    sanitized = sanitized.trim_matches('.').to_owned();
264    if sanitized.is_empty() {
265        sanitized = "source.pdf".to_owned();
266    }
267    if !sanitized.to_ascii_lowercase().ends_with(".pdf") {
268        sanitized.push_str(".pdf");
269    }
270    sanitized
271}
272
273fn title_from_asset_path(asset_path: &str) -> String {
274    let file_name = asset_path.rsplit('/').next().unwrap_or("source.pdf");
275    let stem = file_name
276        .strip_suffix(".pdf")
277        .or_else(|| file_name.strip_suffix(".PDF"))
278        .unwrap_or(file_name);
279    let title = stem.replace(['_', '-'], " ");
280    if title.trim().is_empty() {
281        "Converted PDF".to_owned()
282    } else {
283        title.trim().to_owned()
284    }
285}
286
287fn escape_heading(value: &str) -> String {
288    value.replace('\n', " ").trim().to_owned()
289}
290
291#[cfg(test)]
292mod tests {
293    use super::*;
294
295    #[test]
296    fn rejects_non_pdf_input() {
297        let err = pdf_to_mcd_bytes(b"not a pdf", PdfConversionOptions::default())
298            .expect_err("non-pdf should fail");
299
300        assert_eq!(
301            err.diagnostic().map(|diagnostic| diagnostic.code.as_str()),
302            Some("pdf.signature.invalid")
303        );
304    }
305
306    #[test]
307    fn sanitizes_asset_file_names() {
308        assert_eq!(
309            sanitize_pdf_filename(Some(r"..\Quarterly Report 2026.pdf")),
310            "Quarterly_Report_2026.pdf"
311        );
312        assert_eq!(sanitize_pdf_filename(Some("report")), "report.pdf");
313    }
314
315    #[test]
316    fn renders_empty_pdf_text_as_valid_markdown() {
317        let markdown = markdown_from_pdf_pages("Report", "assets/report.pdf", &[]);
318
319        assert!(markdown.contains("# Report"));
320        assert!(markdown.contains("_No extractable text was found"));
321    }
322
323    #[test]
324    fn converts_simple_pdf_to_valid_mcd_package() {
325        let pdf = minimal_pdf("Hello from PDF");
326        let mcd = pdf_to_mcd_bytes(
327            &pdf,
328            PdfConversionOptions {
329                title: Some("PDF Import".to_owned()),
330                source_filename: Some("import.pdf".to_owned()),
331            },
332        )
333        .expect("pdf converts");
334        let package = crate::McdPackage::from_bytes(&mcd).expect("mcd opens");
335        crate::validate::validate_package(&package).expect("mcd validates");
336        let markdown = package
337            .read_to_string("content/main.md")
338            .expect("markdown exists");
339
340        assert!(markdown.contains("# PDF Import"));
341        assert!(markdown.contains("Hello from PDF"));
342        assert!(package.contains("assets/import.pdf"));
343    }
344
345    fn minimal_pdf(text: &str) -> Vec<u8> {
346        let escaped = text
347            .replace('\\', r"\\")
348            .replace('(', r"\(")
349            .replace(')', r"\)");
350        let content = format!("BT /F1 24 Tf 100 700 Td ({escaped}) Tj ET");
351        let objects = [
352            "<< /Type /Catalog /Pages 2 0 R >>".to_owned(),
353            "<< /Type /Pages /Kids [3 0 R] /Count 1 >>".to_owned(),
354            "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 4 0 R >> >> /Contents 5 0 R >>".to_owned(),
355            "<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>".to_owned(),
356            format!("<< /Length {} >>\nstream\n{}\nendstream", content.len(), content),
357        ];
358        let mut bytes = b"%PDF-1.4\n".to_vec();
359        let mut offsets = Vec::new();
360        for (index, object) in objects.iter().enumerate() {
361            offsets.push(bytes.len());
362            bytes
363                .extend_from_slice(format!("{} 0 obj\n{}\nendobj\n", index + 1, object).as_bytes());
364        }
365        let xref_offset = bytes.len();
366        bytes.extend_from_slice(format!("xref\n0 {}\n", objects.len() + 1).as_bytes());
367        bytes.extend_from_slice(b"0000000000 65535 f \n");
368        for offset in offsets {
369            bytes.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
370        }
371        bytes.extend_from_slice(
372            format!(
373                "trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF\n",
374                objects.len() + 1,
375                xref_offset
376            )
377            .as_bytes(),
378        );
379        bytes
380    }
381}