1use std::io::{Cursor, Write};
4
5use zip::{CompressionMethod, ZipWriter, write::SimpleFileOptions};
6
7use crate::{
8 errors::{Diagnostic, McdError},
9 manifest::{AssetManifestEntry, ConformanceClaim, LayoutManifestEntry, Manifest, McdProfile},
10 package::MCD_MIMETYPE,
11};
12
13#[derive(Debug, Clone, Default, PartialEq, Eq)]
15pub struct PdfConversionOptions {
16 pub title: Option<String>,
18 pub source_filename: Option<String>,
20}
21
22pub fn pdf_to_mcd_bytes(pdf: &[u8], options: PdfConversionOptions) -> crate::Result<Vec<u8>> {
28 if !pdf.starts_with(b"%PDF-") {
29 return Err(McdError::from_diagnostic(Diagnostic::error(
30 "pdf.signature.invalid",
31 "Input does not look like a PDF file.",
32 )));
33 }
34
35 let pages = extract_pdf_pages(pdf)?;
36 let asset_path = format!(
37 "assets/{}",
38 sanitize_pdf_filename(options.source_filename.as_deref())
39 );
40 let title = options
41 .title
42 .filter(|title| !title.trim().is_empty())
43 .unwrap_or_else(|| title_from_asset_path(&asset_path));
44 let markdown = markdown_from_pdf_pages(&title, &asset_path, &pages);
45 let manifest = Manifest {
46 format: "MCD".to_owned(),
47 version: "0.1".to_owned(),
48 profile: McdProfile::Core,
49 conformance: vec![ConformanceClaim::Core],
50 entrypoint: "content/main.md".to_owned(),
51 title: Some(title),
52 encoding: Some("utf-8".to_owned()),
53 tables: Vec::new(),
54 images: Vec::new(),
55 annotations: Vec::new(),
56 assets: vec![AssetManifestEntry {
57 id: Some("source-pdf".to_owned()),
58 path: asset_path.clone(),
59 }],
60 external_data: Vec::new(),
61 provenance: None,
62 layout: None::<LayoutManifestEntry>,
63 };
64
65 let manifest_json = serde_json::to_vec_pretty(&manifest)?;
66 let mut archive = ZipWriter::new(Cursor::new(Vec::new()));
67 let stored = SimpleFileOptions::default().compression_method(CompressionMethod::Stored);
68 let deflated = SimpleFileOptions::default().compression_method(CompressionMethod::Deflated);
69
70 archive.start_file("mimetype", stored)?;
71 archive.write_all(MCD_MIMETYPE.as_bytes())?;
72 archive.write_all(b"\n")?;
73 archive.start_file("manifest.json", deflated)?;
74 archive.write_all(&manifest_json)?;
75 archive.start_file("content/main.md", deflated)?;
76 archive.write_all(markdown.as_bytes())?;
77 archive.start_file(asset_path, deflated)?;
78 archive.write_all(pdf)?;
79
80 Ok(archive.finish()?.into_inner())
81}
82
83fn extract_pdf_pages(pdf: &[u8]) -> crate::Result<Vec<String>> {
84 #[cfg(not(target_arch = "wasm32"))]
85 {
86 match pdf_extract::extract_text_from_mem_by_pages(pdf) {
87 Ok(pages) => Ok(pages),
88 Err(err) => {
89 let fallback_pages = fallback_extract_literal_text(pdf);
90 if fallback_pages.is_empty() {
91 Err(pdf_error(err))
92 } else {
93 Ok(fallback_pages)
94 }
95 }
96 }
97 }
98 #[cfg(target_arch = "wasm32")]
99 {
100 Ok(fallback_extract_literal_text(pdf))
101 }
102}
103
104#[cfg(not(target_arch = "wasm32"))]
105fn pdf_error(err: pdf_extract::OutputError) -> McdError {
106 McdError::from_diagnostic(Diagnostic::error(
107 "pdf.text.extract.failed",
108 format!("Failed to extract text from PDF: {err}"),
109 ))
110}
111
112fn fallback_extract_literal_text(pdf: &[u8]) -> Vec<String> {
113 let mut strings = Vec::new();
114 let mut index = 0;
115 while index < pdf.len() {
116 if pdf[index] != b'(' {
117 index += 1;
118 continue;
119 }
120
121 if let Some((value, end)) = parse_pdf_literal_string(pdf, index + 1) {
122 if looks_like_text_showing_operator(pdf, end) && !value.trim().is_empty() {
123 strings.push(value);
124 }
125 index = end + 1;
126 } else {
127 index += 1;
128 }
129 }
130
131 if strings.is_empty() {
132 Vec::new()
133 } else {
134 vec![strings.join("\n")]
135 }
136}
137
138fn parse_pdf_literal_string(pdf: &[u8], mut index: usize) -> Option<(String, usize)> {
139 let mut value = Vec::new();
140 let mut depth = 1_u32;
141 while index < pdf.len() {
142 let byte = pdf[index];
143 match byte {
144 b'\\' => {
145 index += 1;
146 if index >= pdf.len() {
147 return None;
148 }
149 match pdf[index] {
150 b'n' => value.push(b'\n'),
151 b'r' => value.push(b'\r'),
152 b't' => value.push(b'\t'),
153 b'b' => value.push(0x08),
154 b'f' => value.push(0x0c),
155 b'\n' => {}
156 b'\r' => {
157 if pdf.get(index + 1) == Some(&b'\n') {
158 index += 1;
159 }
160 }
161 escaped => value.push(escaped),
162 }
163 }
164 b'(' => {
165 depth += 1;
166 value.push(byte);
167 }
168 b')' => {
169 depth -= 1;
170 if depth == 0 {
171 return Some((String::from_utf8_lossy(&value).into_owned(), index));
172 }
173 value.push(byte);
174 }
175 _ => value.push(byte),
176 }
177 index += 1;
178 }
179 None
180}
181
182fn looks_like_text_showing_operator(pdf: &[u8], end: usize) -> bool {
183 let tail_start = end.saturating_add(1);
184 let tail_end = tail_start.saturating_add(32).min(pdf.len());
185 let tail = &pdf[tail_start..tail_end];
186 let tail = trim_ascii_start(tail);
187 tail.starts_with(b"Tj")
188 || tail.starts_with(b"'")
189 || tail.starts_with(b"\"")
190 || tail.windows(2).take(16).any(|window| window == b"TJ")
191}
192
193fn trim_ascii_start(mut bytes: &[u8]) -> &[u8] {
194 while let Some((first, rest)) = bytes.split_first() {
195 if !first.is_ascii_whitespace() {
196 break;
197 }
198 bytes = rest;
199 }
200 bytes
201}
202
203fn markdown_from_pdf_pages(title: &str, asset_path: &str, pages: &[String]) -> String {
204 let mut markdown = String::new();
205 markdown.push_str("# ");
206 markdown.push_str(&escape_heading(title));
207 markdown.push_str("\n\n");
208 markdown.push_str("Source PDF asset: `");
209 markdown.push_str(asset_path);
210 markdown.push_str("`.\n");
211
212 if pages.is_empty() || pages.iter().all(|page| page.trim().is_empty()) {
213 markdown.push_str("\n_No extractable text was found in the PDF._\n");
214 return markdown;
215 }
216
217 for (index, page) in pages.iter().enumerate() {
218 let text = normalize_pdf_text(page);
219 if text.trim().is_empty() {
220 continue;
221 }
222 markdown.push_str("\n\n## Page ");
223 markdown.push_str(&(index + 1).to_string());
224 markdown.push_str("\n\n");
225 markdown.push_str(&text);
226 }
227 markdown.push('\n');
228 markdown
229}
230
231fn normalize_pdf_text(text: &str) -> String {
232 text.replace("\r\n", "\n")
233 .replace('\r', "\n")
234 .lines()
235 .map(str::trim_end)
236 .collect::<Vec<_>>()
237 .join("\n")
238 .trim()
239 .to_owned()
240}
241
242fn sanitize_pdf_filename(source_filename: Option<&str>) -> String {
243 let source_filename = source_filename
244 .and_then(|path| {
245 path.rsplit(['/', '\\'])
246 .find(|part| !part.trim().is_empty())
247 })
248 .unwrap_or("source.pdf");
249 let mut sanitized = source_filename
250 .chars()
251 .map(|character| {
252 if character.is_ascii_alphanumeric() || matches!(character, '.' | '-' | '_') {
253 character
254 } else {
255 '_'
256 }
257 })
258 .collect::<String>();
259
260 while sanitized.contains("..") {
261 sanitized = sanitized.replace("..", ".");
262 }
263 sanitized = sanitized.trim_matches('.').to_owned();
264 if sanitized.is_empty() {
265 sanitized = "source.pdf".to_owned();
266 }
267 if !sanitized.to_ascii_lowercase().ends_with(".pdf") {
268 sanitized.push_str(".pdf");
269 }
270 sanitized
271}
272
273fn title_from_asset_path(asset_path: &str) -> String {
274 let file_name = asset_path.rsplit('/').next().unwrap_or("source.pdf");
275 let stem = file_name
276 .strip_suffix(".pdf")
277 .or_else(|| file_name.strip_suffix(".PDF"))
278 .unwrap_or(file_name);
279 let title = stem.replace(['_', '-'], " ");
280 if title.trim().is_empty() {
281 "Converted PDF".to_owned()
282 } else {
283 title.trim().to_owned()
284 }
285}
286
287fn escape_heading(value: &str) -> String {
288 value.replace('\n', " ").trim().to_owned()
289}
290
291#[cfg(test)]
292mod tests {
293 use super::*;
294
295 #[test]
296 fn rejects_non_pdf_input() {
297 let err = pdf_to_mcd_bytes(b"not a pdf", PdfConversionOptions::default())
298 .expect_err("non-pdf should fail");
299
300 assert_eq!(
301 err.diagnostic().map(|diagnostic| diagnostic.code.as_str()),
302 Some("pdf.signature.invalid")
303 );
304 }
305
306 #[test]
307 fn sanitizes_asset_file_names() {
308 assert_eq!(
309 sanitize_pdf_filename(Some(r"..\Quarterly Report 2026.pdf")),
310 "Quarterly_Report_2026.pdf"
311 );
312 assert_eq!(sanitize_pdf_filename(Some("report")), "report.pdf");
313 }
314
315 #[test]
316 fn renders_empty_pdf_text_as_valid_markdown() {
317 let markdown = markdown_from_pdf_pages("Report", "assets/report.pdf", &[]);
318
319 assert!(markdown.contains("# Report"));
320 assert!(markdown.contains("_No extractable text was found"));
321 }
322
323 #[test]
324 fn converts_simple_pdf_to_valid_mcd_package() {
325 let pdf = minimal_pdf("Hello from PDF");
326 let mcd = pdf_to_mcd_bytes(
327 &pdf,
328 PdfConversionOptions {
329 title: Some("PDF Import".to_owned()),
330 source_filename: Some("import.pdf".to_owned()),
331 },
332 )
333 .expect("pdf converts");
334 let package = crate::McdPackage::from_bytes(&mcd).expect("mcd opens");
335 crate::validate::validate_package(&package).expect("mcd validates");
336 let markdown = package
337 .read_to_string("content/main.md")
338 .expect("markdown exists");
339
340 assert!(markdown.contains("# PDF Import"));
341 assert!(markdown.contains("Hello from PDF"));
342 assert!(package.contains("assets/import.pdf"));
343 }
344
345 fn minimal_pdf(text: &str) -> Vec<u8> {
346 let escaped = text
347 .replace('\\', r"\\")
348 .replace('(', r"\(")
349 .replace(')', r"\)");
350 let content = format!("BT /F1 24 Tf 100 700 Td ({escaped}) Tj ET");
351 let objects = [
352 "<< /Type /Catalog /Pages 2 0 R >>".to_owned(),
353 "<< /Type /Pages /Kids [3 0 R] /Count 1 >>".to_owned(),
354 "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 4 0 R >> >> /Contents 5 0 R >>".to_owned(),
355 "<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>".to_owned(),
356 format!("<< /Length {} >>\nstream\n{}\nendstream", content.len(), content),
357 ];
358 let mut bytes = b"%PDF-1.4\n".to_vec();
359 let mut offsets = Vec::new();
360 for (index, object) in objects.iter().enumerate() {
361 offsets.push(bytes.len());
362 bytes
363 .extend_from_slice(format!("{} 0 obj\n{}\nendobj\n", index + 1, object).as_bytes());
364 }
365 let xref_offset = bytes.len();
366 bytes.extend_from_slice(format!("xref\n0 {}\n", objects.len() + 1).as_bytes());
367 bytes.extend_from_slice(b"0000000000 65535 f \n");
368 for offset in offsets {
369 bytes.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
370 }
371 bytes.extend_from_slice(
372 format!(
373 "trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF\n",
374 objects.len() + 1,
375 xref_offset
376 )
377 .as_bytes(),
378 );
379 bytes
380 }
381}