1use std::collections::HashMap;
17use std::sync::Arc;
18
19use lopdf::{Dictionary, Document, Object};
20
21use crate::pdfium_backend::Glyph;
22
23#[derive(Default)]
32struct DocCaches {
33 fonts: HashMap<(lopdf::ObjectId, Vec<u8>), Arc<Font>>,
34 forms: HashMap<lopdf::ObjectId, Arc<lopdf::content::Content>>,
35}
36
37#[derive(Clone, Copy)]
39struct Mat {
40 a: f64,
41 b: f64,
42 c: f64,
43 d: f64,
44 e: f64,
45 f: f64,
46}
47
48impl Mat {
49 const ID: Mat = Mat {
50 a: 1.0,
51 b: 0.0,
52 c: 0.0,
53 d: 1.0,
54 e: 0.0,
55 f: 0.0,
56 };
57
58 fn then(self, m: Mat) -> Mat {
60 Mat {
61 a: self.a * m.a + self.b * m.c,
62 b: self.a * m.b + self.b * m.d,
63 c: self.c * m.a + self.d * m.c,
64 d: self.c * m.b + self.d * m.d,
65 e: self.e * m.a + self.f * m.c + m.e,
66 f: self.e * m.b + self.f * m.d + m.f,
67 }
68 }
69
70 fn apply(self, x: f64, y: f64) -> (f64, f64) {
71 (
72 self.a * x + self.c * y + self.e,
73 self.b * x + self.d * y + self.f,
74 )
75 }
76}
77
78struct Font {
80 two_byte: bool,
82 to_unicode: HashMap<u32, String>,
84 widths: HashMap<u32, f64>,
86 default_width: f64,
87 simple_encoding: Option<HashMap<u8, char>>,
89 fallback_names: HashMap<u8, String>,
93 program_encoding: HashMap<u8, char>,
97 ascent: f64,
98 descent: f64,
99 hash: u64,
100}
101
102impl Font {
103 fn decode_code(&self, code: u32) -> (Option<String>, f64) {
104 let w = self
105 .widths
106 .get(&code)
107 .copied()
108 .unwrap_or(self.default_width);
109 if let Some(s) = self.to_unicode.get(&code) {
110 return (Some(decompose_ligatures(s)), w);
111 }
112 if !self.two_byte {
113 if let Some(name) = self.fallback_names.get(&(code as u8)) {
116 return (Some(format!("/{name}")), w);
117 }
118 if let Some(enc) = &self.simple_encoding {
119 if let Some(&ch) = enc.get(&(code as u8)) {
120 return (Some(decompose_ligatures(&ch.to_string())), w);
121 }
122 }
123 if let Some(&ch) = self.program_encoding.get(&(code as u8)) {
131 return (Some(decompose_ligatures(&ch.to_string())), w);
132 }
133 }
134 (None, w)
135 }
136}
137
138fn decompose_ligatures(s: &str) -> String {
142 if !s.chars().any(|c| ('\u{FB00}'..='\u{FB06}').contains(&c)) {
143 return s.to_string();
144 }
145 s.chars()
146 .map(|c| {
147 match c {
148 '\u{FB00}' => "ff",
149 '\u{FB01}' => "fi",
150 '\u{FB02}' => "fl",
151 '\u{FB03}' => "ffi",
152 '\u{FB04}' => "ffl",
153 '\u{FB05}' => "ft",
154 '\u{FB06}' => "st",
155 _ => return c.to_string(),
156 }
157 .to_string()
158 })
159 .collect()
160}
161
162fn hash_name(name: &[u8]) -> u64 {
163 use std::hash::{Hash, Hasher};
164 let mut h = std::collections::hash_map::DefaultHasher::new();
165 name.hash(&mut h);
166 h.finish()
167}
168
169fn as_dict<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Dictionary> {
171 match obj {
172 Object::Dictionary(d) => Some(d),
173 Object::Reference(id) => doc.get_object(*id).ok().and_then(|o| o.as_dict().ok()),
174 _ => None,
175 }
176}
177
178fn deref<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Object> {
179 match obj {
180 Object::Reference(id) => doc.get_object(*id).ok(),
181 other => Some(other),
182 }
183}
184
185fn parse_font(doc: &Document, name: &[u8], fdict: &Dictionary) -> Font {
187 let subtype: &[u8] = fdict
188 .get(b"Subtype")
189 .ok()
190 .and_then(|o| o.as_name().ok())
191 .unwrap_or(&[]);
192 let two_byte = subtype == b"Type0".as_slice();
193
194 let to_unicode = fdict
195 .get(b"ToUnicode")
196 .ok()
197 .and_then(|o| deref(doc, o))
198 .and_then(|o| o.as_stream().ok())
199 .and_then(|s| s.decompressed_content().ok())
200 .map(|data| parse_tounicode(&data))
201 .unwrap_or_default();
202
203 let (mut widths, mut default_width) = if two_byte {
204 cid_widths(doc, fdict)
205 } else {
206 simple_widths(doc, fdict)
207 };
208
209 let simple_encoding = if two_byte {
210 None
211 } else {
212 Some(simple_encoding_table(doc, fdict))
213 };
214
215 if !two_byte && widths.is_empty() && default_width == 0.0 {
223 if let Some(std14) = base_font_name(fdict).and_then(|n| crate::std14::widths_for(&n)) {
224 if let Some(enc) = &simple_encoding {
225 for (&code, &ch) in enc {
226 if let Some(w) = std14.width(ch) {
227 widths.insert(u32::from(code), w);
228 }
229 }
230 }
231 default_width = 500.0;
234 }
235 }
236 let fallback_names = if two_byte {
237 HashMap::new()
238 } else {
239 differences_gid_names(doc, fdict)
240 };
241 let program_encoding = if two_byte {
242 HashMap::new()
243 } else {
244 type1_program_encoding(doc, fdict)
245 };
246
247 let (ascent, descent) = font_ascent_descent(doc, fdict, two_byte);
248
249 Font {
250 two_byte,
251 to_unicode,
252 widths,
253 default_width,
254 simple_encoding,
255 fallback_names,
256 program_encoding,
257 ascent,
258 descent,
259 hash: hash_name(name),
260 }
261}
262
263fn differences_gid_names(doc: &Document, fdict: &Dictionary) -> HashMap<u8, String> {
270 let mut map = HashMap::new();
271 let Some(Object::Dictionary(enc)) = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o))
272 else {
273 return map;
274 };
275 let Some(Object::Array(diffs)) = enc.get(b"Differences").ok().and_then(|o| deref(doc, o))
276 else {
277 return map;
278 };
279 let mut code = 0u8;
280 for el in diffs {
281 match el {
282 Object::Integer(i) => code = *i as u8,
283 Object::Name(name) => {
284 if glyph_name_to_char(name).is_none() && is_gid_name(name) {
285 map.insert(code, String::from_utf8_lossy(name).into_owned());
286 }
287 code = code.wrapping_add(1);
288 }
289 _ => {}
290 }
291 }
292 map
293}
294
295fn type1_program_encoding(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
303 let mut map = HashMap::new();
304 let Some(desc) = fdict
305 .get(b"FontDescriptor")
306 .ok()
307 .and_then(|o| deref(doc, o))
308 .and_then(|o| o.as_dict().ok())
309 else {
310 return map;
311 };
312 let Some(data) = desc
313 .get(b"FontFile")
314 .ok()
315 .and_then(|o| deref(doc, o))
316 .and_then(|o| o.as_stream().ok())
317 .and_then(|s| s.decompressed_content().ok())
318 else {
319 return map;
320 };
321 let head_end = data
323 .windows(5)
324 .position(|w| w == b"eexec")
325 .unwrap_or(data.len());
326 let head = String::from_utf8_lossy(&data[..head_end]);
327 let toks: Vec<&str> = head.split_whitespace().collect();
329 for w in toks.windows(4) {
330 if w[0] == "dup" && w[3] == "put" {
331 if let (Ok(code), Some(name)) = (w[1].parse::<u32>(), w[2].strip_prefix('/')) {
332 if code <= 255 {
333 if let Some(ch) = glyph_name_to_char(name.as_bytes()) {
334 map.insert(code as u8, ch);
335 }
336 }
337 }
338 }
339 }
340 map
341}
342
343fn is_gid_name(name: &[u8]) -> bool {
349 let Ok(s) = std::str::from_utf8(name) else {
350 return false;
351 };
352 if s.starts_with("afii") || s.starts_with("uni") {
353 return false;
354 }
355 for prefix in ["g", "G", "cid", "CID", "glyph", "index"] {
356 if let Some(rest) = s.strip_prefix(prefix) {
357 if !rest.is_empty() && rest.bytes().all(|b| b.is_ascii_digit()) {
358 return true;
359 }
360 }
361 }
362 let alpha = s.bytes().take_while(|b| b.is_ascii_alphabetic()).count();
366 let digits = s.len() - alpha;
367 (1..=3).contains(&alpha)
368 && digits >= 3
369 && s.as_bytes()[alpha..].iter().all(|b| b.is_ascii_digit())
370}
371
372fn font_ascent_descent(doc: &Document, fdict: &Dictionary, two_byte: bool) -> (f64, f64) {
373 let descr_owner = if two_byte {
375 fdict
376 .get(b"DescendantFonts")
377 .ok()
378 .and_then(|o| deref(doc, o))
379 .and_then(|o| match o {
380 Object::Array(a) => a.first(),
381 _ => None,
382 })
383 .and_then(|o| as_dict(doc, o))
384 } else {
385 Some(fdict)
386 };
387 let fd = descr_owner
388 .and_then(|d| d.get(b"FontDescriptor").ok())
389 .and_then(|o| as_dict(doc, o));
390 let asc = fd
391 .and_then(|d| d.get(b"Ascent").ok())
392 .and_then(|o| {
393 o.as_float()
394 .ok()
395 .or_else(|| o.as_i64().ok().map(|i| i as f32))
396 })
397 .unwrap_or(750.0) as f64;
398 let desc = fd
399 .and_then(|d| d.get(b"Descent").ok())
400 .and_then(|o| {
401 o.as_float()
402 .ok()
403 .or_else(|| o.as_i64().ok().map(|i| i as f32))
404 })
405 .unwrap_or(-250.0) as f64;
406 if asc - desc <= 1.0 {
413 return (750.0, -250.0);
414 }
415 (asc, desc)
416}
417
418fn base_font_name(fdict: &Dictionary) -> Option<Vec<u8>> {
420 let name = fdict.get(b"BaseFont").ok()?.as_name().ok()?;
421 let stripped = match name.iter().position(|&b| b == b'+') {
422 Some(i) if i == 6 => &name[i + 1..],
423 _ => name,
424 };
425 Some(stripped.to_vec())
426}
427
428fn simple_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
430 let mut map = HashMap::new();
431 let first = fdict
432 .get(b"FirstChar")
433 .ok()
434 .and_then(|o| o.as_i64().ok())
435 .unwrap_or(0) as u32;
436 if let Some(Object::Array(arr)) = fdict.get(b"Widths").ok().and_then(|o| deref(doc, o)) {
437 for (i, w) in arr.iter().enumerate() {
438 if let Some(w) = num(w) {
439 map.insert(first + i as u32, w);
440 }
441 }
442 }
443 let dw = fdict
444 .get(b"FontDescriptor")
445 .ok()
446 .and_then(|o| as_dict(doc, o))
447 .and_then(|d| d.get(b"MissingWidth").ok())
448 .and_then(num)
449 .unwrap_or(0.0);
450 (map, dw)
451}
452
453fn cid_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
455 let mut map = HashMap::new();
456 let Some(desc) = fdict
457 .get(b"DescendantFonts")
458 .ok()
459 .and_then(|o| deref(doc, o))
460 .and_then(|o| match o {
461 Object::Array(a) => a.first(),
462 _ => None,
463 })
464 .and_then(|o| as_dict(doc, o))
465 else {
466 return (map, 1000.0);
467 };
468 let dw = desc.get(b"DW").ok().and_then(num).unwrap_or(1000.0);
469 if let Some(Object::Array(w)) = desc.get(b"W").ok().and_then(|o| deref(doc, o)) {
470 let mut i = 0;
471 while i < w.len() {
472 let c = w.get(i).and_then(num);
473 match (c, w.get(i + 1)) {
474 (Some(c), Some(Object::Array(list))) => {
476 for (k, wv) in list.iter().enumerate() {
477 if let Some(wv) = num(wv) {
478 map.insert(c as u32 + k as u32, wv);
479 }
480 }
481 i += 2;
482 }
483 (Some(c1), Some(o2)) => {
485 if let (Some(c2), Some(wv)) = (num(o2), w.get(i + 2).and_then(num)) {
486 for cid in c1 as u32..=c2 as u32 {
487 map.insert(cid, wv);
488 }
489 }
490 i += 3;
491 }
492 _ => break,
493 }
494 }
495 }
496 (map, dw)
497}
498
499fn num(o: &Object) -> Option<f64> {
500 match o {
501 Object::Integer(i) => Some(*i as f64),
502 Object::Real(r) => Some(*r as f64),
503 _ => None,
504 }
505}
506
507fn parse_tounicode(data: &[u8]) -> HashMap<u32, String> {
509 let text = String::from_utf8_lossy(data);
510 let mut map = HashMap::new();
511 let hex = |s: &str| -> Option<Vec<u16>> {
512 let s = s.trim();
513 if !s.starts_with('<') || !s.ends_with('>') {
514 return None;
515 }
516 let h = &s[1..s.len() - 1];
517 let bytes: Vec<u8> = (0..h.len())
518 .step_by(2)
519 .filter_map(|i| u8::from_str_radix(h.get(i..i + 2)?, 16).ok())
520 .collect();
521 Some(
522 bytes
523 .chunks(2)
524 .map(|c| {
525 if c.len() == 2 {
526 u16::from_be_bytes([c[0], c[1]])
527 } else {
528 c[0] as u16
529 }
530 })
531 .collect(),
532 )
533 };
534 let u16s_to_string = |u: &[u16]| String::from_utf16_lossy(u);
535 let code_of = |u: &[u16]| u.iter().fold(0u32, |acc, &x| (acc << 16) | x as u32);
536
537 let tokens: Vec<String> = {
541 let bytes = text.as_bytes();
542 let mut toks = Vec::new();
543 let mut i = 0;
544 while i < bytes.len() {
545 let c = bytes[i];
546 if c.is_ascii_whitespace() {
547 i += 1;
548 } else if c == b'<' {
549 let start = i;
550 while i < bytes.len() && bytes[i] != b'>' {
551 i += 1;
552 }
553 i += 1; toks.push(String::from_utf8_lossy(&bytes[start..i.min(bytes.len())]).into_owned());
555 } else if c == b'[' || c == b']' {
556 toks.push((c as char).to_string());
557 i += 1;
558 } else {
559 let start = i;
560 while i < bytes.len()
561 && !bytes[i].is_ascii_whitespace()
562 && bytes[i] != b'<'
563 && bytes[i] != b'['
564 && bytes[i] != b']'
565 {
566 i += 1;
567 }
568 toks.push(String::from_utf8_lossy(&bytes[start..i]).into_owned());
569 }
570 }
571 toks
572 };
573 let tokens: Vec<&str> = tokens.iter().map(|s| s.as_str()).collect();
574 let mut i = 0;
575 while i < tokens.len() {
576 match tokens[i] {
577 "beginbfchar" => {
578 i += 1;
579 while i + 1 < tokens.len() && tokens[i] != "endbfchar" {
580 if let (Some(src), Some(dst)) = (hex(tokens[i]), hex(tokens[i + 1])) {
581 map.insert(code_of(&src), u16s_to_string(&dst));
582 }
583 i += 2;
584 }
585 }
586 "beginbfrange" => {
587 i += 1;
588 while i + 2 < tokens.len() && tokens[i] != "endbfrange" {
589 let (Some(lo), Some(hi)) = (hex(tokens[i]), hex(tokens[i + 1])) else {
590 i += 1;
591 continue;
592 };
593 let lo = code_of(&lo);
594 let hi = code_of(&hi);
595 if tokens[i + 2] == "[" {
596 let mut j = i + 3;
598 let mut code = lo;
599 while j < tokens.len() && tokens[j] != "]" {
600 if let Some(dst) = hex(tokens[j]) {
601 map.insert(code, u16s_to_string(&dst));
602 }
603 code += 1;
604 j += 1;
605 }
606 i = j + 1;
607 } else if let Some(dst) = hex(tokens[i + 2]) {
608 let base = code_of(&dst);
610 for (k, code) in (lo..=hi).enumerate() {
611 if let Some(ch) = char::from_u32(base + k as u32) {
612 map.insert(code, ch.to_string());
613 }
614 }
615 i += 3;
616 } else {
617 i += 1;
618 }
619 }
620 }
621 _ => i += 1,
622 }
623 }
624 map
625}
626
627fn codes(font: &Font, bytes: &[u8]) -> Vec<u32> {
629 if font.two_byte {
630 bytes
631 .chunks(2)
632 .map(|c| {
633 if c.len() == 2 {
634 ((c[0] as u32) << 8) | c[1] as u32
635 } else {
636 c[0] as u32
637 }
638 })
639 .collect()
640 } else {
641 bytes.iter().map(|&b| b as u32).collect()
642 }
643}
644
645fn page_size(doc: &Document, page_id: lopdf::ObjectId) -> (f32, f32) {
647 let mb = doc
648 .get_object(page_id)
649 .ok()
650 .and_then(|o| o.as_dict().ok())
651 .and_then(|d| {
652 d.get(b"MediaBox").ok().cloned()
654 })
655 .or_else(|| {
656 doc.get_dictionary(page_id)
657 .ok()
658 .and_then(|d| d.get(b"MediaBox").ok().cloned())
659 });
660 if let Some(Object::Array(a)) = mb {
661 let v: Vec<f32> = a.iter().filter_map(|o| num(o).map(|x| x as f32)).collect();
662 if v.len() == 4 {
663 return ((v[2] - v[0]).abs(), (v[3] - v[1]).abs());
664 }
665 }
666 (612.0, 792.0)
667}
668
669pub fn content_diagnosis(bytes: &[u8]) -> String {
675 let Some(doc) = load_document(bytes) else {
676 return "document does not load".into();
677 };
678 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
679 pages.sort_by_key(|(n, _)| *n);
680 let mut out = String::new();
681 let mut caches = DocCaches::default();
682 for (n, pid) in pages.into_iter().take(4) {
683 let content_bytes = doc.get_page_content(pid);
684 let ops = lopdf::content::Content::decode(&content_bytes)
685 .map(|c| c.operations.len())
686 .ok();
687 let res = page_res(&doc, pid);
688 let fonts = res.map(|r| fonts_from_res(&doc, r, &mut caches).len());
689 let glyphs = page_glyphs_cached(&doc, pid, &mut caches).len();
690 out.push_str(&format!(
691 "\n page {n}: content {} B, ops {}, resources {}, fonts {}, glyphs {}",
692 content_bytes.len(),
693 ops.map_or("UNDECODABLE".to_string(), |n| n.to_string()),
694 if res.is_some() { "ok" } else { "MISSING" },
695 fonts.map_or("-".to_string(), |n| n.to_string()),
696 glyphs,
697 ));
698 }
699 out
700}
701
702pub fn text_layer_is_vestigial(pages: &[crate::pdfium_backend::PdfPage]) -> bool {
715 let lines: usize = pages.iter().map(|p| p.cells.len()).sum();
716 if lines == 0 {
717 return true;
718 }
719 let chars: usize = pages
720 .iter()
721 .flat_map(|p| &p.cells)
722 .map(|c| c.text.chars().count())
723 .sum();
724 lines <= pages.len() && chars < 32
725}
726
727pub fn xref_repair_status(bytes: &[u8]) -> String {
731 if Document::load_mem(bytes).is_ok() {
732 return "loads unaided; no repair needed".into();
733 }
734 match pad_short_xref_entries(bytes) {
735 Ok(fixed) => match Document::load_mem(&fixed) {
736 Ok(_) => "repaired: cross-reference entries padded to 20 bytes".into(),
737 Err(e) => format!("padded the entries, but it still will not load: {e}"),
738 },
739 Err(why) => format!("repair declined — {why}"),
740 }
741}
742
743fn load_document(bytes: &[u8]) -> Option<Document> {
758 let mut fallback = None;
763 if let Some(doc) = best_effort_load(bytes, &mut fallback) {
764 return Some(doc);
765 }
766 let xref_fixed = pad_short_xref_entries(bytes).ok();
767 if let Some(fixed) = &xref_fixed {
768 if let Some(doc) = best_effort_load(fixed, &mut fallback) {
769 return Some(doc);
770 }
771 }
772 let lengths_fixed = fix_stream_lengths(xref_fixed.as_deref().unwrap_or(bytes));
775 if let Some(doc) = best_effort_load(&lengths_fixed, &mut fallback) {
776 return Some(doc);
777 }
778 fallback
779}
780
781fn best_effort_load(data: &[u8], fallback: &mut Option<Document>) -> Option<Document> {
784 match Document::load_mem(data) {
785 Ok(doc) if has_page_content(&doc) => Some(doc),
786 Ok(doc) => {
787 fallback.get_or_insert(doc);
788 None
789 }
790 Err(_) => None,
791 }
792}
793
794fn has_page_content(doc: &Document) -> bool {
798 doc.get_pages()
799 .into_values()
800 .take(4)
801 .any(|pid| !doc.get_page_content(pid).is_empty())
802}
803
804fn fix_stream_lengths(bytes: &[u8]) -> Vec<u8> {
817 let mut out = bytes.to_vec();
818 let mut i = 0;
819 while let Some(rel) = find(&out[i..], b"stream") {
820 let kw = i + rel;
821 i = kw + 6;
822 if kw >= 3 && &out[kw - 3..kw] == b"end" {
824 continue;
825 }
826 let mut data = kw + 6;
828 if out.get(data..data + 2) == Some(b"\r\n".as_slice()) {
829 data += 2;
830 } else if matches!(out.get(data), Some(b'\n' | b'\r')) {
831 data += 1;
832 }
833 let Some(end) = find(&out[data..], b"endstream").map(|r| data + r) else {
834 continue;
835 };
836 let dict_start = out[..kw].iter().rposition(|&c| c == b'<').unwrap_or(0);
838 let Some(lrel) = find(&out[dict_start..kw], b"/Length") else {
839 continue;
840 };
841 let mut d = dict_start + lrel + 7;
842 while matches!(out.get(d), Some(b' ')) {
843 d += 1;
844 }
845 let digits = out[d..].iter().take_while(|c| c.is_ascii_digit()).count();
846 if digits == 0 {
847 continue;
848 }
849 let declared: usize = match std::str::from_utf8(&out[d..d + digits])
850 .ok()
851 .and_then(|s| s.parse().ok())
852 {
853 Some(v) => v,
854 None => continue,
855 };
856 let actual = end - data;
857 let replacement = actual.to_string();
860 if actual == declared || replacement.len() > digits {
861 continue;
862 }
863 out[d..d + digits].fill(b' ');
864 out[d..d + replacement.len()].copy_from_slice(replacement.as_bytes());
865 }
866 out
867}
868
869fn find(haystack: &[u8], needle: &[u8]) -> Option<usize> {
870 haystack.windows(needle.len()).position(|w| w == needle)
871}
872
873fn pad_short_xref_entries(bytes: &[u8]) -> Result<Vec<u8>, &'static str> {
876 let is_boundary = |i: usize| i == 0 || matches!(bytes[i - 1], b'\n' | b'\r');
879 let mut starts = (0..bytes.len().saturating_sub(4))
880 .filter(|&i| &bytes[i..i + 4] == b"xref" && is_boundary(i));
881 let xref_at = starts
882 .next()
883 .ok_or("no classic `xref` section (an xref stream?)")?;
884 if starts.next().is_some() {
885 return Err("more than one xref section (incremental update)");
886 }
887 let last_obj = bytes
888 .windows(3)
889 .rposition(|w| w == b"obj")
890 .ok_or("no objects found")?;
891 if last_obj > xref_at {
892 return Err("an object follows the xref — padding would move it");
893 }
894
895 let mut out = bytes[..xref_at].to_vec();
896 out.extend_from_slice(b"xref\n");
897 let mut i = xref_at + 4;
898 let skip_ws = |i: &mut usize| {
899 while matches!(bytes.get(*i), Some(b'\r' | b'\n' | b' ')) {
900 *i += 1;
901 }
902 };
903 loop {
904 skip_ws(&mut i);
905 if bytes[i..].starts_with(b"trailer") {
907 out.extend_from_slice(&bytes[i..]);
908 return Ok(out);
909 }
910 let header_end = i + bytes[i..]
911 .iter()
912 .position(|c| matches!(c, b'\n' | b'\r'))
913 .ok_or("subsection header runs off the end")?;
914 let header = std::str::from_utf8(&bytes[i..header_end])
915 .map_err(|_| "subsection header is not text")?
916 .trim();
917 let mut parts = header.split_whitespace();
918 let count: usize = parts
919 .nth(1)
920 .and_then(|c| c.parse().ok())
921 .ok_or("unparseable subsection header")?;
922 if parts.next().is_some() || count == 0 {
923 return Err("unexpected subsection header shape");
924 }
925 out.extend_from_slice(header.as_bytes());
926 out.push(b'\n');
927 i = header_end;
928 for _ in 0..count {
929 skip_ws(&mut i);
930 let entry = bytes.get(i..i + 18).ok_or("xref entry runs off the end")?;
932 let well_formed = entry[..10].iter().all(u8::is_ascii_digit)
933 && entry[10] == b' '
934 && entry[11..16].iter().all(u8::is_ascii_digit)
935 && entry[16] == b' '
936 && matches!(entry[17], b'n' | b'f');
937 if !well_formed {
938 return Err("xref entry is not `nnnnnnnnnn ggggg n`");
939 }
940 out.extend_from_slice(entry);
941 out.extend_from_slice(b" \n"); i += 18;
943 }
944 }
945}
946
947pub fn debug_glyphs(bytes: &[u8], index: usize) -> Vec<(char, f32, f32, f32, f32)> {
950 let Some(doc) = load_document(bytes) else {
951 return Vec::new();
952 };
953 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
954 pages.sort_by_key(|(n, _)| *n);
955 let Some((_, pid)) = pages.get(index) else {
956 return Vec::new();
957 };
958 page_glyphs(&doc, *pid)
959 .into_iter()
960 .map(|g| (g.ch, g.ll, g.lr, g.lb, g.lt))
961 .collect()
962}
963
964pub fn pdf_textlines(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
968 let Some(doc) = load_document(bytes) else {
969 return Vec::new();
970 };
971 let mut caches = DocCaches::default();
972 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
973 pages.sort_by_key(|(n, _)| *n);
974 pages
975 .into_iter()
976 .map(|(_, pid)| {
977 let (w, h) = page_size(&doc, pid);
978 let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
979 let cells = crate::dp_lines::line_cells(&glyphs, h, true);
980 (w, h, cells)
981 })
982 .collect()
983}
984
985pub fn pdf_words(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
990 let Some(doc) = load_document(bytes) else {
991 return Vec::new();
992 };
993 let mut caches = DocCaches::default();
994 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
995 pages.sort_by_key(|(n, _)| *n);
996 pages
997 .into_iter()
998 .map(|(_, pid)| {
999 let (w, h) = page_size(&doc, pid);
1000 let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1001 let cells = crate::dp_lines::word_cells(&glyphs, h, true);
1002 (w, h, cells)
1003 })
1004 .collect()
1005}
1006
1007#[derive(Default)]
1011pub struct PageParserCells {
1012 pub prose: Vec<crate::pdfium_backend::TextCell>,
1013 pub words: Vec<crate::pdfium_backend::TextCell>,
1014 pub code: Vec<crate::pdfium_backend::TextCell>,
1015}
1016
1017pub struct PageTextParser {
1030 doc: Document,
1031 caches: DocCaches,
1032 pages: Vec<lopdf::ObjectId>,
1034}
1035
1036impl PageTextParser {
1037 pub fn open(bytes: &[u8]) -> Option<Self> {
1040 let doc = load_document(bytes)?;
1041 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1042 pages.sort_by_key(|(n, _)| *n);
1043 Some(Self {
1044 doc,
1045 caches: DocCaches::default(),
1046 pages: pages.into_iter().map(|(_, pid)| pid).collect(),
1047 })
1048 }
1049
1050 pub fn cells(&mut self, index: usize) -> PageParserCells {
1054 let Some(&pid) = self.pages.get(index) else {
1055 return PageParserCells::default();
1056 };
1057 let (_w, h) = page_size(&self.doc, pid);
1058 let glyphs = page_glyphs_cached(&self.doc, pid, &mut self.caches);
1059 let (prose, words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1060 PageParserCells {
1061 prose,
1062 words,
1063 code: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1064 }
1065 }
1066}
1067
1068pub fn pdf_all_cells(bytes: &[u8]) -> Vec<PageParserCells> {
1073 let Some(mut parser) = PageTextParser::open(bytes) else {
1074 return Vec::new();
1075 };
1076 (0..parser.pages.len()).map(|i| parser.cells(i)).collect()
1077}
1078
1079pub fn pdf_text_pages(bytes: &[u8]) -> Vec<crate::pdfium_backend::PdfPage> {
1086 let Some(doc) = load_document(bytes) else {
1087 return Vec::new();
1088 };
1089 let mut caches = DocCaches::default();
1090 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1091 pages.sort_by_key(|(n, _)| *n);
1092 pages
1093 .into_iter()
1094 .map(|(_, pid)| {
1095 let (w, h) = page_size(&doc, pid);
1096 let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1097 let (mut prose, mut words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1098 drop_overpainted_cells(&mut prose);
1099 drop_overpainted_cells(&mut words);
1100 crate::pdfium_backend::PdfPage {
1101 #[cfg(feature = "ocr-prep")]
1102 image_layout: None,
1103 width: w,
1104 height: h,
1105 scale: 1.0,
1107 cells: prose,
1108 code_cells: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1109 word_cells: words,
1110 #[cfg(feature = "ocr-prep")]
1111 image: image::RgbImage::new(1, 1),
1112 links: Vec::new(),
1113 rotation: 0,
1114 }
1115 })
1116 .collect()
1117}
1118
1119fn drop_overpainted_cells(cells: &mut Vec<crate::pdfium_backend::TextCell>) {
1138 let mut paint = vec![false; cells.len()];
1139 for i in 0..cells.len() {
1140 for j in 0..cells.len() {
1141 if i == j || cells[i].text == cells[j].text {
1142 continue;
1143 }
1144 let (a, b) = (&cells[i], &cells[j]);
1145 let vo = (a.b.min(b.b) - a.t.max(b.t)).max(0.0);
1147 if vo < 0.6 * (a.b - a.t).min(b.b - b.t) {
1148 continue;
1149 }
1150 let ho = (a.r.min(b.r) - a.l.max(b.l)).max(0.0);
1152 if ho >= 0.8 * (a.r - a.l) && (a.r - a.l) <= (b.r - b.l) {
1153 paint[i] = true;
1154 paint[j] = true;
1155 }
1156 }
1157 }
1158 let mut keep = paint.iter().map(|p| !p);
1159 cells.retain(|_| keep.next().unwrap());
1160}
1161
1162#[derive(Clone, Copy)]
1166struct TextState {
1167 tc: f64,
1168 tw: f64,
1169 th: f64,
1170 tl: f64,
1171 trise: f64,
1172 fsize: f64,
1173}
1174
1175impl TextState {
1176 const INIT: TextState = TextState {
1177 tc: 0.0,
1178 tw: 0.0,
1179 th: 1.0,
1180 tl: 0.0,
1181 trise: 0.0,
1182 fsize: 0.0,
1183 };
1184}
1185
1186fn page_res(doc: &Document, page_id: lopdf::ObjectId) -> Option<&Dictionary> {
1189 let (inline, ids) = doc.get_page_resources(page_id).ok()?;
1190 if let Some(d) = inline {
1191 return Some(d);
1192 }
1193 ids.into_iter().find_map(|id| doc.get_dictionary(id).ok())
1194}
1195
1196fn fonts_from_res(
1200 doc: &Document,
1201 res: &Dictionary,
1202 caches: &mut DocCaches,
1203) -> HashMap<Vec<u8>, Arc<Font>> {
1204 let mut map = HashMap::new();
1205 let font_dict = res
1206 .get(b"Font")
1207 .ok()
1208 .and_then(|o| deref(doc, o))
1209 .and_then(|o| o.as_dict().ok());
1210 if let Some(fd) = font_dict {
1211 for (name, value) in fd.iter() {
1212 let font = match value {
1213 Object::Reference(id) => {
1214 let key = (*id, name.clone());
1215 if let Some(f) = caches.fonts.get(&key) {
1216 Arc::clone(f)
1217 } else if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1218 let f = Arc::new(parse_font(doc, name, fdict));
1219 caches.fonts.insert(key, Arc::clone(&f));
1220 f
1221 } else {
1222 continue;
1223 }
1224 }
1225 _ => {
1226 if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1227 Arc::new(parse_font(doc, name, fdict))
1228 } else {
1229 continue;
1230 }
1231 }
1232 };
1233 map.insert(name.clone(), font);
1234 }
1235 }
1236 map
1237}
1238
1239pub(crate) fn page_glyphs(doc: &Document, page_id: lopdf::ObjectId) -> Vec<Glyph> {
1241 page_glyphs_cached(doc, page_id, &mut DocCaches::default())
1242}
1243
1244fn page_glyphs_cached(
1247 doc: &Document,
1248 page_id: lopdf::ObjectId,
1249 caches: &mut DocCaches,
1250) -> Vec<Glyph> {
1251 let mut out = Vec::new();
1252 let content_bytes = doc.get_page_content(page_id);
1255 let Ok(content) = lopdf::content::Content::decode(&content_bytes) else {
1256 return out;
1257 };
1258 if let Some(res) = page_res(doc, page_id) {
1259 run_content(
1260 doc,
1261 res,
1262 &content,
1263 Mat::ID,
1264 TextState::INIT,
1265 0,
1266 caches,
1267 &mut out,
1268 );
1269 }
1270 out
1271}
1272
1273#[allow(clippy::too_many_arguments)]
1278fn run_content(
1279 doc: &Document,
1280 res: &Dictionary,
1281 content: &lopdf::content::Content,
1282 base_ctm: Mat,
1283 init: TextState,
1284 depth: u32,
1285 caches: &mut DocCaches,
1286 out: &mut Vec<Glyph>,
1287) {
1288 let fonts = fonts_from_res(doc, res, caches);
1289 let xobjects = res
1290 .get(b"XObject")
1291 .ok()
1292 .and_then(|o| deref(doc, o))
1293 .and_then(|o| o.as_dict().ok());
1294
1295 #[allow(clippy::type_complexity)]
1300 let mut gstate_stack: Vec<(Mat, f64, f64, f64, f64, f64, f64, Option<&Arc<Font>>)> = Vec::new();
1301 let mut ctm = base_ctm;
1302 let mut tm = Mat::ID;
1303 let mut tlm = Mat::ID;
1304 let mut font: Option<&Arc<Font>> = None;
1305 let mut fsize = init.fsize;
1306 let mut tc = init.tc; let mut tw = init.tw; let mut th = init.th; let mut tl = init.tl; let mut trise = init.trise;
1311
1312 let op_f = |operands: &[Object], i: usize| operands.get(i).and_then(num).unwrap_or(0.0);
1313
1314 for op in &content.operations {
1315 let operands = &op.operands;
1316 match op.operator.as_str() {
1317 "q" => gstate_stack.push((ctm, tc, tw, th, tl, trise, fsize, font)),
1318 "Q" => {
1319 if let Some((c, a, b, h, l, r, fs, f)) = gstate_stack.pop() {
1320 ctm = c;
1321 tc = a;
1322 tw = b;
1323 th = h;
1324 tl = l;
1325 trise = r;
1326 fsize = fs;
1327 font = f;
1328 }
1329 }
1330 "cm" => {
1331 let m = Mat {
1332 a: op_f(operands, 0),
1333 b: op_f(operands, 1),
1334 c: op_f(operands, 2),
1335 d: op_f(operands, 3),
1336 e: op_f(operands, 4),
1337 f: op_f(operands, 5),
1338 };
1339 ctm = m.then(ctm);
1340 }
1341 "BT" => {
1342 tm = Mat::ID;
1343 tlm = Mat::ID;
1344 }
1345 "ET" => {}
1346 "Tf" => {
1347 if let Some(Object::Name(n)) = operands.first() {
1348 font = fonts.get(n.as_slice());
1349 }
1350 fsize = op_f(operands, 1);
1351 }
1352 "Td" => {
1353 tlm = Mat {
1354 a: 1.0,
1355 b: 0.0,
1356 c: 0.0,
1357 d: 1.0,
1358 e: op_f(operands, 0),
1359 f: op_f(operands, 1),
1360 }
1361 .then(tlm);
1362 tm = tlm;
1363 }
1364 "TD" => {
1365 tl = -op_f(operands, 1);
1366 tlm = Mat {
1367 a: 1.0,
1368 b: 0.0,
1369 c: 0.0,
1370 d: 1.0,
1371 e: op_f(operands, 0),
1372 f: op_f(operands, 1),
1373 }
1374 .then(tlm);
1375 tm = tlm;
1376 }
1377 "Tm" => {
1378 tlm = Mat {
1379 a: op_f(operands, 0),
1380 b: op_f(operands, 1),
1381 c: op_f(operands, 2),
1382 d: op_f(operands, 3),
1383 e: op_f(operands, 4),
1384 f: op_f(operands, 5),
1385 };
1386 tm = tlm;
1387 }
1388 "T*" => {
1389 tlm = Mat {
1390 a: 1.0,
1391 b: 0.0,
1392 c: 0.0,
1393 d: 1.0,
1394 e: 0.0,
1395 f: -tl,
1396 }
1397 .then(tlm);
1398 tm = tlm;
1399 }
1400 "Tc" => tc = op_f(operands, 0),
1401 "Tw" => tw = op_f(operands, 0),
1402 "Tz" => th = op_f(operands, 0) / 100.0,
1403 "TL" => tl = op_f(operands, 0),
1404 "Ts" => trise = op_f(operands, 0),
1405 "Tj" | "'" | "\"" => {
1406 if op.operator == "'" || op.operator == "\"" {
1407 tlm = Mat {
1409 a: 1.0,
1410 b: 0.0,
1411 c: 0.0,
1412 d: 1.0,
1413 e: 0.0,
1414 f: -tl,
1415 }
1416 .then(tlm);
1417 tm = tlm;
1418 }
1419 if op.operator == "\"" {
1420 tw = op_f(operands, 0);
1423 tc = op_f(operands, 1);
1424 }
1425 if let (Some(f), Some(Object::String(s, _))) = (font, operands.last()) {
1426 show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out);
1427 }
1428 }
1429 "TJ" => {
1430 if let (Some(f), Some(Object::Array(arr))) = (font, operands.first()) {
1431 for el in arr {
1432 match el {
1433 Object::String(s, _) => {
1434 show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out)
1435 }
1436 other => {
1437 if let Some(adj) = num(other) {
1438 let tx = -adj / 1000.0 * fsize * th;
1440 tm = Mat {
1441 a: 1.0,
1442 b: 0.0,
1443 c: 0.0,
1444 d: 1.0,
1445 e: tx,
1446 f: 0.0,
1447 }
1448 .then(tm);
1449 }
1450 }
1451 }
1452 }
1453 }
1454 }
1455 "Do" => {
1456 if depth >= 8 {
1459 continue;
1460 }
1461 let Some(Object::Name(n)) = operands.first() else {
1462 continue;
1463 };
1464 let obj = xobjects.and_then(|d| d.get(n.as_slice()).ok());
1465 let form_id = match obj {
1466 Some(Object::Reference(id)) => Some(*id),
1467 _ => None,
1468 };
1469 let stream = obj
1470 .and_then(|o| deref(doc, o))
1471 .and_then(|o| o.as_stream().ok());
1472 let Some(stream) = stream else { continue };
1473 let is_form = stream
1474 .dict
1475 .get(b"Subtype")
1476 .ok()
1477 .and_then(|o| o.as_name().ok())
1478 == Some(b"Form".as_slice());
1479 if !is_form {
1480 continue;
1481 }
1482 let cached = form_id.and_then(|id| caches.forms.get(&id).cloned());
1485 let form_content = match cached {
1486 Some(c) => c,
1487 None => {
1488 let Ok(data) = stream.decompressed_content() else {
1489 continue;
1490 };
1491 let Ok(c) = lopdf::content::Content::decode(&data) else {
1492 continue;
1493 };
1494 let c = Arc::new(c);
1495 if let Some(id) = form_id {
1496 caches.forms.insert(id, Arc::clone(&c));
1497 }
1498 c
1499 }
1500 };
1501 let form_mat = match stream.dict.get(b"Matrix").ok() {
1503 Some(Object::Array(a)) if a.len() == 6 => {
1504 let v: Vec<f64> = a.iter().filter_map(num).collect();
1505 if v.len() == 6 {
1506 Mat {
1507 a: v[0],
1508 b: v[1],
1509 c: v[2],
1510 d: v[3],
1511 e: v[4],
1512 f: v[5],
1513 }
1514 } else {
1515 Mat::ID
1516 }
1517 }
1518 _ => Mat::ID,
1519 };
1520 let form_res = stream
1522 .dict
1523 .get(b"Resources")
1524 .ok()
1525 .and_then(|o| deref(doc, o))
1526 .and_then(|o| o.as_dict().ok())
1527 .unwrap_or(res);
1528 let state = TextState {
1529 tc,
1530 tw,
1531 th,
1532 tl,
1533 trise,
1534 fsize,
1535 };
1536 run_content(
1537 doc,
1538 form_res,
1539 &form_content,
1540 form_mat.then(ctm),
1541 state,
1542 depth + 1,
1543 caches,
1544 out,
1545 );
1546 }
1547 _ => {}
1548 }
1549 }
1550}
1551
1552#[allow(clippy::too_many_arguments)]
1553fn show_text(
1554 font: &Font,
1555 bytes: &[u8],
1556 fsize: f64,
1557 tc: f64,
1558 tw: f64,
1559 th: f64,
1560 trise: f64,
1561 tm: &mut Mat,
1562 ctm: Mat,
1563 out: &mut Vec<Glyph>,
1564) {
1565 for code in codes(font, bytes) {
1566 let (text, w) = font.decode_code(code);
1567 let w0 = w / 1000.0; let scale = Mat {
1570 a: fsize * th,
1571 b: 0.0,
1572 c: 0.0,
1573 d: fsize,
1574 e: 0.0,
1575 f: trise,
1576 };
1577 let trm = scale.then(*tm).then(ctm);
1578 let (x0, y0) = trm.apply(0.0, font.descent / 1000.0);
1580 let (x1, _y1) = trm.apply(w0, font.descent / 1000.0);
1581 let (_x2, y2) = trm.apply(0.0, font.ascent / 1000.0);
1582 let (left, right) = (x0.min(x1), x0.max(x1));
1583 let (bot, top) = (y0.min(y2), y0.max(y2));
1584 if let Some(s) = text {
1585 for ch in s.chars() {
1587 if ch != '\u{0}' {
1588 out.push(Glyph {
1589 ch,
1590 l: left as f32,
1591 b: bot as f32,
1592 r: right as f32,
1593 t: top as f32,
1594 ll: left as f32,
1595 lb: bot as f32,
1596 lr: right as f32,
1597 lt: top as f32,
1598 font: font.hash,
1599 });
1600 }
1601 }
1602 }
1603 let is_space = !font.two_byte && code == 32;
1605 let tx = (w0 * fsize + tc + if is_space { tw } else { 0.0 }) * th;
1606 *tm = Mat {
1607 a: 1.0,
1608 b: 0.0,
1609 c: 0.0,
1610 d: 1.0,
1611 e: tx,
1612 f: 0.0,
1613 }
1614 .then(*tm);
1615 }
1616}
1617
1618fn simple_encoding_table(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
1622 let enc = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o));
1623 let base_name = match enc {
1624 Some(Object::Name(n)) => n.clone(),
1625 Some(Object::Dictionary(d)) => d
1626 .get(b"BaseEncoding")
1627 .ok()
1628 .and_then(|o| o.as_name().ok())
1629 .map(|n| n.to_vec())
1630 .unwrap_or_default(),
1631 _ => Vec::new(),
1632 };
1633 let mut m = if base_name == b"MacRomanEncoding" {
1634 macroman_table()
1635 } else if base_name.is_empty() {
1636 tex_math_builtin(fdict).unwrap_or_else(winansi_table)
1643 } else {
1644 winansi_table()
1645 };
1646 if let Some(Object::Dictionary(d)) = enc {
1648 if let Some(Object::Array(diffs)) = d.get(b"Differences").ok().and_then(|o| deref(doc, o)) {
1649 let mut code = 0u8;
1650 for el in diffs {
1651 match el {
1652 Object::Integer(i) => code = *i as u8,
1653 Object::Name(name) => {
1654 if let Some(ch) = glyph_name_to_char(name) {
1655 m.insert(code, ch);
1656 }
1657 code = code.wrapping_add(1);
1658 }
1659 _ => {}
1660 }
1661 }
1662 }
1663 }
1664 m
1665}
1666
1667fn tex_math_builtin(fdict: &Dictionary) -> Option<HashMap<u8, char>> {
1673 const CMSY: [char; 128] = [
1674 '−', '·', '×', '∗', '÷', '⋄', '±', '∓', '⊕', '⊖', '⊗', '⊘', '⊙', '◯', '∘', '•', '≍', '≡',
1675 '⊆', '⊇', '≤', '≥', '≼', '≽', '∼', '≈', '⊂', '⊃', '≪', '≫', '≺', '≻', '←', '→', '↑', '↓',
1676 '↔', '↗', '↘', '≃', '⇐', '⇒', '⇑', '⇓', '⇔', '↖', '↙', '∝', '′', '∞', '∈', '∋', '△', '▽',
1677 '\u{338}', '↦', '∀', '∃', '¬', '∅', 'ℜ', 'ℑ', '⊤', '⊥', 'ℵ', 'A', 'B', 'C', 'D', 'E', 'F',
1678 'G', 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X',
1679 'Y', 'Z', '∪', '∩', '⊎', '∧', '∨', '⊢', '⊣', '⌊', '⌋', '⌈', '⌉', '{', '}', '⟨', '⟩', '|',
1680 '∥', '↕', '⇕', '\\', '≀', '√', '∐', '∇', '∫', '⊔', '⊓', '⊑', '⊒', '§', '†', '‡', '¶', '♣',
1681 '♢', '♡', '♠',
1682 ];
1683 const CMMI: [char; 128] = [
1684 'Γ', 'Δ', 'Θ', 'Λ', 'Ξ', 'Π', 'Σ', 'Υ', 'Φ', 'Ψ', 'Ω', 'α', 'β', 'γ', 'δ', 'ε', 'ζ', 'η',
1685 'θ', 'ι', 'κ', 'λ', 'μ', 'ν', 'ξ', 'π', 'ρ', 'σ', 'τ', 'υ', 'φ', 'χ', 'ψ', 'ω', 'ϵ', 'ϑ',
1686 'ϖ', 'ϱ', 'ς', 'ϕ', '↼', '↽', '⇀', '⇁', '↩', '↪', '▷', '◁', '0', '1', '2', '3', '4', '5',
1687 '6', '7', '8', '9', '.', ',', '<', '/', '>', '⋆', '∂', 'A', 'B', 'C', 'D', 'E', 'F', 'G',
1688 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y',
1689 'Z', '♭', '♮', '♯', '⌣', '⌢', 'ℓ', 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k',
1690 'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', 'ı', 'ȷ', '℘',
1691 '\u{20d7}', '⁀',
1692 ];
1693 let name = base_font_name(fdict)?;
1694 let up = name.to_ascii_uppercase();
1695 let table: &[char; 128] = if up.starts_with(b"CMSY") || up.starts_with(b"CMBSY") {
1696 &CMSY
1697 } else if up.starts_with(b"CMMI") {
1698 &CMMI
1699 } else {
1700 return None;
1701 };
1702 Some(
1703 table
1704 .iter()
1705 .enumerate()
1706 .map(|(i, &c)| (i as u8, c))
1707 .collect(),
1708 )
1709}
1710
1711fn glyph_name_to_char(name: &[u8]) -> Option<char> {
1716 let s = std::str::from_utf8(name).ok()?;
1717 if let Some(hex) = s.strip_prefix("uni") {
1718 if let Ok(cp) = u32::from_str_radix(hex.get(0..4)?, 16) {
1719 return char::from_u32(cp);
1720 }
1721 }
1722 if s.len() == 1 {
1724 let b = s.as_bytes()[0];
1725 if b.is_ascii_alphabetic() {
1726 return Some(b as char);
1727 }
1728 }
1729 let resolved = match s {
1730 "space" => ' ',
1731 "exclam" => '!',
1732 "quotedbl" => '"',
1733 "numbersign" => '#',
1734 "dollar" => '$',
1735 "percent" => '%',
1736 "ampersand" => '&',
1737 "quotesingle" => '\'',
1738 "parenleft" => '(',
1739 "parenright" => ')',
1740 "asterisk" => '*',
1741 "plus" => '+',
1742 "comma" => ',',
1743 "hyphen" => '-',
1744 "period" => '.',
1745 "slash" => '/',
1746 "zero" => '0',
1747 "one" => '1',
1748 "two" => '2',
1749 "three" => '3',
1750 "four" => '4',
1751 "five" => '5',
1752 "six" => '6',
1753 "seven" => '7',
1754 "eight" => '8',
1755 "nine" => '9',
1756 "colon" => ':',
1757 "semicolon" => ';',
1758 "less" => '<',
1759 "equal" => '=',
1760 "greater" => '>',
1761 "question" => '?',
1762 "at" => '@',
1763 "bracketleft" => '[',
1764 "backslash" => '\\',
1765 "bracketright" => ']',
1766 "asciicircum" => '^',
1767 "underscore" => '_',
1768 "grave" => '`',
1769 "braceleft" => '{',
1770 "bar" => '|',
1771 "braceright" => '}',
1772 "asciitilde" => '~',
1773 "bullet" => '\u{2022}',
1774 "periodcentered" => '\u{00B7}',
1775 "endash" => '\u{2013}',
1776 "emdash" => '\u{2014}',
1777 "quoteright" => '\u{2019}',
1778 "quoteleft" => '\u{2018}',
1779 "quotedblleft" => '\u{201C}',
1780 "quotedblright" => '\u{201D}',
1781 "quotedblbase" => '\u{201E}',
1782 "quotesinglbase" => '\u{201A}',
1783 "ff" => '\u{FB00}',
1788 "fi" => '\u{FB01}',
1789 "fl" => '\u{FB02}',
1790 "ffi" => '\u{FB03}',
1791 "ffl" => '\u{FB04}',
1792 "ft" => '\u{FB05}',
1793 "st" => '\u{FB06}',
1794 "degree" => '\u{00B0}',
1795 "trademark" => '\u{2122}',
1796 "registered" => '\u{00AE}',
1797 "copyright" => '\u{00A9}',
1798 "ellipsis" => '\u{2026}',
1799 "minus" => '\u{2212}',
1800 "fraction" => '\u{2044}',
1801 "nbspace" => '\u{00A0}',
1802 "alpha" => '\u{03B1}',
1807 "beta" => '\u{03B2}',
1808 "gamma" => '\u{03B3}',
1809 "delta" => '\u{03B4}',
1810 "epsilon" | "epsilon1" => '\u{03B5}',
1811 "zeta" => '\u{03B6}',
1812 "eta" => '\u{03B7}',
1813 "theta" | "theta1" => '\u{03B8}',
1814 "iota" => '\u{03B9}',
1815 "kappa" => '\u{03BA}',
1816 "lambda" => '\u{03BB}',
1817 "mu" => '\u{03BC}',
1818 "nu" => '\u{03BD}',
1819 "xi" => '\u{03BE}',
1820 "omicron" => '\u{03BF}',
1821 "pi" | "pi1" => '\u{03C0}',
1822 "rho" | "rho1" => '\u{03C1}',
1823 "sigma" => '\u{03C3}',
1824 "sigma1" => '\u{03C2}',
1825 "tau" => '\u{03C4}',
1826 "upsilon" => '\u{03C5}',
1827 "phi" | "phi1" => '\u{03C6}',
1828 "chi" => '\u{03C7}',
1829 "psi" => '\u{03C8}',
1830 "omega" | "omega1" => '\u{03C9}',
1831 "Gamma" => '\u{0393}',
1832 "Delta" => '\u{0394}',
1833 "Theta" => '\u{0398}',
1834 "Lambda" => '\u{039B}',
1835 "Xi" => '\u{039E}',
1836 "Pi" => '\u{03A0}',
1837 "Sigma" => '\u{03A3}',
1838 "Upsilon" => '\u{03A5}',
1839 "Phi" => '\u{03A6}',
1840 "Psi" => '\u{03A8}',
1841 "Omega" => '\u{03A9}',
1842 "lessequal" => '\u{2264}',
1843 "greaterequal" => '\u{2265}',
1844 "notequal" => '\u{2260}',
1845 "approxequal" => '\u{2248}',
1846 "equivalence" => '\u{2261}',
1847 "element" => '\u{2208}',
1848 "plusminus" => '\u{00B1}',
1849 "multiply" => '\u{00D7}',
1850 "divide" => '\u{00F7}',
1851 "infinity" => '\u{221E}',
1852 "partialdiff" => '\u{2202}',
1853 "gradient" => '\u{2207}',
1854 "summation" => '\u{2211}',
1855 "product" => '\u{220F}',
1856 "integral" => '\u{222B}',
1857 "radical" => '\u{221A}',
1858 "proportional" => '\u{221D}',
1859 "arrowright" => '\u{2192}',
1860 "arrowleft" => '\u{2190}',
1861 "arrowup" => '\u{2191}',
1862 "arrowdown" => '\u{2193}',
1863 "arrowboth" => '\u{2194}',
1864 "arrowdblright" => '\u{21D2}',
1865 "logicaland" => '\u{2227}',
1866 "logicalor" => '\u{2228}',
1867 "intersection" => '\u{2229}',
1868 "union" => '\u{222A}',
1869 "similar" => '\u{223C}',
1870 "congruent" => '\u{2245}',
1871 "dotmath" => '\u{22C5}',
1872 "asteriskmath" => '\u{2217}',
1873 _ => {
1874 if let Some((base, _)) = s.split_once('.') {
1876 if !base.is_empty() {
1877 return glyph_name_to_char(base.as_bytes());
1878 }
1879 }
1880 return None;
1881 }
1882 };
1883 Some(resolved)
1884}
1885
1886fn winansi_table() -> HashMap<u8, char> {
1888 let mut m = HashMap::new();
1889 for b in 0x20u8..=0x7e {
1890 m.insert(b, b as char);
1891 }
1892 let extra: &[(u8, char)] = &[
1894 (0x91, '\u{2018}'),
1895 (0x92, '\u{2019}'),
1896 (0x93, '\u{201C}'),
1897 (0x94, '\u{201D}'),
1898 (0x95, '\u{2022}'),
1899 (0x96, '\u{2013}'),
1900 (0x97, '\u{2014}'),
1901 (0x85, '\u{2026}'),
1902 (0xA0, '\u{00A0}'),
1903 ];
1904 for &(b, c) in extra {
1905 m.insert(b, c);
1906 }
1907 for b in 0xA1u8..=0xFF {
1908 m.entry(b).or_insert(b as char);
1909 }
1910 m
1911}
1912
1913fn macroman_table() -> HashMap<u8, char> {
1916 let mut m = HashMap::new();
1917 for b in 0x20u8..=0x7e {
1918 m.insert(b, b as char);
1919 }
1920 let high: &[(u8, char)] = &[
1921 (0xA5, '\u{2022}'), (0xD0, '\u{2013}'), (0xD1, '\u{2014}'), (0xD2, '\u{201C}'),
1925 (0xD3, '\u{201D}'),
1926 (0xD4, '\u{2018}'),
1927 (0xD5, '\u{2019}'),
1928 (0xCA, '\u{00A0}'),
1929 (0xC9, '\u{2026}'),
1930 (0xDE, '\u{FB01}'),
1931 (0xDF, '\u{FB02}'),
1932 ];
1933 for &(b, c) in high {
1934 m.insert(b, c);
1935 }
1936 m
1937}
1938
1939#[cfg(test)]
1940mod xref_repair {
1941 fn pdf_with_xref(two_byte_eol: bool) -> Vec<u8> {
1945 let content = b"BT /F1 12 Tf 72 700 Td (Invoice 922769430725) Tj ET\n";
1946 let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
1947 let objs: Vec<Vec<u8>> = vec![
1948 b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
1949 b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
1950 b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
1951 /Resources<</Font<</F1 5 0 R>>>>>>"
1952 .to_vec(),
1953 [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
1954 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".to_vec(),
1955 ];
1956
1957 let mut out = b"%PDF-1.4\n".to_vec();
1958 let mut offsets = Vec::new();
1959 for (i, body) in objs.iter().enumerate() {
1960 offsets.push(out.len());
1961 out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
1962 out.extend_from_slice(body);
1963 out.extend_from_slice(b"endobj\n");
1964 }
1965 let xref_at = out.len();
1966 let eol: &[u8] = if two_byte_eol { b" \n" } else { b"\n" };
1967 out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
1968 out.extend_from_slice(b"0000000000 65535 f");
1969 out.extend_from_slice(eol);
1970 for off in &offsets {
1971 out.extend_from_slice(format!("{off:010} 00000 n").as_bytes());
1972 out.extend_from_slice(eol);
1973 }
1974 out.extend_from_slice(
1975 format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
1976 );
1977 out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
1978 out
1979 }
1980
1981 #[test]
1987 fn short_xref_entries_still_parse() {
1988 let good = pdf_with_xref(true);
1989 let broken = pdf_with_xref(false);
1990 assert!(
1991 broken.len() < good.len(),
1992 "the broken file is the shorter one"
1993 );
1994 assert!(
1995 lopdf::Document::load_mem(&good).is_ok(),
1996 "the control file must load unaided"
1997 );
1998 assert!(
1999 lopdf::Document::load_mem(&broken).is_err(),
2000 "lopdf rejects 19-byte entries — if this ever passes, drop the repair"
2001 );
2002
2003 let cells = |b: &[u8]| -> Vec<String> {
2004 super::pdf_textlines(b)
2005 .into_iter()
2006 .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2007 .collect()
2008 };
2009 let from_good = cells(&good);
2010 assert!(
2011 from_good.iter().any(|t| t.contains("922769430725")),
2012 "control text: {from_good:?}"
2013 );
2014 assert_eq!(
2015 cells(&broken),
2016 from_good,
2017 "repair must match the good parse"
2018 );
2019 }
2020
2021 #[test]
2026 fn overstated_stream_length_still_yields_content() {
2027 let good = pdf_with_xref(true);
2028 let broken = {
2031 let at = good
2032 .windows(8)
2033 .position(|w| w == b"/Length ")
2034 .expect("a /Length")
2035 + 8;
2036 let digits = good[at..].iter().take_while(|c| c.is_ascii_digit()).count();
2037 let n: usize = std::str::from_utf8(&good[at..at + digits])
2038 .unwrap()
2039 .parse()
2040 .unwrap();
2041 let inflated = (n + 1).to_string();
2042 assert_eq!(inflated.len(), digits, "keep the digit count");
2043 let mut b = good.clone();
2044 b[at..at + digits].copy_from_slice(inflated.as_bytes());
2045 b
2046 };
2047 assert_eq!(broken.len(), good.len(), "the defect must not move bytes");
2048 let raw = lopdf::Document::load_mem(&broken).expect("still loads");
2050 assert!(
2051 raw.get_pages()
2052 .into_values()
2053 .all(|p| raw.get_page_content(p).is_empty()),
2054 "lopdf should drop the stream — if it stops, drop this repair"
2055 );
2056 let text = |b: &[u8]| -> Vec<String> {
2058 super::pdf_textlines(b)
2059 .into_iter()
2060 .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2061 .collect()
2062 };
2063 let expected = text(&good);
2064 assert!(!expected.is_empty(), "control must produce text");
2065 assert_eq!(text(&broken), expected);
2066 }
2067
2068 #[test]
2072 fn repair_declines_when_padding_would_move_objects() {
2073 let mut incremental = pdf_with_xref(false);
2074 incremental.extend_from_slice(b"6 0 obj<</Type/Whatever>>endobj\n");
2075 let declined = super::pad_short_xref_entries(&incremental).unwrap_err();
2076 assert!(
2077 declined.contains("object follows the xref"),
2078 "reason: {declined}"
2079 );
2080 }
2081}
2082
2083#[cfg(test)]
2089mod base14_fonts {
2090 fn pdf_with_font(fontdict: &[u8], text: &[u8]) -> Vec<u8> {
2092 let content = [b"BT /F1 12 Tf 72 700 Td (".as_slice(), text, b") Tj ET\n"].concat();
2093 let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2094 let objs: Vec<Vec<u8>> = vec![
2095 b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2096 b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2097 b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2098 /Resources<</Font<</F1 5 0 R>>>>>>"
2099 .to_vec(),
2100 [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
2101 fontdict.to_vec(),
2102 ];
2103 let mut out = b"%PDF-1.4\n".to_vec();
2104 let mut offsets = Vec::new();
2105 for (i, body) in objs.iter().enumerate() {
2106 offsets.push(out.len());
2107 out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2108 out.extend_from_slice(body);
2109 out.extend_from_slice(b"endobj\n");
2110 }
2111 let xref_at = out.len();
2112 out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2113 out.extend_from_slice(b"0000000000 65535 f \n");
2114 for off in &offsets {
2115 out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2116 }
2117 out.extend_from_slice(
2118 format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2119 );
2120 out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2121 out
2122 }
2123
2124 fn cells(pdf: &[u8]) -> Vec<crate::pdfium_backend::TextCell> {
2126 super::pdf_textlines(pdf)
2127 .into_iter()
2128 .flat_map(|(_, _, c)| c)
2129 .collect()
2130 }
2131
2132 #[test]
2134 fn standard14_faces_get_builtin_widths() {
2135 for fontdict in [
2136 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>".as_slice(),
2138 b"<</Type/Font/Subtype/Type1/BaseFont/Times-BoldItalic>>",
2140 b"<</Type/Font/Subtype/TrueType/BaseFont/Arial,Bold>>",
2142 b"<</Type/Font/Subtype/Type1/BaseFont/ABCDEF+Courier-Oblique>>",
2143 ] {
2144 let pdf = pdf_with_font(fontdict, b"Words have width now");
2145 let cs = cells(&pdf);
2146 let text: String = cs
2147 .iter()
2148 .map(|c| c.text.as_str())
2149 .collect::<Vec<_>>()
2150 .join(" ");
2151 assert!(
2152 text.contains("Words have width now"),
2153 "{}: text lost: {text:?}",
2154 String::from_utf8_lossy(fontdict)
2155 );
2156 assert!(
2157 cs.iter().all(|c| c.r > c.l),
2158 "{}: zero-width cells: {cs:?}",
2159 String::from_utf8_lossy(fontdict)
2160 );
2161 }
2162 }
2163
2164 #[test]
2167 fn explicit_widths_win_and_unknown_faces_are_untouched() {
2168 let explicit = pdf_with_font(
2172 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/FirstChar 65\
2173 /Widths[100 100 100 100]/Encoding/WinAnsiEncoding>>",
2174 b"ABBA",
2175 );
2176 let builtin = pdf_with_font(
2177 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>",
2178 b"ABBA",
2179 );
2180 let w = |pdf: &[u8]| {
2181 let cs = cells(pdf);
2182 assert_eq!(cs.len(), 1, "one word cell: {cs:?}");
2183 cs[0].r - cs[0].l
2184 };
2185 let (we, wb) = (w(&explicit), w(&builtin));
2186 assert!(
2187 (we - 4.8).abs() < 0.1,
2188 "explicit widths must win: got {we}, want 4×100×12/1000"
2189 );
2190 assert!(
2191 wb > 2.0 * we,
2192 "built-in Helvetica is much wider: {wb} vs {we}"
2193 );
2194
2195 let unknown = pdf_with_font(
2199 b"<</Type/Font/Subtype/Type1/BaseFont/FancyCorp-Display>>",
2200 b"Mystery",
2201 );
2202 let cs = cells(&unknown);
2203 let text: String = cs.iter().map(|c| c.text.as_str()).collect();
2204 assert!(text.contains("Mystery"), "text still decodes: {cs:?}");
2205 }
2206}
2207
2208#[cfg(test)]
2209mod overpainted {
2210 use crate::pdfium_backend::TextCell;
2211
2212 fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
2213 TextCell {
2214 text: text.into(),
2215 l,
2216 t,
2217 r,
2218 b,
2219 }
2220 }
2221
2222 #[test]
2226 fn stacked_logo_glyphs_are_dropped() {
2227 let mut cells = vec![
2228 cell("\"", 72.7, 21.5, 86.4, 31.5),
2229 cell("==", 59.4, 21.5, 99.6, 31.5),
2230 cell("Herr", 65.2, 151.3, 81.7, 161.3),
2231 ];
2232 super::drop_overpainted_cells(&mut cells);
2233 assert_eq!(cells.len(), 1, "cells: {cells:?}");
2234 assert_eq!(cells[0].text, "Herr");
2235 }
2236
2237 #[test]
2241 fn prose_and_double_draw_are_kept() {
2242 let mut cells = vec![
2243 cell("Telefon", 354.3, 133.2, 381.5, 143.2),
2244 cell("0676/2000", 387.3, 133.2, 428.7, 143.2),
2245 cell("Bold", 100.0, 50.0, 130.0, 60.0),
2246 cell("Bold", 100.3, 50.0, 130.3, 60.0),
2247 ];
2248 super::drop_overpainted_cells(&mut cells);
2249 assert_eq!(cells.len(), 4);
2250 }
2251}
2252
2253#[cfg(test)]
2254mod vestigial_layer {
2255 use crate::pdfium_backend::{PdfPage, TextCell};
2256
2257 fn page_with(texts: &[&str]) -> PdfPage {
2258 let cells = texts
2259 .iter()
2260 .enumerate()
2261 .map(|(i, t)| TextCell {
2262 text: t.to_string(),
2263 l: 10.0,
2264 t: 10.0 + 12.0 * i as f32,
2265 r: 90.0,
2266 b: 20.0 + 12.0 * i as f32,
2267 })
2268 .collect();
2269 PdfPage::from_cells(595.0, 842.0, 1.0, cells)
2270 }
2271
2272 #[test]
2277 fn typed_in_form_fields_are_not_a_text_layer() {
2278 let pages = vec![
2279 page_with(&["03", "05", "2025"]),
2280 page_with(&[]),
2281 page_with(&[]),
2282 ];
2283 assert!(super::text_layer_is_vestigial(&pages));
2284 assert!(super::text_layer_is_vestigial(&[page_with(&[])]));
2285 }
2286
2287 #[test]
2290 fn sparse_but_real_documents_pass() {
2291 let one_pager = vec![page_with(&[
2292 "Confidential briefing",
2293 "Prepared for the board meeting",
2294 "Do not distribute",
2295 ])];
2296 assert!(!super::text_layer_is_vestigial(&one_pager));
2297 }
2298}