1use std::collections::HashMap;
17use std::rc::Rc;
18
19use lopdf::{Dictionary, Document, Object};
20
21use crate::pdfium_backend::Glyph;
22
23#[derive(Default)]
32struct DocCaches {
33 fonts: HashMap<(lopdf::ObjectId, Vec<u8>), Rc<Font>>,
34 forms: HashMap<lopdf::ObjectId, Rc<lopdf::content::Content>>,
35}
36
37#[derive(Clone, Copy)]
39struct Mat {
40 a: f64,
41 b: f64,
42 c: f64,
43 d: f64,
44 e: f64,
45 f: f64,
46}
47
48impl Mat {
49 const ID: Mat = Mat {
50 a: 1.0,
51 b: 0.0,
52 c: 0.0,
53 d: 1.0,
54 e: 0.0,
55 f: 0.0,
56 };
57
58 fn then(self, m: Mat) -> Mat {
60 Mat {
61 a: self.a * m.a + self.b * m.c,
62 b: self.a * m.b + self.b * m.d,
63 c: self.c * m.a + self.d * m.c,
64 d: self.c * m.b + self.d * m.d,
65 e: self.e * m.a + self.f * m.c + m.e,
66 f: self.e * m.b + self.f * m.d + m.f,
67 }
68 }
69
70 fn apply(self, x: f64, y: f64) -> (f64, f64) {
71 (
72 self.a * x + self.c * y + self.e,
73 self.b * x + self.d * y + self.f,
74 )
75 }
76}
77
78struct Font {
80 two_byte: bool,
82 to_unicode: HashMap<u32, String>,
84 widths: HashMap<u32, f64>,
86 default_width: f64,
87 simple_encoding: Option<HashMap<u8, char>>,
89 fallback_names: HashMap<u8, String>,
93 program_encoding: HashMap<u8, char>,
97 ascent: f64,
98 descent: f64,
99 hash: u64,
100}
101
102impl Font {
103 fn decode_code(&self, code: u32) -> (Option<String>, f64) {
104 let w = self
105 .widths
106 .get(&code)
107 .copied()
108 .unwrap_or(self.default_width);
109 if let Some(s) = self.to_unicode.get(&code) {
110 return (Some(decompose_ligatures(s)), w);
111 }
112 if !self.two_byte {
113 if let Some(name) = self.fallback_names.get(&(code as u8)) {
116 return (Some(format!("/{name}")), w);
117 }
118 if let Some(enc) = &self.simple_encoding {
119 if let Some(&ch) = enc.get(&(code as u8)) {
120 return (Some(decompose_ligatures(&ch.to_string())), w);
121 }
122 }
123 if let Some(&ch) = self.program_encoding.get(&(code as u8)) {
131 return (Some(decompose_ligatures(&ch.to_string())), w);
132 }
133 }
134 (None, w)
135 }
136}
137
138fn decompose_ligatures(s: &str) -> String {
142 if !s.chars().any(|c| ('\u{FB00}'..='\u{FB06}').contains(&c)) {
143 return s.to_string();
144 }
145 s.chars()
146 .map(|c| {
147 match c {
148 '\u{FB00}' => "ff",
149 '\u{FB01}' => "fi",
150 '\u{FB02}' => "fl",
151 '\u{FB03}' => "ffi",
152 '\u{FB04}' => "ffl",
153 '\u{FB05}' => "ft",
154 '\u{FB06}' => "st",
155 _ => return c.to_string(),
156 }
157 .to_string()
158 })
159 .collect()
160}
161
162fn hash_name(name: &[u8]) -> u64 {
163 use std::hash::{Hash, Hasher};
164 let mut h = std::collections::hash_map::DefaultHasher::new();
165 name.hash(&mut h);
166 h.finish()
167}
168
169fn as_dict<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Dictionary> {
171 match obj {
172 Object::Dictionary(d) => Some(d),
173 Object::Reference(id) => doc.get_object(*id).ok().and_then(|o| o.as_dict().ok()),
174 _ => None,
175 }
176}
177
178fn deref<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Object> {
179 match obj {
180 Object::Reference(id) => doc.get_object(*id).ok(),
181 other => Some(other),
182 }
183}
184
185fn parse_font(doc: &Document, name: &[u8], fdict: &Dictionary) -> Font {
187 let subtype: &[u8] = fdict
188 .get(b"Subtype")
189 .ok()
190 .and_then(|o| o.as_name().ok())
191 .unwrap_or(&[]);
192 let two_byte = subtype == b"Type0".as_slice();
193
194 let to_unicode = fdict
195 .get(b"ToUnicode")
196 .ok()
197 .and_then(|o| deref(doc, o))
198 .and_then(|o| o.as_stream().ok())
199 .and_then(|s| s.decompressed_content().ok())
200 .map(|data| parse_tounicode(&data))
201 .unwrap_or_default();
202
203 let (mut widths, mut default_width) = if two_byte {
204 cid_widths(doc, fdict)
205 } else {
206 simple_widths(doc, fdict)
207 };
208
209 let simple_encoding = if two_byte {
210 None
211 } else {
212 Some(simple_encoding_table(doc, fdict))
213 };
214
215 if !two_byte && widths.is_empty() && default_width == 0.0 {
223 if let Some(std14) = base_font_name(fdict).and_then(|n| crate::std14::widths_for(&n)) {
224 if let Some(enc) = &simple_encoding {
225 for (&code, &ch) in enc {
226 if let Some(w) = std14.width(ch) {
227 widths.insert(u32::from(code), w);
228 }
229 }
230 }
231 default_width = 500.0;
234 }
235 }
236 let fallback_names = if two_byte {
237 HashMap::new()
238 } else {
239 differences_gid_names(doc, fdict)
240 };
241 let program_encoding = if two_byte {
242 HashMap::new()
243 } else {
244 type1_program_encoding(doc, fdict)
245 };
246
247 let (ascent, descent) = font_ascent_descent(doc, fdict, two_byte);
248
249 Font {
250 two_byte,
251 to_unicode,
252 widths,
253 default_width,
254 simple_encoding,
255 fallback_names,
256 program_encoding,
257 ascent,
258 descent,
259 hash: hash_name(name),
260 }
261}
262
263fn differences_gid_names(doc: &Document, fdict: &Dictionary) -> HashMap<u8, String> {
270 let mut map = HashMap::new();
271 let Some(Object::Dictionary(enc)) = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o))
272 else {
273 return map;
274 };
275 let Some(Object::Array(diffs)) = enc.get(b"Differences").ok().and_then(|o| deref(doc, o))
276 else {
277 return map;
278 };
279 let mut code = 0u8;
280 for el in diffs {
281 match el {
282 Object::Integer(i) => code = *i as u8,
283 Object::Name(name) => {
284 if glyph_name_to_char(name).is_none() && is_gid_name(name) {
285 map.insert(code, String::from_utf8_lossy(name).into_owned());
286 }
287 code = code.wrapping_add(1);
288 }
289 _ => {}
290 }
291 }
292 map
293}
294
295fn type1_program_encoding(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
303 let mut map = HashMap::new();
304 let Some(desc) = fdict
305 .get(b"FontDescriptor")
306 .ok()
307 .and_then(|o| deref(doc, o))
308 .and_then(|o| o.as_dict().ok())
309 else {
310 return map;
311 };
312 let Some(data) = desc
313 .get(b"FontFile")
314 .ok()
315 .and_then(|o| deref(doc, o))
316 .and_then(|o| o.as_stream().ok())
317 .and_then(|s| s.decompressed_content().ok())
318 else {
319 return map;
320 };
321 let head_end = data
323 .windows(5)
324 .position(|w| w == b"eexec")
325 .unwrap_or(data.len());
326 let head = String::from_utf8_lossy(&data[..head_end]);
327 let toks: Vec<&str> = head.split_whitespace().collect();
329 for w in toks.windows(4) {
330 if w[0] == "dup" && w[3] == "put" {
331 if let (Ok(code), Some(name)) = (w[1].parse::<u32>(), w[2].strip_prefix('/')) {
332 if code <= 255 {
333 if let Some(ch) = glyph_name_to_char(name.as_bytes()) {
334 map.insert(code as u8, ch);
335 }
336 }
337 }
338 }
339 }
340 map
341}
342
343fn is_gid_name(name: &[u8]) -> bool {
349 let Ok(s) = std::str::from_utf8(name) else {
350 return false;
351 };
352 if s.starts_with("afii") || s.starts_with("uni") {
353 return false;
354 }
355 for prefix in ["g", "G", "cid", "CID", "glyph", "index"] {
356 if let Some(rest) = s.strip_prefix(prefix) {
357 if !rest.is_empty() && rest.bytes().all(|b| b.is_ascii_digit()) {
358 return true;
359 }
360 }
361 }
362 let alpha = s.bytes().take_while(|b| b.is_ascii_alphabetic()).count();
366 let digits = s.len() - alpha;
367 (1..=3).contains(&alpha)
368 && digits >= 3
369 && s.as_bytes()[alpha..].iter().all(|b| b.is_ascii_digit())
370}
371
372fn font_ascent_descent(doc: &Document, fdict: &Dictionary, two_byte: bool) -> (f64, f64) {
373 let descr_owner = if two_byte {
375 fdict
376 .get(b"DescendantFonts")
377 .ok()
378 .and_then(|o| deref(doc, o))
379 .and_then(|o| match o {
380 Object::Array(a) => a.first(),
381 _ => None,
382 })
383 .and_then(|o| as_dict(doc, o))
384 } else {
385 Some(fdict)
386 };
387 let fd = descr_owner
388 .and_then(|d| d.get(b"FontDescriptor").ok())
389 .and_then(|o| as_dict(doc, o));
390 let asc = fd
391 .and_then(|d| d.get(b"Ascent").ok())
392 .and_then(|o| {
393 o.as_float()
394 .ok()
395 .or_else(|| o.as_i64().ok().map(|i| i as f32))
396 })
397 .unwrap_or(750.0) as f64;
398 let desc = fd
399 .and_then(|d| d.get(b"Descent").ok())
400 .and_then(|o| {
401 o.as_float()
402 .ok()
403 .or_else(|| o.as_i64().ok().map(|i| i as f32))
404 })
405 .unwrap_or(-250.0) as f64;
406 if asc - desc <= 1.0 {
413 return (750.0, -250.0);
414 }
415 (asc, desc)
416}
417
418fn base_font_name(fdict: &Dictionary) -> Option<Vec<u8>> {
420 let name = fdict.get(b"BaseFont").ok()?.as_name().ok()?;
421 let stripped = match name.iter().position(|&b| b == b'+') {
422 Some(i) if i == 6 => &name[i + 1..],
423 _ => name,
424 };
425 Some(stripped.to_vec())
426}
427
428fn simple_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
430 let mut map = HashMap::new();
431 let first = fdict
432 .get(b"FirstChar")
433 .ok()
434 .and_then(|o| o.as_i64().ok())
435 .unwrap_or(0) as u32;
436 if let Some(Object::Array(arr)) = fdict.get(b"Widths").ok().and_then(|o| deref(doc, o)) {
437 for (i, w) in arr.iter().enumerate() {
438 if let Some(w) = num(w) {
439 map.insert(first + i as u32, w);
440 }
441 }
442 }
443 let dw = fdict
444 .get(b"FontDescriptor")
445 .ok()
446 .and_then(|o| as_dict(doc, o))
447 .and_then(|d| d.get(b"MissingWidth").ok())
448 .and_then(num)
449 .unwrap_or(0.0);
450 (map, dw)
451}
452
453fn cid_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
455 let mut map = HashMap::new();
456 let Some(desc) = fdict
457 .get(b"DescendantFonts")
458 .ok()
459 .and_then(|o| deref(doc, o))
460 .and_then(|o| match o {
461 Object::Array(a) => a.first(),
462 _ => None,
463 })
464 .and_then(|o| as_dict(doc, o))
465 else {
466 return (map, 1000.0);
467 };
468 let dw = desc.get(b"DW").ok().and_then(num).unwrap_or(1000.0);
469 if let Some(Object::Array(w)) = desc.get(b"W").ok().and_then(|o| deref(doc, o)) {
470 let mut i = 0;
471 while i < w.len() {
472 let c = w.get(i).and_then(num);
473 match (c, w.get(i + 1)) {
474 (Some(c), Some(Object::Array(list))) => {
476 for (k, wv) in list.iter().enumerate() {
477 if let Some(wv) = num(wv) {
478 map.insert(c as u32 + k as u32, wv);
479 }
480 }
481 i += 2;
482 }
483 (Some(c1), Some(o2)) => {
485 if let (Some(c2), Some(wv)) = (num(o2), w.get(i + 2).and_then(num)) {
486 for cid in c1 as u32..=c2 as u32 {
487 map.insert(cid, wv);
488 }
489 }
490 i += 3;
491 }
492 _ => break,
493 }
494 }
495 }
496 (map, dw)
497}
498
499fn num(o: &Object) -> Option<f64> {
500 match o {
501 Object::Integer(i) => Some(*i as f64),
502 Object::Real(r) => Some(*r as f64),
503 _ => None,
504 }
505}
506
507fn parse_tounicode(data: &[u8]) -> HashMap<u32, String> {
509 let text = String::from_utf8_lossy(data);
510 let mut map = HashMap::new();
511 let hex = |s: &str| -> Option<Vec<u16>> {
512 let s = s.trim();
513 if !s.starts_with('<') || !s.ends_with('>') {
514 return None;
515 }
516 let h = &s[1..s.len() - 1];
517 let bytes: Vec<u8> = (0..h.len())
518 .step_by(2)
519 .filter_map(|i| u8::from_str_radix(h.get(i..i + 2)?, 16).ok())
520 .collect();
521 Some(
522 bytes
523 .chunks(2)
524 .map(|c| {
525 if c.len() == 2 {
526 u16::from_be_bytes([c[0], c[1]])
527 } else {
528 c[0] as u16
529 }
530 })
531 .collect(),
532 )
533 };
534 let u16s_to_string = |u: &[u16]| String::from_utf16_lossy(u);
535 let code_of = |u: &[u16]| u.iter().fold(0u32, |acc, &x| (acc << 16) | x as u32);
536
537 let tokens: Vec<String> = {
541 let bytes = text.as_bytes();
542 let mut toks = Vec::new();
543 let mut i = 0;
544 while i < bytes.len() {
545 let c = bytes[i];
546 if c.is_ascii_whitespace() {
547 i += 1;
548 } else if c == b'<' {
549 let start = i;
550 while i < bytes.len() && bytes[i] != b'>' {
551 i += 1;
552 }
553 i += 1; toks.push(String::from_utf8_lossy(&bytes[start..i.min(bytes.len())]).into_owned());
555 } else if c == b'[' || c == b']' {
556 toks.push((c as char).to_string());
557 i += 1;
558 } else {
559 let start = i;
560 while i < bytes.len()
561 && !bytes[i].is_ascii_whitespace()
562 && bytes[i] != b'<'
563 && bytes[i] != b'['
564 && bytes[i] != b']'
565 {
566 i += 1;
567 }
568 toks.push(String::from_utf8_lossy(&bytes[start..i]).into_owned());
569 }
570 }
571 toks
572 };
573 let tokens: Vec<&str> = tokens.iter().map(|s| s.as_str()).collect();
574 let mut i = 0;
575 while i < tokens.len() {
576 match tokens[i] {
577 "beginbfchar" => {
578 i += 1;
579 while i + 1 < tokens.len() && tokens[i] != "endbfchar" {
580 if let (Some(src), Some(dst)) = (hex(tokens[i]), hex(tokens[i + 1])) {
581 map.insert(code_of(&src), u16s_to_string(&dst));
582 }
583 i += 2;
584 }
585 }
586 "beginbfrange" => {
587 i += 1;
588 while i + 2 < tokens.len() && tokens[i] != "endbfrange" {
589 let (Some(lo), Some(hi)) = (hex(tokens[i]), hex(tokens[i + 1])) else {
590 i += 1;
591 continue;
592 };
593 let lo = code_of(&lo);
594 let hi = code_of(&hi);
595 if tokens[i + 2] == "[" {
596 let mut j = i + 3;
598 let mut code = lo;
599 while j < tokens.len() && tokens[j] != "]" {
600 if let Some(dst) = hex(tokens[j]) {
601 map.insert(code, u16s_to_string(&dst));
602 }
603 code += 1;
604 j += 1;
605 }
606 i = j + 1;
607 } else if let Some(dst) = hex(tokens[i + 2]) {
608 let base = code_of(&dst);
610 for (k, code) in (lo..=hi).enumerate() {
611 if let Some(ch) = char::from_u32(base + k as u32) {
612 map.insert(code, ch.to_string());
613 }
614 }
615 i += 3;
616 } else {
617 i += 1;
618 }
619 }
620 }
621 _ => i += 1,
622 }
623 }
624 map
625}
626
627fn codes(font: &Font, bytes: &[u8]) -> Vec<u32> {
629 if font.two_byte {
630 bytes
631 .chunks(2)
632 .map(|c| {
633 if c.len() == 2 {
634 ((c[0] as u32) << 8) | c[1] as u32
635 } else {
636 c[0] as u32
637 }
638 })
639 .collect()
640 } else {
641 bytes.iter().map(|&b| b as u32).collect()
642 }
643}
644
645fn page_size(doc: &Document, page_id: lopdf::ObjectId) -> (f32, f32) {
647 let mb = doc
648 .get_object(page_id)
649 .ok()
650 .and_then(|o| o.as_dict().ok())
651 .and_then(|d| {
652 d.get(b"MediaBox").ok().cloned()
654 })
655 .or_else(|| {
656 doc.get_dictionary(page_id)
657 .ok()
658 .and_then(|d| d.get(b"MediaBox").ok().cloned())
659 });
660 if let Some(Object::Array(a)) = mb {
661 let v: Vec<f32> = a.iter().filter_map(|o| num(o).map(|x| x as f32)).collect();
662 if v.len() == 4 {
663 return ((v[2] - v[0]).abs(), (v[3] - v[1]).abs());
664 }
665 }
666 (612.0, 792.0)
667}
668
669pub fn content_diagnosis(bytes: &[u8]) -> String {
675 let Some(doc) = load_document(bytes) else {
676 return "document does not load".into();
677 };
678 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
679 pages.sort_by_key(|(n, _)| *n);
680 let mut out = String::new();
681 let mut caches = DocCaches::default();
682 for (n, pid) in pages.into_iter().take(4) {
683 let content_bytes = doc.get_page_content(pid);
684 let ops = lopdf::content::Content::decode(&content_bytes)
685 .map(|c| c.operations.len())
686 .ok();
687 let res = page_res(&doc, pid);
688 let fonts = res.map(|r| fonts_from_res(&doc, r, &mut caches).len());
689 let glyphs = page_glyphs_cached(&doc, pid, &mut caches).len();
690 out.push_str(&format!(
691 "\n page {n}: content {} B, ops {}, resources {}, fonts {}, glyphs {}",
692 content_bytes.len(),
693 ops.map_or("UNDECODABLE".to_string(), |n| n.to_string()),
694 if res.is_some() { "ok" } else { "MISSING" },
695 fonts.map_or("-".to_string(), |n| n.to_string()),
696 glyphs,
697 ));
698 }
699 out
700}
701
702pub fn text_layer_is_vestigial(pages: &[crate::pdfium_backend::PdfPage]) -> bool {
715 let lines: usize = pages.iter().map(|p| p.cells.len()).sum();
716 if lines == 0 {
717 return true;
718 }
719 let chars: usize = pages
720 .iter()
721 .flat_map(|p| &p.cells)
722 .map(|c| c.text.chars().count())
723 .sum();
724 lines <= pages.len() && chars < 32
725}
726
727pub fn xref_repair_status(bytes: &[u8]) -> String {
731 if Document::load_mem(bytes).is_ok() {
732 return "loads unaided; no repair needed".into();
733 }
734 match pad_short_xref_entries(bytes) {
735 Ok(fixed) => match Document::load_mem(&fixed) {
736 Ok(_) => "repaired: cross-reference entries padded to 20 bytes".into(),
737 Err(e) => format!("padded the entries, but it still will not load: {e}"),
738 },
739 Err(why) => format!("repair declined — {why}"),
740 }
741}
742
743fn load_document(bytes: &[u8]) -> Option<Document> {
758 let mut fallback = None;
763 if let Some(doc) = best_effort_load(bytes, &mut fallback) {
764 return Some(doc);
765 }
766 let xref_fixed = pad_short_xref_entries(bytes).ok();
767 if let Some(fixed) = &xref_fixed {
768 if let Some(doc) = best_effort_load(fixed, &mut fallback) {
769 return Some(doc);
770 }
771 }
772 let lengths_fixed = fix_stream_lengths(xref_fixed.as_deref().unwrap_or(bytes));
775 if let Some(doc) = best_effort_load(&lengths_fixed, &mut fallback) {
776 return Some(doc);
777 }
778 fallback
779}
780
781fn best_effort_load(data: &[u8], fallback: &mut Option<Document>) -> Option<Document> {
784 match Document::load_mem(data) {
785 Ok(doc) if has_page_content(&doc) => Some(doc),
786 Ok(doc) => {
787 fallback.get_or_insert(doc);
788 None
789 }
790 Err(_) => None,
791 }
792}
793
794fn has_page_content(doc: &Document) -> bool {
798 doc.get_pages()
799 .into_values()
800 .take(4)
801 .any(|pid| !doc.get_page_content(pid).is_empty())
802}
803
804fn fix_stream_lengths(bytes: &[u8]) -> Vec<u8> {
817 let mut out = bytes.to_vec();
818 let mut i = 0;
819 while let Some(rel) = find(&out[i..], b"stream") {
820 let kw = i + rel;
821 i = kw + 6;
822 if kw >= 3 && &out[kw - 3..kw] == b"end" {
824 continue;
825 }
826 let mut data = kw + 6;
828 if out.get(data..data + 2) == Some(b"\r\n".as_slice()) {
829 data += 2;
830 } else if matches!(out.get(data), Some(b'\n' | b'\r')) {
831 data += 1;
832 }
833 let Some(end) = find(&out[data..], b"endstream").map(|r| data + r) else {
834 continue;
835 };
836 let dict_start = out[..kw].iter().rposition(|&c| c == b'<').unwrap_or(0);
838 let Some(lrel) = find(&out[dict_start..kw], b"/Length") else {
839 continue;
840 };
841 let mut d = dict_start + lrel + 7;
842 while matches!(out.get(d), Some(b' ')) {
843 d += 1;
844 }
845 let digits = out[d..].iter().take_while(|c| c.is_ascii_digit()).count();
846 if digits == 0 {
847 continue;
848 }
849 let declared: usize = match std::str::from_utf8(&out[d..d + digits])
850 .ok()
851 .and_then(|s| s.parse().ok())
852 {
853 Some(v) => v,
854 None => continue,
855 };
856 let actual = end - data;
857 let replacement = actual.to_string();
860 if actual == declared || replacement.len() > digits {
861 continue;
862 }
863 out[d..d + digits].fill(b' ');
864 out[d..d + replacement.len()].copy_from_slice(replacement.as_bytes());
865 }
866 out
867}
868
869fn find(haystack: &[u8], needle: &[u8]) -> Option<usize> {
870 haystack.windows(needle.len()).position(|w| w == needle)
871}
872
873fn pad_short_xref_entries(bytes: &[u8]) -> Result<Vec<u8>, &'static str> {
876 let is_boundary = |i: usize| i == 0 || matches!(bytes[i - 1], b'\n' | b'\r');
879 let mut starts = (0..bytes.len().saturating_sub(4))
880 .filter(|&i| &bytes[i..i + 4] == b"xref" && is_boundary(i));
881 let xref_at = starts
882 .next()
883 .ok_or("no classic `xref` section (an xref stream?)")?;
884 if starts.next().is_some() {
885 return Err("more than one xref section (incremental update)");
886 }
887 let last_obj = bytes
888 .windows(3)
889 .rposition(|w| w == b"obj")
890 .ok_or("no objects found")?;
891 if last_obj > xref_at {
892 return Err("an object follows the xref — padding would move it");
893 }
894
895 let mut out = bytes[..xref_at].to_vec();
896 out.extend_from_slice(b"xref\n");
897 let mut i = xref_at + 4;
898 let skip_ws = |i: &mut usize| {
899 while matches!(bytes.get(*i), Some(b'\r' | b'\n' | b' ')) {
900 *i += 1;
901 }
902 };
903 loop {
904 skip_ws(&mut i);
905 if bytes[i..].starts_with(b"trailer") {
907 out.extend_from_slice(&bytes[i..]);
908 return Ok(out);
909 }
910 let header_end = i + bytes[i..]
911 .iter()
912 .position(|c| matches!(c, b'\n' | b'\r'))
913 .ok_or("subsection header runs off the end")?;
914 let header = std::str::from_utf8(&bytes[i..header_end])
915 .map_err(|_| "subsection header is not text")?
916 .trim();
917 let mut parts = header.split_whitespace();
918 let count: usize = parts
919 .nth(1)
920 .and_then(|c| c.parse().ok())
921 .ok_or("unparseable subsection header")?;
922 if parts.next().is_some() || count == 0 {
923 return Err("unexpected subsection header shape");
924 }
925 out.extend_from_slice(header.as_bytes());
926 out.push(b'\n');
927 i = header_end;
928 for _ in 0..count {
929 skip_ws(&mut i);
930 let entry = bytes.get(i..i + 18).ok_or("xref entry runs off the end")?;
932 let well_formed = entry[..10].iter().all(u8::is_ascii_digit)
933 && entry[10] == b' '
934 && entry[11..16].iter().all(u8::is_ascii_digit)
935 && entry[16] == b' '
936 && matches!(entry[17], b'n' | b'f');
937 if !well_formed {
938 return Err("xref entry is not `nnnnnnnnnn ggggg n`");
939 }
940 out.extend_from_slice(entry);
941 out.extend_from_slice(b" \n"); i += 18;
943 }
944 }
945}
946
947pub fn debug_glyphs(bytes: &[u8], index: usize) -> Vec<(char, f32, f32, f32, f32)> {
950 let Some(doc) = load_document(bytes) else {
951 return Vec::new();
952 };
953 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
954 pages.sort_by_key(|(n, _)| *n);
955 let Some((_, pid)) = pages.get(index) else {
956 return Vec::new();
957 };
958 page_glyphs(&doc, *pid)
959 .into_iter()
960 .map(|g| (g.ch, g.ll, g.lr, g.lb, g.lt))
961 .collect()
962}
963
964pub fn pdf_textlines(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
968 let Some(doc) = load_document(bytes) else {
969 return Vec::new();
970 };
971 let mut caches = DocCaches::default();
972 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
973 pages.sort_by_key(|(n, _)| *n);
974 pages
975 .into_iter()
976 .map(|(_, pid)| {
977 let (w, h) = page_size(&doc, pid);
978 let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
979 let cells = crate::dp_lines::line_cells(&glyphs, h, true);
980 (w, h, cells)
981 })
982 .collect()
983}
984
985pub fn pdf_words(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
990 let Some(doc) = load_document(bytes) else {
991 return Vec::new();
992 };
993 let mut caches = DocCaches::default();
994 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
995 pages.sort_by_key(|(n, _)| *n);
996 pages
997 .into_iter()
998 .map(|(_, pid)| {
999 let (w, h) = page_size(&doc, pid);
1000 let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1001 let cells = crate::dp_lines::word_cells(&glyphs, h, true);
1002 (w, h, cells)
1003 })
1004 .collect()
1005}
1006
1007#[derive(Default)]
1011pub struct PageParserCells {
1012 pub prose: Vec<crate::pdfium_backend::TextCell>,
1013 pub words: Vec<crate::pdfium_backend::TextCell>,
1014 pub code: Vec<crate::pdfium_backend::TextCell>,
1015}
1016
1017pub fn pdf_all_cells(bytes: &[u8]) -> Vec<PageParserCells> {
1022 let Some(doc) = load_document(bytes) else {
1023 return Vec::new();
1024 };
1025 let mut caches = DocCaches::default();
1026 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1027 pages.sort_by_key(|(n, _)| *n);
1028 pages
1029 .into_iter()
1030 .map(|(_, pid)| {
1031 let (_w, h) = page_size(&doc, pid);
1032 let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1033 let (prose, words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1034 PageParserCells {
1035 prose,
1036 words,
1037 code: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1038 }
1039 })
1040 .collect()
1041}
1042
1043pub fn pdf_text_pages(bytes: &[u8]) -> Vec<crate::pdfium_backend::PdfPage> {
1050 let Some(doc) = load_document(bytes) else {
1051 return Vec::new();
1052 };
1053 let mut caches = DocCaches::default();
1054 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1055 pages.sort_by_key(|(n, _)| *n);
1056 pages
1057 .into_iter()
1058 .map(|(_, pid)| {
1059 let (w, h) = page_size(&doc, pid);
1060 let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1061 let (mut prose, mut words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1062 drop_overpainted_cells(&mut prose);
1063 drop_overpainted_cells(&mut words);
1064 crate::pdfium_backend::PdfPage {
1065 #[cfg(feature = "ocr-prep")]
1066 image_layout: None,
1067 width: w,
1068 height: h,
1069 scale: 1.0,
1071 cells: prose,
1072 code_cells: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1073 word_cells: words,
1074 #[cfg(feature = "ocr-prep")]
1075 image: image::RgbImage::new(1, 1),
1076 links: Vec::new(),
1077 }
1078 })
1079 .collect()
1080}
1081
1082fn drop_overpainted_cells(cells: &mut Vec<crate::pdfium_backend::TextCell>) {
1101 let mut paint = vec![false; cells.len()];
1102 for i in 0..cells.len() {
1103 for j in 0..cells.len() {
1104 if i == j || cells[i].text == cells[j].text {
1105 continue;
1106 }
1107 let (a, b) = (&cells[i], &cells[j]);
1108 let vo = (a.b.min(b.b) - a.t.max(b.t)).max(0.0);
1110 if vo < 0.6 * (a.b - a.t).min(b.b - b.t) {
1111 continue;
1112 }
1113 let ho = (a.r.min(b.r) - a.l.max(b.l)).max(0.0);
1115 if ho >= 0.8 * (a.r - a.l) && (a.r - a.l) <= (b.r - b.l) {
1116 paint[i] = true;
1117 paint[j] = true;
1118 }
1119 }
1120 }
1121 let mut keep = paint.iter().map(|p| !p);
1122 cells.retain(|_| keep.next().unwrap());
1123}
1124
1125#[derive(Clone, Copy)]
1129struct TextState {
1130 tc: f64,
1131 tw: f64,
1132 th: f64,
1133 tl: f64,
1134 trise: f64,
1135 fsize: f64,
1136}
1137
1138impl TextState {
1139 const INIT: TextState = TextState {
1140 tc: 0.0,
1141 tw: 0.0,
1142 th: 1.0,
1143 tl: 0.0,
1144 trise: 0.0,
1145 fsize: 0.0,
1146 };
1147}
1148
1149fn page_res(doc: &Document, page_id: lopdf::ObjectId) -> Option<&Dictionary> {
1152 let (inline, ids) = doc.get_page_resources(page_id).ok()?;
1153 if let Some(d) = inline {
1154 return Some(d);
1155 }
1156 ids.into_iter().find_map(|id| doc.get_dictionary(id).ok())
1157}
1158
1159fn fonts_from_res(
1163 doc: &Document,
1164 res: &Dictionary,
1165 caches: &mut DocCaches,
1166) -> HashMap<Vec<u8>, Rc<Font>> {
1167 let mut map = HashMap::new();
1168 let font_dict = res
1169 .get(b"Font")
1170 .ok()
1171 .and_then(|o| deref(doc, o))
1172 .and_then(|o| o.as_dict().ok());
1173 if let Some(fd) = font_dict {
1174 for (name, value) in fd.iter() {
1175 let font = match value {
1176 Object::Reference(id) => {
1177 let key = (*id, name.clone());
1178 if let Some(f) = caches.fonts.get(&key) {
1179 Rc::clone(f)
1180 } else if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1181 let f = Rc::new(parse_font(doc, name, fdict));
1182 caches.fonts.insert(key, Rc::clone(&f));
1183 f
1184 } else {
1185 continue;
1186 }
1187 }
1188 _ => {
1189 if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1190 Rc::new(parse_font(doc, name, fdict))
1191 } else {
1192 continue;
1193 }
1194 }
1195 };
1196 map.insert(name.clone(), font);
1197 }
1198 }
1199 map
1200}
1201
1202pub(crate) fn page_glyphs(doc: &Document, page_id: lopdf::ObjectId) -> Vec<Glyph> {
1204 page_glyphs_cached(doc, page_id, &mut DocCaches::default())
1205}
1206
1207fn page_glyphs_cached(
1210 doc: &Document,
1211 page_id: lopdf::ObjectId,
1212 caches: &mut DocCaches,
1213) -> Vec<Glyph> {
1214 let mut out = Vec::new();
1215 let content_bytes = doc.get_page_content(page_id);
1218 let Ok(content) = lopdf::content::Content::decode(&content_bytes) else {
1219 return out;
1220 };
1221 if let Some(res) = page_res(doc, page_id) {
1222 run_content(
1223 doc,
1224 res,
1225 &content,
1226 Mat::ID,
1227 TextState::INIT,
1228 0,
1229 caches,
1230 &mut out,
1231 );
1232 }
1233 out
1234}
1235
1236#[allow(clippy::too_many_arguments)]
1241fn run_content(
1242 doc: &Document,
1243 res: &Dictionary,
1244 content: &lopdf::content::Content,
1245 base_ctm: Mat,
1246 init: TextState,
1247 depth: u32,
1248 caches: &mut DocCaches,
1249 out: &mut Vec<Glyph>,
1250) {
1251 let fonts = fonts_from_res(doc, res, caches);
1252 let xobjects = res
1253 .get(b"XObject")
1254 .ok()
1255 .and_then(|o| deref(doc, o))
1256 .and_then(|o| o.as_dict().ok());
1257
1258 #[allow(clippy::type_complexity)]
1263 let mut gstate_stack: Vec<(Mat, f64, f64, f64, f64, f64, f64, Option<&Rc<Font>>)> = Vec::new();
1264 let mut ctm = base_ctm;
1265 let mut tm = Mat::ID;
1266 let mut tlm = Mat::ID;
1267 let mut font: Option<&Rc<Font>> = None;
1268 let mut fsize = init.fsize;
1269 let mut tc = init.tc; let mut tw = init.tw; let mut th = init.th; let mut tl = init.tl; let mut trise = init.trise;
1274
1275 let op_f = |operands: &[Object], i: usize| operands.get(i).and_then(num).unwrap_or(0.0);
1276
1277 for op in &content.operations {
1278 let operands = &op.operands;
1279 match op.operator.as_str() {
1280 "q" => gstate_stack.push((ctm, tc, tw, th, tl, trise, fsize, font)),
1281 "Q" => {
1282 if let Some((c, a, b, h, l, r, fs, f)) = gstate_stack.pop() {
1283 ctm = c;
1284 tc = a;
1285 tw = b;
1286 th = h;
1287 tl = l;
1288 trise = r;
1289 fsize = fs;
1290 font = f;
1291 }
1292 }
1293 "cm" => {
1294 let m = Mat {
1295 a: op_f(operands, 0),
1296 b: op_f(operands, 1),
1297 c: op_f(operands, 2),
1298 d: op_f(operands, 3),
1299 e: op_f(operands, 4),
1300 f: op_f(operands, 5),
1301 };
1302 ctm = m.then(ctm);
1303 }
1304 "BT" => {
1305 tm = Mat::ID;
1306 tlm = Mat::ID;
1307 }
1308 "ET" => {}
1309 "Tf" => {
1310 if let Some(Object::Name(n)) = operands.first() {
1311 font = fonts.get(n.as_slice());
1312 }
1313 fsize = op_f(operands, 1);
1314 }
1315 "Td" => {
1316 tlm = Mat {
1317 a: 1.0,
1318 b: 0.0,
1319 c: 0.0,
1320 d: 1.0,
1321 e: op_f(operands, 0),
1322 f: op_f(operands, 1),
1323 }
1324 .then(tlm);
1325 tm = tlm;
1326 }
1327 "TD" => {
1328 tl = -op_f(operands, 1);
1329 tlm = Mat {
1330 a: 1.0,
1331 b: 0.0,
1332 c: 0.0,
1333 d: 1.0,
1334 e: op_f(operands, 0),
1335 f: op_f(operands, 1),
1336 }
1337 .then(tlm);
1338 tm = tlm;
1339 }
1340 "Tm" => {
1341 tlm = Mat {
1342 a: op_f(operands, 0),
1343 b: op_f(operands, 1),
1344 c: op_f(operands, 2),
1345 d: op_f(operands, 3),
1346 e: op_f(operands, 4),
1347 f: op_f(operands, 5),
1348 };
1349 tm = tlm;
1350 }
1351 "T*" => {
1352 tlm = Mat {
1353 a: 1.0,
1354 b: 0.0,
1355 c: 0.0,
1356 d: 1.0,
1357 e: 0.0,
1358 f: -tl,
1359 }
1360 .then(tlm);
1361 tm = tlm;
1362 }
1363 "Tc" => tc = op_f(operands, 0),
1364 "Tw" => tw = op_f(operands, 0),
1365 "Tz" => th = op_f(operands, 0) / 100.0,
1366 "TL" => tl = op_f(operands, 0),
1367 "Ts" => trise = op_f(operands, 0),
1368 "Tj" | "'" | "\"" => {
1369 if op.operator == "'" || op.operator == "\"" {
1370 tlm = Mat {
1372 a: 1.0,
1373 b: 0.0,
1374 c: 0.0,
1375 d: 1.0,
1376 e: 0.0,
1377 f: -tl,
1378 }
1379 .then(tlm);
1380 tm = tlm;
1381 }
1382 if op.operator == "\"" {
1383 tw = op_f(operands, 0);
1386 tc = op_f(operands, 1);
1387 }
1388 if let (Some(f), Some(Object::String(s, _))) = (font, operands.last()) {
1389 show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out);
1390 }
1391 }
1392 "TJ" => {
1393 if let (Some(f), Some(Object::Array(arr))) = (font, operands.first()) {
1394 for el in arr {
1395 match el {
1396 Object::String(s, _) => {
1397 show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out)
1398 }
1399 other => {
1400 if let Some(adj) = num(other) {
1401 let tx = -adj / 1000.0 * fsize * th;
1403 tm = Mat {
1404 a: 1.0,
1405 b: 0.0,
1406 c: 0.0,
1407 d: 1.0,
1408 e: tx,
1409 f: 0.0,
1410 }
1411 .then(tm);
1412 }
1413 }
1414 }
1415 }
1416 }
1417 }
1418 "Do" => {
1419 if depth >= 8 {
1422 continue;
1423 }
1424 let Some(Object::Name(n)) = operands.first() else {
1425 continue;
1426 };
1427 let obj = xobjects.and_then(|d| d.get(n.as_slice()).ok());
1428 let form_id = match obj {
1429 Some(Object::Reference(id)) => Some(*id),
1430 _ => None,
1431 };
1432 let stream = obj
1433 .and_then(|o| deref(doc, o))
1434 .and_then(|o| o.as_stream().ok());
1435 let Some(stream) = stream else { continue };
1436 let is_form = stream
1437 .dict
1438 .get(b"Subtype")
1439 .ok()
1440 .and_then(|o| o.as_name().ok())
1441 == Some(b"Form".as_slice());
1442 if !is_form {
1443 continue;
1444 }
1445 let cached = form_id.and_then(|id| caches.forms.get(&id).cloned());
1448 let form_content = match cached {
1449 Some(c) => c,
1450 None => {
1451 let Ok(data) = stream.decompressed_content() else {
1452 continue;
1453 };
1454 let Ok(c) = lopdf::content::Content::decode(&data) else {
1455 continue;
1456 };
1457 let c = Rc::new(c);
1458 if let Some(id) = form_id {
1459 caches.forms.insert(id, Rc::clone(&c));
1460 }
1461 c
1462 }
1463 };
1464 let form_mat = match stream.dict.get(b"Matrix").ok() {
1466 Some(Object::Array(a)) if a.len() == 6 => {
1467 let v: Vec<f64> = a.iter().filter_map(num).collect();
1468 if v.len() == 6 {
1469 Mat {
1470 a: v[0],
1471 b: v[1],
1472 c: v[2],
1473 d: v[3],
1474 e: v[4],
1475 f: v[5],
1476 }
1477 } else {
1478 Mat::ID
1479 }
1480 }
1481 _ => Mat::ID,
1482 };
1483 let form_res = stream
1485 .dict
1486 .get(b"Resources")
1487 .ok()
1488 .and_then(|o| deref(doc, o))
1489 .and_then(|o| o.as_dict().ok())
1490 .unwrap_or(res);
1491 let state = TextState {
1492 tc,
1493 tw,
1494 th,
1495 tl,
1496 trise,
1497 fsize,
1498 };
1499 run_content(
1500 doc,
1501 form_res,
1502 &form_content,
1503 form_mat.then(ctm),
1504 state,
1505 depth + 1,
1506 caches,
1507 out,
1508 );
1509 }
1510 _ => {}
1511 }
1512 }
1513}
1514
1515#[allow(clippy::too_many_arguments)]
1516fn show_text(
1517 font: &Font,
1518 bytes: &[u8],
1519 fsize: f64,
1520 tc: f64,
1521 tw: f64,
1522 th: f64,
1523 trise: f64,
1524 tm: &mut Mat,
1525 ctm: Mat,
1526 out: &mut Vec<Glyph>,
1527) {
1528 for code in codes(font, bytes) {
1529 let (text, w) = font.decode_code(code);
1530 let w0 = w / 1000.0; let scale = Mat {
1533 a: fsize * th,
1534 b: 0.0,
1535 c: 0.0,
1536 d: fsize,
1537 e: 0.0,
1538 f: trise,
1539 };
1540 let trm = scale.then(*tm).then(ctm);
1541 let (x0, y0) = trm.apply(0.0, font.descent / 1000.0);
1543 let (x1, _y1) = trm.apply(w0, font.descent / 1000.0);
1544 let (_x2, y2) = trm.apply(0.0, font.ascent / 1000.0);
1545 let (left, right) = (x0.min(x1), x0.max(x1));
1546 let (bot, top) = (y0.min(y2), y0.max(y2));
1547 if let Some(s) = text {
1548 for ch in s.chars() {
1550 if ch != '\u{0}' {
1551 out.push(Glyph {
1552 ch,
1553 l: left as f32,
1554 b: bot as f32,
1555 r: right as f32,
1556 t: top as f32,
1557 ll: left as f32,
1558 lb: bot as f32,
1559 lr: right as f32,
1560 lt: top as f32,
1561 font: font.hash,
1562 });
1563 }
1564 }
1565 }
1566 let is_space = !font.two_byte && code == 32;
1568 let tx = (w0 * fsize + tc + if is_space { tw } else { 0.0 }) * th;
1569 *tm = Mat {
1570 a: 1.0,
1571 b: 0.0,
1572 c: 0.0,
1573 d: 1.0,
1574 e: tx,
1575 f: 0.0,
1576 }
1577 .then(*tm);
1578 }
1579}
1580
1581fn simple_encoding_table(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
1585 let enc = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o));
1586 let base_name = match enc {
1587 Some(Object::Name(n)) => n.clone(),
1588 Some(Object::Dictionary(d)) => d
1589 .get(b"BaseEncoding")
1590 .ok()
1591 .and_then(|o| o.as_name().ok())
1592 .map(|n| n.to_vec())
1593 .unwrap_or_default(),
1594 _ => Vec::new(),
1595 };
1596 let mut m = if base_name == b"MacRomanEncoding" {
1597 macroman_table()
1598 } else if base_name.is_empty() {
1599 tex_math_builtin(fdict).unwrap_or_else(winansi_table)
1606 } else {
1607 winansi_table()
1608 };
1609 if let Some(Object::Dictionary(d)) = enc {
1611 if let Some(Object::Array(diffs)) = d.get(b"Differences").ok().and_then(|o| deref(doc, o)) {
1612 let mut code = 0u8;
1613 for el in diffs {
1614 match el {
1615 Object::Integer(i) => code = *i as u8,
1616 Object::Name(name) => {
1617 if let Some(ch) = glyph_name_to_char(name) {
1618 m.insert(code, ch);
1619 }
1620 code = code.wrapping_add(1);
1621 }
1622 _ => {}
1623 }
1624 }
1625 }
1626 }
1627 m
1628}
1629
1630fn tex_math_builtin(fdict: &Dictionary) -> Option<HashMap<u8, char>> {
1636 const CMSY: [char; 128] = [
1637 '−', '·', '×', '∗', '÷', '⋄', '±', '∓', '⊕', '⊖', '⊗', '⊘', '⊙', '◯', '∘', '•', '≍', '≡',
1638 '⊆', '⊇', '≤', '≥', '≼', '≽', '∼', '≈', '⊂', '⊃', '≪', '≫', '≺', '≻', '←', '→', '↑', '↓',
1639 '↔', '↗', '↘', '≃', '⇐', '⇒', '⇑', '⇓', '⇔', '↖', '↙', '∝', '′', '∞', '∈', '∋', '△', '▽',
1640 '\u{338}', '↦', '∀', '∃', '¬', '∅', 'ℜ', 'ℑ', '⊤', '⊥', 'ℵ', 'A', 'B', 'C', 'D', 'E', 'F',
1641 'G', 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X',
1642 'Y', 'Z', '∪', '∩', '⊎', '∧', '∨', '⊢', '⊣', '⌊', '⌋', '⌈', '⌉', '{', '}', '⟨', '⟩', '|',
1643 '∥', '↕', '⇕', '\\', '≀', '√', '∐', '∇', '∫', '⊔', '⊓', '⊑', '⊒', '§', '†', '‡', '¶', '♣',
1644 '♢', '♡', '♠',
1645 ];
1646 const CMMI: [char; 128] = [
1647 'Γ', 'Δ', 'Θ', 'Λ', 'Ξ', 'Π', 'Σ', 'Υ', 'Φ', 'Ψ', 'Ω', 'α', 'β', 'γ', 'δ', 'ε', 'ζ', 'η',
1648 'θ', 'ι', 'κ', 'λ', 'μ', 'ν', 'ξ', 'π', 'ρ', 'σ', 'τ', 'υ', 'φ', 'χ', 'ψ', 'ω', 'ϵ', 'ϑ',
1649 'ϖ', 'ϱ', 'ς', 'ϕ', '↼', '↽', '⇀', '⇁', '↩', '↪', '▷', '◁', '0', '1', '2', '3', '4', '5',
1650 '6', '7', '8', '9', '.', ',', '<', '/', '>', '⋆', '∂', 'A', 'B', 'C', 'D', 'E', 'F', 'G',
1651 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y',
1652 'Z', '♭', '♮', '♯', '⌣', '⌢', 'ℓ', 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k',
1653 'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', 'ı', 'ȷ', '℘',
1654 '\u{20d7}', '⁀',
1655 ];
1656 let name = base_font_name(fdict)?;
1657 let up = name.to_ascii_uppercase();
1658 let table: &[char; 128] = if up.starts_with(b"CMSY") || up.starts_with(b"CMBSY") {
1659 &CMSY
1660 } else if up.starts_with(b"CMMI") {
1661 &CMMI
1662 } else {
1663 return None;
1664 };
1665 Some(
1666 table
1667 .iter()
1668 .enumerate()
1669 .map(|(i, &c)| (i as u8, c))
1670 .collect(),
1671 )
1672}
1673
1674fn glyph_name_to_char(name: &[u8]) -> Option<char> {
1679 let s = std::str::from_utf8(name).ok()?;
1680 if let Some(hex) = s.strip_prefix("uni") {
1681 if let Ok(cp) = u32::from_str_radix(hex.get(0..4)?, 16) {
1682 return char::from_u32(cp);
1683 }
1684 }
1685 if s.len() == 1 {
1687 let b = s.as_bytes()[0];
1688 if b.is_ascii_alphabetic() {
1689 return Some(b as char);
1690 }
1691 }
1692 let resolved = match s {
1693 "space" => ' ',
1694 "exclam" => '!',
1695 "quotedbl" => '"',
1696 "numbersign" => '#',
1697 "dollar" => '$',
1698 "percent" => '%',
1699 "ampersand" => '&',
1700 "quotesingle" => '\'',
1701 "parenleft" => '(',
1702 "parenright" => ')',
1703 "asterisk" => '*',
1704 "plus" => '+',
1705 "comma" => ',',
1706 "hyphen" => '-',
1707 "period" => '.',
1708 "slash" => '/',
1709 "zero" => '0',
1710 "one" => '1',
1711 "two" => '2',
1712 "three" => '3',
1713 "four" => '4',
1714 "five" => '5',
1715 "six" => '6',
1716 "seven" => '7',
1717 "eight" => '8',
1718 "nine" => '9',
1719 "colon" => ':',
1720 "semicolon" => ';',
1721 "less" => '<',
1722 "equal" => '=',
1723 "greater" => '>',
1724 "question" => '?',
1725 "at" => '@',
1726 "bracketleft" => '[',
1727 "backslash" => '\\',
1728 "bracketright" => ']',
1729 "asciicircum" => '^',
1730 "underscore" => '_',
1731 "grave" => '`',
1732 "braceleft" => '{',
1733 "bar" => '|',
1734 "braceright" => '}',
1735 "asciitilde" => '~',
1736 "bullet" => '\u{2022}',
1737 "periodcentered" => '\u{00B7}',
1738 "endash" => '\u{2013}',
1739 "emdash" => '\u{2014}',
1740 "quoteright" => '\u{2019}',
1741 "quoteleft" => '\u{2018}',
1742 "quotedblleft" => '\u{201C}',
1743 "quotedblright" => '\u{201D}',
1744 "quotedblbase" => '\u{201E}',
1745 "quotesinglbase" => '\u{201A}',
1746 "ff" => '\u{FB00}',
1751 "fi" => '\u{FB01}',
1752 "fl" => '\u{FB02}',
1753 "ffi" => '\u{FB03}',
1754 "ffl" => '\u{FB04}',
1755 "ft" => '\u{FB05}',
1756 "st" => '\u{FB06}',
1757 "degree" => '\u{00B0}',
1758 "trademark" => '\u{2122}',
1759 "registered" => '\u{00AE}',
1760 "copyright" => '\u{00A9}',
1761 "ellipsis" => '\u{2026}',
1762 "minus" => '\u{2212}',
1763 "fraction" => '\u{2044}',
1764 "nbspace" => '\u{00A0}',
1765 "alpha" => '\u{03B1}',
1770 "beta" => '\u{03B2}',
1771 "gamma" => '\u{03B3}',
1772 "delta" => '\u{03B4}',
1773 "epsilon" | "epsilon1" => '\u{03B5}',
1774 "zeta" => '\u{03B6}',
1775 "eta" => '\u{03B7}',
1776 "theta" | "theta1" => '\u{03B8}',
1777 "iota" => '\u{03B9}',
1778 "kappa" => '\u{03BA}',
1779 "lambda" => '\u{03BB}',
1780 "mu" => '\u{03BC}',
1781 "nu" => '\u{03BD}',
1782 "xi" => '\u{03BE}',
1783 "omicron" => '\u{03BF}',
1784 "pi" | "pi1" => '\u{03C0}',
1785 "rho" | "rho1" => '\u{03C1}',
1786 "sigma" => '\u{03C3}',
1787 "sigma1" => '\u{03C2}',
1788 "tau" => '\u{03C4}',
1789 "upsilon" => '\u{03C5}',
1790 "phi" | "phi1" => '\u{03C6}',
1791 "chi" => '\u{03C7}',
1792 "psi" => '\u{03C8}',
1793 "omega" | "omega1" => '\u{03C9}',
1794 "Gamma" => '\u{0393}',
1795 "Delta" => '\u{0394}',
1796 "Theta" => '\u{0398}',
1797 "Lambda" => '\u{039B}',
1798 "Xi" => '\u{039E}',
1799 "Pi" => '\u{03A0}',
1800 "Sigma" => '\u{03A3}',
1801 "Upsilon" => '\u{03A5}',
1802 "Phi" => '\u{03A6}',
1803 "Psi" => '\u{03A8}',
1804 "Omega" => '\u{03A9}',
1805 "lessequal" => '\u{2264}',
1806 "greaterequal" => '\u{2265}',
1807 "notequal" => '\u{2260}',
1808 "approxequal" => '\u{2248}',
1809 "equivalence" => '\u{2261}',
1810 "element" => '\u{2208}',
1811 "plusminus" => '\u{00B1}',
1812 "multiply" => '\u{00D7}',
1813 "divide" => '\u{00F7}',
1814 "infinity" => '\u{221E}',
1815 "partialdiff" => '\u{2202}',
1816 "gradient" => '\u{2207}',
1817 "summation" => '\u{2211}',
1818 "product" => '\u{220F}',
1819 "integral" => '\u{222B}',
1820 "radical" => '\u{221A}',
1821 "proportional" => '\u{221D}',
1822 "arrowright" => '\u{2192}',
1823 "arrowleft" => '\u{2190}',
1824 "arrowup" => '\u{2191}',
1825 "arrowdown" => '\u{2193}',
1826 "arrowboth" => '\u{2194}',
1827 "arrowdblright" => '\u{21D2}',
1828 "logicaland" => '\u{2227}',
1829 "logicalor" => '\u{2228}',
1830 "intersection" => '\u{2229}',
1831 "union" => '\u{222A}',
1832 "similar" => '\u{223C}',
1833 "congruent" => '\u{2245}',
1834 "dotmath" => '\u{22C5}',
1835 "asteriskmath" => '\u{2217}',
1836 _ => {
1837 if let Some((base, _)) = s.split_once('.') {
1839 if !base.is_empty() {
1840 return glyph_name_to_char(base.as_bytes());
1841 }
1842 }
1843 return None;
1844 }
1845 };
1846 Some(resolved)
1847}
1848
1849fn winansi_table() -> HashMap<u8, char> {
1851 let mut m = HashMap::new();
1852 for b in 0x20u8..=0x7e {
1853 m.insert(b, b as char);
1854 }
1855 let extra: &[(u8, char)] = &[
1857 (0x91, '\u{2018}'),
1858 (0x92, '\u{2019}'),
1859 (0x93, '\u{201C}'),
1860 (0x94, '\u{201D}'),
1861 (0x95, '\u{2022}'),
1862 (0x96, '\u{2013}'),
1863 (0x97, '\u{2014}'),
1864 (0x85, '\u{2026}'),
1865 (0xA0, '\u{00A0}'),
1866 ];
1867 for &(b, c) in extra {
1868 m.insert(b, c);
1869 }
1870 for b in 0xA1u8..=0xFF {
1871 m.entry(b).or_insert(b as char);
1872 }
1873 m
1874}
1875
1876fn macroman_table() -> HashMap<u8, char> {
1879 let mut m = HashMap::new();
1880 for b in 0x20u8..=0x7e {
1881 m.insert(b, b as char);
1882 }
1883 let high: &[(u8, char)] = &[
1884 (0xA5, '\u{2022}'), (0xD0, '\u{2013}'), (0xD1, '\u{2014}'), (0xD2, '\u{201C}'),
1888 (0xD3, '\u{201D}'),
1889 (0xD4, '\u{2018}'),
1890 (0xD5, '\u{2019}'),
1891 (0xCA, '\u{00A0}'),
1892 (0xC9, '\u{2026}'),
1893 (0xDE, '\u{FB01}'),
1894 (0xDF, '\u{FB02}'),
1895 ];
1896 for &(b, c) in high {
1897 m.insert(b, c);
1898 }
1899 m
1900}
1901
1902#[cfg(test)]
1903mod xref_repair {
1904 fn pdf_with_xref(two_byte_eol: bool) -> Vec<u8> {
1908 let content = b"BT /F1 12 Tf 72 700 Td (Invoice 922769430725) Tj ET\n";
1909 let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
1910 let objs: Vec<Vec<u8>> = vec![
1911 b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
1912 b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
1913 b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
1914 /Resources<</Font<</F1 5 0 R>>>>>>"
1915 .to_vec(),
1916 [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
1917 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".to_vec(),
1918 ];
1919
1920 let mut out = b"%PDF-1.4\n".to_vec();
1921 let mut offsets = Vec::new();
1922 for (i, body) in objs.iter().enumerate() {
1923 offsets.push(out.len());
1924 out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
1925 out.extend_from_slice(body);
1926 out.extend_from_slice(b"endobj\n");
1927 }
1928 let xref_at = out.len();
1929 let eol: &[u8] = if two_byte_eol { b" \n" } else { b"\n" };
1930 out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
1931 out.extend_from_slice(b"0000000000 65535 f");
1932 out.extend_from_slice(eol);
1933 for off in &offsets {
1934 out.extend_from_slice(format!("{off:010} 00000 n").as_bytes());
1935 out.extend_from_slice(eol);
1936 }
1937 out.extend_from_slice(
1938 format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
1939 );
1940 out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
1941 out
1942 }
1943
1944 #[test]
1950 fn short_xref_entries_still_parse() {
1951 let good = pdf_with_xref(true);
1952 let broken = pdf_with_xref(false);
1953 assert!(
1954 broken.len() < good.len(),
1955 "the broken file is the shorter one"
1956 );
1957 assert!(
1958 lopdf::Document::load_mem(&good).is_ok(),
1959 "the control file must load unaided"
1960 );
1961 assert!(
1962 lopdf::Document::load_mem(&broken).is_err(),
1963 "lopdf rejects 19-byte entries — if this ever passes, drop the repair"
1964 );
1965
1966 let cells = |b: &[u8]| -> Vec<String> {
1967 super::pdf_textlines(b)
1968 .into_iter()
1969 .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
1970 .collect()
1971 };
1972 let from_good = cells(&good);
1973 assert!(
1974 from_good.iter().any(|t| t.contains("922769430725")),
1975 "control text: {from_good:?}"
1976 );
1977 assert_eq!(
1978 cells(&broken),
1979 from_good,
1980 "repair must match the good parse"
1981 );
1982 }
1983
1984 #[test]
1989 fn overstated_stream_length_still_yields_content() {
1990 let good = pdf_with_xref(true);
1991 let broken = {
1994 let at = good
1995 .windows(8)
1996 .position(|w| w == b"/Length ")
1997 .expect("a /Length")
1998 + 8;
1999 let digits = good[at..].iter().take_while(|c| c.is_ascii_digit()).count();
2000 let n: usize = std::str::from_utf8(&good[at..at + digits])
2001 .unwrap()
2002 .parse()
2003 .unwrap();
2004 let inflated = (n + 1).to_string();
2005 assert_eq!(inflated.len(), digits, "keep the digit count");
2006 let mut b = good.clone();
2007 b[at..at + digits].copy_from_slice(inflated.as_bytes());
2008 b
2009 };
2010 assert_eq!(broken.len(), good.len(), "the defect must not move bytes");
2011 let raw = lopdf::Document::load_mem(&broken).expect("still loads");
2013 assert!(
2014 raw.get_pages()
2015 .into_values()
2016 .all(|p| raw.get_page_content(p).is_empty()),
2017 "lopdf should drop the stream — if it stops, drop this repair"
2018 );
2019 let text = |b: &[u8]| -> Vec<String> {
2021 super::pdf_textlines(b)
2022 .into_iter()
2023 .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2024 .collect()
2025 };
2026 let expected = text(&good);
2027 assert!(!expected.is_empty(), "control must produce text");
2028 assert_eq!(text(&broken), expected);
2029 }
2030
2031 #[test]
2035 fn repair_declines_when_padding_would_move_objects() {
2036 let mut incremental = pdf_with_xref(false);
2037 incremental.extend_from_slice(b"6 0 obj<</Type/Whatever>>endobj\n");
2038 let declined = super::pad_short_xref_entries(&incremental).unwrap_err();
2039 assert!(
2040 declined.contains("object follows the xref"),
2041 "reason: {declined}"
2042 );
2043 }
2044}
2045
2046#[cfg(test)]
2052mod base14_fonts {
2053 fn pdf_with_font(fontdict: &[u8], text: &[u8]) -> Vec<u8> {
2055 let content = [b"BT /F1 12 Tf 72 700 Td (".as_slice(), text, b") Tj ET\n"].concat();
2056 let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2057 let objs: Vec<Vec<u8>> = vec![
2058 b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2059 b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2060 b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2061 /Resources<</Font<</F1 5 0 R>>>>>>"
2062 .to_vec(),
2063 [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
2064 fontdict.to_vec(),
2065 ];
2066 let mut out = b"%PDF-1.4\n".to_vec();
2067 let mut offsets = Vec::new();
2068 for (i, body) in objs.iter().enumerate() {
2069 offsets.push(out.len());
2070 out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2071 out.extend_from_slice(body);
2072 out.extend_from_slice(b"endobj\n");
2073 }
2074 let xref_at = out.len();
2075 out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2076 out.extend_from_slice(b"0000000000 65535 f \n");
2077 for off in &offsets {
2078 out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2079 }
2080 out.extend_from_slice(
2081 format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2082 );
2083 out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2084 out
2085 }
2086
2087 fn cells(pdf: &[u8]) -> Vec<crate::pdfium_backend::TextCell> {
2089 super::pdf_textlines(pdf)
2090 .into_iter()
2091 .flat_map(|(_, _, c)| c)
2092 .collect()
2093 }
2094
2095 #[test]
2097 fn standard14_faces_get_builtin_widths() {
2098 for fontdict in [
2099 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>".as_slice(),
2101 b"<</Type/Font/Subtype/Type1/BaseFont/Times-BoldItalic>>",
2103 b"<</Type/Font/Subtype/TrueType/BaseFont/Arial,Bold>>",
2105 b"<</Type/Font/Subtype/Type1/BaseFont/ABCDEF+Courier-Oblique>>",
2106 ] {
2107 let pdf = pdf_with_font(fontdict, b"Words have width now");
2108 let cs = cells(&pdf);
2109 let text: String = cs
2110 .iter()
2111 .map(|c| c.text.as_str())
2112 .collect::<Vec<_>>()
2113 .join(" ");
2114 assert!(
2115 text.contains("Words have width now"),
2116 "{}: text lost: {text:?}",
2117 String::from_utf8_lossy(fontdict)
2118 );
2119 assert!(
2120 cs.iter().all(|c| c.r > c.l),
2121 "{}: zero-width cells: {cs:?}",
2122 String::from_utf8_lossy(fontdict)
2123 );
2124 }
2125 }
2126
2127 #[test]
2130 fn explicit_widths_win_and_unknown_faces_are_untouched() {
2131 let explicit = pdf_with_font(
2135 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/FirstChar 65\
2136 /Widths[100 100 100 100]/Encoding/WinAnsiEncoding>>",
2137 b"ABBA",
2138 );
2139 let builtin = pdf_with_font(
2140 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>",
2141 b"ABBA",
2142 );
2143 let w = |pdf: &[u8]| {
2144 let cs = cells(pdf);
2145 assert_eq!(cs.len(), 1, "one word cell: {cs:?}");
2146 cs[0].r - cs[0].l
2147 };
2148 let (we, wb) = (w(&explicit), w(&builtin));
2149 assert!(
2150 (we - 4.8).abs() < 0.1,
2151 "explicit widths must win: got {we}, want 4×100×12/1000"
2152 );
2153 assert!(
2154 wb > 2.0 * we,
2155 "built-in Helvetica is much wider: {wb} vs {we}"
2156 );
2157
2158 let unknown = pdf_with_font(
2162 b"<</Type/Font/Subtype/Type1/BaseFont/FancyCorp-Display>>",
2163 b"Mystery",
2164 );
2165 let cs = cells(&unknown);
2166 let text: String = cs.iter().map(|c| c.text.as_str()).collect();
2167 assert!(text.contains("Mystery"), "text still decodes: {cs:?}");
2168 }
2169}
2170
2171#[cfg(test)]
2172mod overpainted {
2173 use crate::pdfium_backend::TextCell;
2174
2175 fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
2176 TextCell {
2177 text: text.into(),
2178 l,
2179 t,
2180 r,
2181 b,
2182 }
2183 }
2184
2185 #[test]
2189 fn stacked_logo_glyphs_are_dropped() {
2190 let mut cells = vec![
2191 cell("\"", 72.7, 21.5, 86.4, 31.5),
2192 cell("==", 59.4, 21.5, 99.6, 31.5),
2193 cell("Herr", 65.2, 151.3, 81.7, 161.3),
2194 ];
2195 super::drop_overpainted_cells(&mut cells);
2196 assert_eq!(cells.len(), 1, "cells: {cells:?}");
2197 assert_eq!(cells[0].text, "Herr");
2198 }
2199
2200 #[test]
2204 fn prose_and_double_draw_are_kept() {
2205 let mut cells = vec![
2206 cell("Telefon", 354.3, 133.2, 381.5, 143.2),
2207 cell("0676/2000", 387.3, 133.2, 428.7, 143.2),
2208 cell("Bold", 100.0, 50.0, 130.0, 60.0),
2209 cell("Bold", 100.3, 50.0, 130.3, 60.0),
2210 ];
2211 super::drop_overpainted_cells(&mut cells);
2212 assert_eq!(cells.len(), 4);
2213 }
2214}
2215
2216#[cfg(test)]
2217mod vestigial_layer {
2218 use crate::pdfium_backend::{PdfPage, TextCell};
2219
2220 fn page_with(texts: &[&str]) -> PdfPage {
2221 let cells = texts
2222 .iter()
2223 .enumerate()
2224 .map(|(i, t)| TextCell {
2225 text: t.to_string(),
2226 l: 10.0,
2227 t: 10.0 + 12.0 * i as f32,
2228 r: 90.0,
2229 b: 20.0 + 12.0 * i as f32,
2230 })
2231 .collect();
2232 PdfPage::from_cells(595.0, 842.0, 1.0, cells)
2233 }
2234
2235 #[test]
2240 fn typed_in_form_fields_are_not_a_text_layer() {
2241 let pages = vec![
2242 page_with(&["03", "05", "2025"]),
2243 page_with(&[]),
2244 page_with(&[]),
2245 ];
2246 assert!(super::text_layer_is_vestigial(&pages));
2247 assert!(super::text_layer_is_vestigial(&[page_with(&[])]));
2248 }
2249
2250 #[test]
2253 fn sparse_but_real_documents_pass() {
2254 let one_pager = vec![page_with(&[
2255 "Confidential briefing",
2256 "Prepared for the board meeting",
2257 "Do not distribute",
2258 ])];
2259 assert!(!super::text_layer_is_vestigial(&one_pager));
2260 }
2261}