1use std::collections::HashMap;
17use std::rc::Rc;
18
19use lopdf::{Dictionary, Document, Object};
20
21use crate::pdfium_backend::Glyph;
22
23#[derive(Default)]
32struct DocCaches {
33 fonts: HashMap<(lopdf::ObjectId, Vec<u8>), Rc<Font>>,
34 forms: HashMap<lopdf::ObjectId, Rc<lopdf::content::Content>>,
35}
36
37#[derive(Clone, Copy)]
39struct Mat {
40 a: f64,
41 b: f64,
42 c: f64,
43 d: f64,
44 e: f64,
45 f: f64,
46}
47
48impl Mat {
49 const ID: Mat = Mat {
50 a: 1.0,
51 b: 0.0,
52 c: 0.0,
53 d: 1.0,
54 e: 0.0,
55 f: 0.0,
56 };
57
58 fn then(self, m: Mat) -> Mat {
60 Mat {
61 a: self.a * m.a + self.b * m.c,
62 b: self.a * m.b + self.b * m.d,
63 c: self.c * m.a + self.d * m.c,
64 d: self.c * m.b + self.d * m.d,
65 e: self.e * m.a + self.f * m.c + m.e,
66 f: self.e * m.b + self.f * m.d + m.f,
67 }
68 }
69
70 fn apply(self, x: f64, y: f64) -> (f64, f64) {
71 (
72 self.a * x + self.c * y + self.e,
73 self.b * x + self.d * y + self.f,
74 )
75 }
76}
77
78struct Font {
80 two_byte: bool,
82 to_unicode: HashMap<u32, String>,
84 widths: HashMap<u32, f64>,
86 default_width: f64,
87 simple_encoding: Option<HashMap<u8, char>>,
89 fallback_names: HashMap<u8, String>,
93 program_encoding: HashMap<u8, char>,
97 ascent: f64,
98 descent: f64,
99 hash: u64,
100}
101
102impl Font {
103 fn decode_code(&self, code: u32) -> (Option<String>, f64) {
104 let w = self
105 .widths
106 .get(&code)
107 .copied()
108 .unwrap_or(self.default_width);
109 if let Some(s) = self.to_unicode.get(&code) {
110 return (Some(decompose_ligatures(s)), w);
111 }
112 if !self.two_byte {
113 if let Some(name) = self.fallback_names.get(&(code as u8)) {
116 return (Some(format!("/{name}")), w);
117 }
118 if let Some(enc) = &self.simple_encoding {
119 if let Some(&ch) = enc.get(&(code as u8)) {
120 return (Some(decompose_ligatures(&ch.to_string())), w);
121 }
122 }
123 if let Some(&ch) = self.program_encoding.get(&(code as u8)) {
131 return (Some(decompose_ligatures(&ch.to_string())), w);
132 }
133 }
134 (None, w)
135 }
136}
137
138fn decompose_ligatures(s: &str) -> String {
142 if !s.chars().any(|c| ('\u{FB00}'..='\u{FB06}').contains(&c)) {
143 return s.to_string();
144 }
145 s.chars()
146 .map(|c| {
147 match c {
148 '\u{FB00}' => "ff",
149 '\u{FB01}' => "fi",
150 '\u{FB02}' => "fl",
151 '\u{FB03}' => "ffi",
152 '\u{FB04}' => "ffl",
153 '\u{FB05}' => "ft",
154 '\u{FB06}' => "st",
155 _ => return c.to_string(),
156 }
157 .to_string()
158 })
159 .collect()
160}
161
162fn hash_name(name: &[u8]) -> u64 {
163 use std::hash::{Hash, Hasher};
164 let mut h = std::collections::hash_map::DefaultHasher::new();
165 name.hash(&mut h);
166 h.finish()
167}
168
169fn as_dict<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Dictionary> {
171 match obj {
172 Object::Dictionary(d) => Some(d),
173 Object::Reference(id) => doc.get_object(*id).ok().and_then(|o| o.as_dict().ok()),
174 _ => None,
175 }
176}
177
178fn deref<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Object> {
179 match obj {
180 Object::Reference(id) => doc.get_object(*id).ok(),
181 other => Some(other),
182 }
183}
184
185fn parse_font(doc: &Document, name: &[u8], fdict: &Dictionary) -> Font {
187 let subtype: &[u8] = fdict
188 .get(b"Subtype")
189 .ok()
190 .and_then(|o| o.as_name().ok())
191 .unwrap_or(&[]);
192 let two_byte = subtype == b"Type0".as_slice();
193
194 let to_unicode = fdict
195 .get(b"ToUnicode")
196 .ok()
197 .and_then(|o| deref(doc, o))
198 .and_then(|o| o.as_stream().ok())
199 .and_then(|s| s.decompressed_content().ok())
200 .map(|data| parse_tounicode(&data))
201 .unwrap_or_default();
202
203 let (mut widths, mut default_width) = if two_byte {
204 cid_widths(doc, fdict)
205 } else {
206 simple_widths(doc, fdict)
207 };
208
209 let simple_encoding = if two_byte {
210 None
211 } else {
212 Some(simple_encoding_table(doc, fdict))
213 };
214
215 if !two_byte && widths.is_empty() && default_width == 0.0 {
223 if let Some(std14) = base_font_name(fdict).and_then(|n| crate::std14::widths_for(&n)) {
224 if let Some(enc) = &simple_encoding {
225 for (&code, &ch) in enc {
226 if let Some(w) = std14.width(ch) {
227 widths.insert(u32::from(code), w);
228 }
229 }
230 }
231 default_width = 500.0;
234 }
235 }
236 let fallback_names = if two_byte {
237 HashMap::new()
238 } else {
239 differences_gid_names(doc, fdict)
240 };
241 let program_encoding = if two_byte {
242 HashMap::new()
243 } else {
244 type1_program_encoding(doc, fdict)
245 };
246
247 let (ascent, descent) = font_ascent_descent(doc, fdict, two_byte);
248
249 Font {
250 two_byte,
251 to_unicode,
252 widths,
253 default_width,
254 simple_encoding,
255 fallback_names,
256 program_encoding,
257 ascent,
258 descent,
259 hash: hash_name(name),
260 }
261}
262
263fn differences_gid_names(doc: &Document, fdict: &Dictionary) -> HashMap<u8, String> {
270 let mut map = HashMap::new();
271 let Some(Object::Dictionary(enc)) = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o))
272 else {
273 return map;
274 };
275 let Some(Object::Array(diffs)) = enc.get(b"Differences").ok().and_then(|o| deref(doc, o))
276 else {
277 return map;
278 };
279 let mut code = 0u8;
280 for el in diffs {
281 match el {
282 Object::Integer(i) => code = *i as u8,
283 Object::Name(name) => {
284 if glyph_name_to_char(name).is_none() && is_gid_name(name) {
285 map.insert(code, String::from_utf8_lossy(name).into_owned());
286 }
287 code = code.wrapping_add(1);
288 }
289 _ => {}
290 }
291 }
292 map
293}
294
295fn type1_program_encoding(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
303 let mut map = HashMap::new();
304 let Some(desc) = fdict
305 .get(b"FontDescriptor")
306 .ok()
307 .and_then(|o| deref(doc, o))
308 .and_then(|o| o.as_dict().ok())
309 else {
310 return map;
311 };
312 let Some(data) = desc
313 .get(b"FontFile")
314 .ok()
315 .and_then(|o| deref(doc, o))
316 .and_then(|o| o.as_stream().ok())
317 .and_then(|s| s.decompressed_content().ok())
318 else {
319 return map;
320 };
321 let head_end = data
323 .windows(5)
324 .position(|w| w == b"eexec")
325 .unwrap_or(data.len());
326 let head = String::from_utf8_lossy(&data[..head_end]);
327 let toks: Vec<&str> = head.split_whitespace().collect();
329 for w in toks.windows(4) {
330 if w[0] == "dup" && w[3] == "put" {
331 if let (Ok(code), Some(name)) = (w[1].parse::<u32>(), w[2].strip_prefix('/')) {
332 if code <= 255 {
333 if let Some(ch) = glyph_name_to_char(name.as_bytes()) {
334 map.insert(code as u8, ch);
335 }
336 }
337 }
338 }
339 }
340 map
341}
342
343fn is_gid_name(name: &[u8]) -> bool {
349 let Ok(s) = std::str::from_utf8(name) else {
350 return false;
351 };
352 if s.starts_with("afii") || s.starts_with("uni") {
353 return false;
354 }
355 for prefix in ["g", "G", "cid", "CID", "glyph", "index"] {
356 if let Some(rest) = s.strip_prefix(prefix) {
357 if !rest.is_empty() && rest.bytes().all(|b| b.is_ascii_digit()) {
358 return true;
359 }
360 }
361 }
362 let alpha = s.bytes().take_while(|b| b.is_ascii_alphabetic()).count();
366 let digits = s.len() - alpha;
367 (1..=3).contains(&alpha)
368 && digits >= 3
369 && s.as_bytes()[alpha..].iter().all(|b| b.is_ascii_digit())
370}
371
372fn font_ascent_descent(doc: &Document, fdict: &Dictionary, two_byte: bool) -> (f64, f64) {
373 let descr_owner = if two_byte {
375 fdict
376 .get(b"DescendantFonts")
377 .ok()
378 .and_then(|o| deref(doc, o))
379 .and_then(|o| match o {
380 Object::Array(a) => a.first(),
381 _ => None,
382 })
383 .and_then(|o| as_dict(doc, o))
384 } else {
385 Some(fdict)
386 };
387 let fd = descr_owner
388 .and_then(|d| d.get(b"FontDescriptor").ok())
389 .and_then(|o| as_dict(doc, o));
390 let asc = fd
391 .and_then(|d| d.get(b"Ascent").ok())
392 .and_then(|o| {
393 o.as_float()
394 .ok()
395 .or_else(|| o.as_i64().ok().map(|i| i as f32))
396 })
397 .unwrap_or(750.0) as f64;
398 let desc = fd
399 .and_then(|d| d.get(b"Descent").ok())
400 .and_then(|o| {
401 o.as_float()
402 .ok()
403 .or_else(|| o.as_i64().ok().map(|i| i as f32))
404 })
405 .unwrap_or(-250.0) as f64;
406 if asc - desc <= 1.0 {
413 return (750.0, -250.0);
414 }
415 (asc, desc)
416}
417
418fn base_font_name(fdict: &Dictionary) -> Option<Vec<u8>> {
420 let name = fdict.get(b"BaseFont").ok()?.as_name().ok()?;
421 let stripped = match name.iter().position(|&b| b == b'+') {
422 Some(i) if i == 6 => &name[i + 1..],
423 _ => name,
424 };
425 Some(stripped.to_vec())
426}
427
428fn simple_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
430 let mut map = HashMap::new();
431 let first = fdict
432 .get(b"FirstChar")
433 .ok()
434 .and_then(|o| o.as_i64().ok())
435 .unwrap_or(0) as u32;
436 if let Some(Object::Array(arr)) = fdict.get(b"Widths").ok().and_then(|o| deref(doc, o)) {
437 for (i, w) in arr.iter().enumerate() {
438 if let Some(w) = num(w) {
439 map.insert(first + i as u32, w);
440 }
441 }
442 }
443 let dw = fdict
444 .get(b"FontDescriptor")
445 .ok()
446 .and_then(|o| as_dict(doc, o))
447 .and_then(|d| d.get(b"MissingWidth").ok())
448 .and_then(num)
449 .unwrap_or(0.0);
450 (map, dw)
451}
452
453fn cid_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
455 let mut map = HashMap::new();
456 let Some(desc) = fdict
457 .get(b"DescendantFonts")
458 .ok()
459 .and_then(|o| deref(doc, o))
460 .and_then(|o| match o {
461 Object::Array(a) => a.first(),
462 _ => None,
463 })
464 .and_then(|o| as_dict(doc, o))
465 else {
466 return (map, 1000.0);
467 };
468 let dw = desc.get(b"DW").ok().and_then(num).unwrap_or(1000.0);
469 if let Some(Object::Array(w)) = desc.get(b"W").ok().and_then(|o| deref(doc, o)) {
470 let mut i = 0;
471 while i < w.len() {
472 let c = w.get(i).and_then(num);
473 match (c, w.get(i + 1)) {
474 (Some(c), Some(Object::Array(list))) => {
476 for (k, wv) in list.iter().enumerate() {
477 if let Some(wv) = num(wv) {
478 map.insert(c as u32 + k as u32, wv);
479 }
480 }
481 i += 2;
482 }
483 (Some(c1), Some(o2)) => {
485 if let (Some(c2), Some(wv)) = (num(o2), w.get(i + 2).and_then(num)) {
486 for cid in c1 as u32..=c2 as u32 {
487 map.insert(cid, wv);
488 }
489 }
490 i += 3;
491 }
492 _ => break,
493 }
494 }
495 }
496 (map, dw)
497}
498
499fn num(o: &Object) -> Option<f64> {
500 match o {
501 Object::Integer(i) => Some(*i as f64),
502 Object::Real(r) => Some(*r as f64),
503 _ => None,
504 }
505}
506
507fn parse_tounicode(data: &[u8]) -> HashMap<u32, String> {
509 let text = String::from_utf8_lossy(data);
510 let mut map = HashMap::new();
511 let hex = |s: &str| -> Option<Vec<u16>> {
512 let s = s.trim();
513 if !s.starts_with('<') || !s.ends_with('>') {
514 return None;
515 }
516 let h = &s[1..s.len() - 1];
517 let bytes: Vec<u8> = (0..h.len())
518 .step_by(2)
519 .filter_map(|i| u8::from_str_radix(h.get(i..i + 2)?, 16).ok())
520 .collect();
521 Some(
522 bytes
523 .chunks(2)
524 .map(|c| {
525 if c.len() == 2 {
526 u16::from_be_bytes([c[0], c[1]])
527 } else {
528 c[0] as u16
529 }
530 })
531 .collect(),
532 )
533 };
534 let u16s_to_string = |u: &[u16]| String::from_utf16_lossy(u);
535 let code_of = |u: &[u16]| u.iter().fold(0u32, |acc, &x| (acc << 16) | x as u32);
536
537 let tokens: Vec<String> = {
541 let bytes = text.as_bytes();
542 let mut toks = Vec::new();
543 let mut i = 0;
544 while i < bytes.len() {
545 let c = bytes[i];
546 if c.is_ascii_whitespace() {
547 i += 1;
548 } else if c == b'<' {
549 let start = i;
550 while i < bytes.len() && bytes[i] != b'>' {
551 i += 1;
552 }
553 i += 1; toks.push(String::from_utf8_lossy(&bytes[start..i.min(bytes.len())]).into_owned());
555 } else if c == b'[' || c == b']' {
556 toks.push((c as char).to_string());
557 i += 1;
558 } else {
559 let start = i;
560 while i < bytes.len()
561 && !bytes[i].is_ascii_whitespace()
562 && bytes[i] != b'<'
563 && bytes[i] != b'['
564 && bytes[i] != b']'
565 {
566 i += 1;
567 }
568 toks.push(String::from_utf8_lossy(&bytes[start..i]).into_owned());
569 }
570 }
571 toks
572 };
573 let tokens: Vec<&str> = tokens.iter().map(|s| s.as_str()).collect();
574 let mut i = 0;
575 while i < tokens.len() {
576 match tokens[i] {
577 "beginbfchar" => {
578 i += 1;
579 while i + 1 < tokens.len() && tokens[i] != "endbfchar" {
580 if let (Some(src), Some(dst)) = (hex(tokens[i]), hex(tokens[i + 1])) {
581 map.insert(code_of(&src), u16s_to_string(&dst));
582 }
583 i += 2;
584 }
585 }
586 "beginbfrange" => {
587 i += 1;
588 while i + 2 < tokens.len() && tokens[i] != "endbfrange" {
589 let (Some(lo), Some(hi)) = (hex(tokens[i]), hex(tokens[i + 1])) else {
590 i += 1;
591 continue;
592 };
593 let lo = code_of(&lo);
594 let hi = code_of(&hi);
595 if tokens[i + 2] == "[" {
596 let mut j = i + 3;
598 let mut code = lo;
599 while j < tokens.len() && tokens[j] != "]" {
600 if let Some(dst) = hex(tokens[j]) {
601 map.insert(code, u16s_to_string(&dst));
602 }
603 code += 1;
604 j += 1;
605 }
606 i = j + 1;
607 } else if let Some(dst) = hex(tokens[i + 2]) {
608 let base = code_of(&dst);
610 for (k, code) in (lo..=hi).enumerate() {
611 if let Some(ch) = char::from_u32(base + k as u32) {
612 map.insert(code, ch.to_string());
613 }
614 }
615 i += 3;
616 } else {
617 i += 1;
618 }
619 }
620 }
621 _ => i += 1,
622 }
623 }
624 map
625}
626
627fn codes(font: &Font, bytes: &[u8]) -> Vec<u32> {
629 if font.two_byte {
630 bytes
631 .chunks(2)
632 .map(|c| {
633 if c.len() == 2 {
634 ((c[0] as u32) << 8) | c[1] as u32
635 } else {
636 c[0] as u32
637 }
638 })
639 .collect()
640 } else {
641 bytes.iter().map(|&b| b as u32).collect()
642 }
643}
644
645fn page_size(doc: &Document, page_id: lopdf::ObjectId) -> (f32, f32) {
647 let mb = doc
648 .get_object(page_id)
649 .ok()
650 .and_then(|o| o.as_dict().ok())
651 .and_then(|d| {
652 d.get(b"MediaBox").ok().cloned()
654 })
655 .or_else(|| {
656 doc.get_dictionary(page_id)
657 .ok()
658 .and_then(|d| d.get(b"MediaBox").ok().cloned())
659 });
660 if let Some(Object::Array(a)) = mb {
661 let v: Vec<f32> = a.iter().filter_map(|o| num(o).map(|x| x as f32)).collect();
662 if v.len() == 4 {
663 return ((v[2] - v[0]).abs(), (v[3] - v[1]).abs());
664 }
665 }
666 (612.0, 792.0)
667}
668
669pub fn content_diagnosis(bytes: &[u8]) -> String {
675 let Some(doc) = load_document(bytes) else {
676 return "document does not load".into();
677 };
678 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
679 pages.sort_by_key(|(n, _)| *n);
680 let mut out = String::new();
681 let mut caches = DocCaches::default();
682 for (n, pid) in pages.into_iter().take(4) {
683 let content_bytes = doc.get_page_content(pid);
684 let ops = lopdf::content::Content::decode(&content_bytes)
685 .map(|c| c.operations.len())
686 .ok();
687 let res = page_res(&doc, pid);
688 let fonts = res.map(|r| fonts_from_res(&doc, r, &mut caches).len());
689 let glyphs = page_glyphs_cached(&doc, pid, &mut caches).len();
690 out.push_str(&format!(
691 "\n page {n}: content {} B, ops {}, resources {}, fonts {}, glyphs {}",
692 content_bytes.len(),
693 ops.map_or("UNDECODABLE".to_string(), |n| n.to_string()),
694 if res.is_some() { "ok" } else { "MISSING" },
695 fonts.map_or("-".to_string(), |n| n.to_string()),
696 glyphs,
697 ));
698 }
699 out
700}
701
702pub fn text_layer_is_vestigial(pages: &[crate::pdfium_backend::PdfPage]) -> bool {
715 let lines: usize = pages.iter().map(|p| p.cells.len()).sum();
716 if lines == 0 {
717 return true;
718 }
719 let chars: usize = pages
720 .iter()
721 .flat_map(|p| &p.cells)
722 .map(|c| c.text.chars().count())
723 .sum();
724 lines <= pages.len() && chars < 32
725}
726
727pub fn xref_repair_status(bytes: &[u8]) -> String {
731 if Document::load_mem(bytes).is_ok() {
732 return "loads unaided; no repair needed".into();
733 }
734 match pad_short_xref_entries(bytes) {
735 Ok(fixed) => match Document::load_mem(&fixed) {
736 Ok(_) => "repaired: cross-reference entries padded to 20 bytes".into(),
737 Err(e) => format!("padded the entries, but it still will not load: {e}"),
738 },
739 Err(why) => format!("repair declined — {why}"),
740 }
741}
742
743fn load_document(bytes: &[u8]) -> Option<Document> {
758 let mut fallback = None;
763 if let Some(doc) = best_effort_load(bytes, &mut fallback) {
764 return Some(doc);
765 }
766 let xref_fixed = pad_short_xref_entries(bytes).ok();
767 if let Some(fixed) = &xref_fixed {
768 if let Some(doc) = best_effort_load(fixed, &mut fallback) {
769 return Some(doc);
770 }
771 }
772 let lengths_fixed = fix_stream_lengths(xref_fixed.as_deref().unwrap_or(bytes));
775 if let Some(doc) = best_effort_load(&lengths_fixed, &mut fallback) {
776 return Some(doc);
777 }
778 fallback
779}
780
781fn best_effort_load(data: &[u8], fallback: &mut Option<Document>) -> Option<Document> {
784 match Document::load_mem(data) {
785 Ok(doc) if has_page_content(&doc) => Some(doc),
786 Ok(doc) => {
787 fallback.get_or_insert(doc);
788 None
789 }
790 Err(_) => None,
791 }
792}
793
794fn has_page_content(doc: &Document) -> bool {
798 doc.get_pages()
799 .into_values()
800 .take(4)
801 .any(|pid| !doc.get_page_content(pid).is_empty())
802}
803
804fn fix_stream_lengths(bytes: &[u8]) -> Vec<u8> {
817 let mut out = bytes.to_vec();
818 let mut i = 0;
819 while let Some(rel) = find(&out[i..], b"stream") {
820 let kw = i + rel;
821 i = kw + 6;
822 if kw >= 3 && &out[kw - 3..kw] == b"end" {
824 continue;
825 }
826 let mut data = kw + 6;
828 if out.get(data..data + 2) == Some(b"\r\n".as_slice()) {
829 data += 2;
830 } else if matches!(out.get(data), Some(b'\n' | b'\r')) {
831 data += 1;
832 }
833 let Some(end) = find(&out[data..], b"endstream").map(|r| data + r) else {
834 continue;
835 };
836 let dict_start = out[..kw].iter().rposition(|&c| c == b'<').unwrap_or(0);
838 let Some(lrel) = find(&out[dict_start..kw], b"/Length") else {
839 continue;
840 };
841 let mut d = dict_start + lrel + 7;
842 while matches!(out.get(d), Some(b' ')) {
843 d += 1;
844 }
845 let digits = out[d..].iter().take_while(|c| c.is_ascii_digit()).count();
846 if digits == 0 {
847 continue;
848 }
849 let declared: usize = match std::str::from_utf8(&out[d..d + digits])
850 .ok()
851 .and_then(|s| s.parse().ok())
852 {
853 Some(v) => v,
854 None => continue,
855 };
856 let actual = end - data;
857 let replacement = actual.to_string();
860 if actual == declared || replacement.len() > digits {
861 continue;
862 }
863 out[d..d + digits].fill(b' ');
864 out[d..d + replacement.len()].copy_from_slice(replacement.as_bytes());
865 }
866 out
867}
868
869fn find(haystack: &[u8], needle: &[u8]) -> Option<usize> {
870 haystack.windows(needle.len()).position(|w| w == needle)
871}
872
873fn pad_short_xref_entries(bytes: &[u8]) -> Result<Vec<u8>, &'static str> {
876 let is_boundary = |i: usize| i == 0 || matches!(bytes[i - 1], b'\n' | b'\r');
879 let mut starts = (0..bytes.len().saturating_sub(4))
880 .filter(|&i| &bytes[i..i + 4] == b"xref" && is_boundary(i));
881 let xref_at = starts
882 .next()
883 .ok_or("no classic `xref` section (an xref stream?)")?;
884 if starts.next().is_some() {
885 return Err("more than one xref section (incremental update)");
886 }
887 let last_obj = bytes
888 .windows(3)
889 .rposition(|w| w == b"obj")
890 .ok_or("no objects found")?;
891 if last_obj > xref_at {
892 return Err("an object follows the xref — padding would move it");
893 }
894
895 let mut out = bytes[..xref_at].to_vec();
896 out.extend_from_slice(b"xref\n");
897 let mut i = xref_at + 4;
898 let skip_ws = |i: &mut usize| {
899 while matches!(bytes.get(*i), Some(b'\r' | b'\n' | b' ')) {
900 *i += 1;
901 }
902 };
903 loop {
904 skip_ws(&mut i);
905 if bytes[i..].starts_with(b"trailer") {
907 out.extend_from_slice(&bytes[i..]);
908 return Ok(out);
909 }
910 let header_end = i + bytes[i..]
911 .iter()
912 .position(|c| matches!(c, b'\n' | b'\r'))
913 .ok_or("subsection header runs off the end")?;
914 let header = std::str::from_utf8(&bytes[i..header_end])
915 .map_err(|_| "subsection header is not text")?
916 .trim();
917 let mut parts = header.split_whitespace();
918 let count: usize = parts
919 .nth(1)
920 .and_then(|c| c.parse().ok())
921 .ok_or("unparseable subsection header")?;
922 if parts.next().is_some() || count == 0 {
923 return Err("unexpected subsection header shape");
924 }
925 out.extend_from_slice(header.as_bytes());
926 out.push(b'\n');
927 i = header_end;
928 for _ in 0..count {
929 skip_ws(&mut i);
930 let entry = bytes.get(i..i + 18).ok_or("xref entry runs off the end")?;
932 let well_formed = entry[..10].iter().all(u8::is_ascii_digit)
933 && entry[10] == b' '
934 && entry[11..16].iter().all(u8::is_ascii_digit)
935 && entry[16] == b' '
936 && matches!(entry[17], b'n' | b'f');
937 if !well_formed {
938 return Err("xref entry is not `nnnnnnnnnn ggggg n`");
939 }
940 out.extend_from_slice(entry);
941 out.extend_from_slice(b" \n"); i += 18;
943 }
944 }
945}
946
947pub fn debug_glyphs(bytes: &[u8], index: usize) -> Vec<(char, f32, f32, f32, f32)> {
950 let Some(doc) = load_document(bytes) else {
951 return Vec::new();
952 };
953 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
954 pages.sort_by_key(|(n, _)| *n);
955 let Some((_, pid)) = pages.get(index) else {
956 return Vec::new();
957 };
958 page_glyphs(&doc, *pid)
959 .into_iter()
960 .map(|g| (g.ch, g.ll, g.lr, g.lb, g.lt))
961 .collect()
962}
963
964pub fn pdf_textlines(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
968 let Some(doc) = load_document(bytes) else {
969 return Vec::new();
970 };
971 let mut caches = DocCaches::default();
972 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
973 pages.sort_by_key(|(n, _)| *n);
974 pages
975 .into_iter()
976 .map(|(_, pid)| {
977 let (w, h) = page_size(&doc, pid);
978 let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
979 let cells = crate::dp_lines::line_cells(&glyphs, h, true);
980 (w, h, cells)
981 })
982 .collect()
983}
984
985pub fn pdf_words(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
990 let Some(doc) = load_document(bytes) else {
991 return Vec::new();
992 };
993 let mut caches = DocCaches::default();
994 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
995 pages.sort_by_key(|(n, _)| *n);
996 pages
997 .into_iter()
998 .map(|(_, pid)| {
999 let (w, h) = page_size(&doc, pid);
1000 let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1001 let cells = crate::dp_lines::word_cells(&glyphs, h, true);
1002 (w, h, cells)
1003 })
1004 .collect()
1005}
1006
1007#[derive(Default)]
1011pub struct PageParserCells {
1012 pub prose: Vec<crate::pdfium_backend::TextCell>,
1013 pub words: Vec<crate::pdfium_backend::TextCell>,
1014 pub code: Vec<crate::pdfium_backend::TextCell>,
1015}
1016
1017pub fn pdf_all_cells(bytes: &[u8]) -> Vec<PageParserCells> {
1022 let Some(doc) = load_document(bytes) else {
1023 return Vec::new();
1024 };
1025 let mut caches = DocCaches::default();
1026 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1027 pages.sort_by_key(|(n, _)| *n);
1028 pages
1029 .into_iter()
1030 .map(|(_, pid)| {
1031 let (_w, h) = page_size(&doc, pid);
1032 let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1033 let (prose, words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1034 PageParserCells {
1035 prose,
1036 words,
1037 code: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1038 }
1039 })
1040 .collect()
1041}
1042
1043pub fn pdf_text_pages(bytes: &[u8]) -> Vec<crate::pdfium_backend::PdfPage> {
1050 let Some(doc) = load_document(bytes) else {
1051 return Vec::new();
1052 };
1053 let mut caches = DocCaches::default();
1054 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1055 pages.sort_by_key(|(n, _)| *n);
1056 pages
1057 .into_iter()
1058 .map(|(_, pid)| {
1059 let (w, h) = page_size(&doc, pid);
1060 let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1061 let (mut prose, mut words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1062 drop_overpainted_cells(&mut prose);
1063 drop_overpainted_cells(&mut words);
1064 crate::pdfium_backend::PdfPage {
1065 #[cfg(feature = "ocr-prep")]
1066 image_layout: None,
1067 width: w,
1068 height: h,
1069 scale: 1.0,
1071 cells: prose,
1072 code_cells: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1073 word_cells: words,
1074 #[cfg(feature = "ocr-prep")]
1075 image: image::RgbImage::new(1, 1),
1076 links: Vec::new(),
1077 }
1078 })
1079 .collect()
1080}
1081
1082fn drop_overpainted_cells(cells: &mut Vec<crate::pdfium_backend::TextCell>) {
1101 let mut paint = vec![false; cells.len()];
1102 for i in 0..cells.len() {
1103 for j in 0..cells.len() {
1104 if i == j || cells[i].text == cells[j].text {
1105 continue;
1106 }
1107 let (a, b) = (&cells[i], &cells[j]);
1108 let vo = (a.b.min(b.b) - a.t.max(b.t)).max(0.0);
1110 if vo < 0.6 * (a.b - a.t).min(b.b - b.t) {
1111 continue;
1112 }
1113 let ho = (a.r.min(b.r) - a.l.max(b.l)).max(0.0);
1115 if ho >= 0.8 * (a.r - a.l) && (a.r - a.l) <= (b.r - b.l) {
1116 paint[i] = true;
1117 paint[j] = true;
1118 }
1119 }
1120 }
1121 let mut keep = paint.iter().map(|p| !p);
1122 cells.retain(|_| keep.next().unwrap());
1123}
1124
1125#[derive(Clone, Copy)]
1129struct TextState {
1130 tc: f64,
1131 tw: f64,
1132 th: f64,
1133 tl: f64,
1134 trise: f64,
1135 fsize: f64,
1136}
1137
1138impl TextState {
1139 const INIT: TextState = TextState {
1140 tc: 0.0,
1141 tw: 0.0,
1142 th: 1.0,
1143 tl: 0.0,
1144 trise: 0.0,
1145 fsize: 0.0,
1146 };
1147}
1148
1149fn page_res(doc: &Document, page_id: lopdf::ObjectId) -> Option<&Dictionary> {
1152 let (inline, ids) = doc.get_page_resources(page_id).ok()?;
1153 if let Some(d) = inline {
1154 return Some(d);
1155 }
1156 ids.into_iter().find_map(|id| doc.get_dictionary(id).ok())
1157}
1158
1159fn fonts_from_res(
1163 doc: &Document,
1164 res: &Dictionary,
1165 caches: &mut DocCaches,
1166) -> HashMap<Vec<u8>, Rc<Font>> {
1167 let mut map = HashMap::new();
1168 let font_dict = res
1169 .get(b"Font")
1170 .ok()
1171 .and_then(|o| deref(doc, o))
1172 .and_then(|o| o.as_dict().ok());
1173 if let Some(fd) = font_dict {
1174 for (name, value) in fd.iter() {
1175 let font = match value {
1176 Object::Reference(id) => {
1177 let key = (*id, name.clone());
1178 if let Some(f) = caches.fonts.get(&key) {
1179 Rc::clone(f)
1180 } else if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1181 let f = Rc::new(parse_font(doc, name, fdict));
1182 caches.fonts.insert(key, Rc::clone(&f));
1183 f
1184 } else {
1185 continue;
1186 }
1187 }
1188 _ => {
1189 if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1190 Rc::new(parse_font(doc, name, fdict))
1191 } else {
1192 continue;
1193 }
1194 }
1195 };
1196 map.insert(name.clone(), font);
1197 }
1198 }
1199 map
1200}
1201
1202pub(crate) fn page_glyphs(doc: &Document, page_id: lopdf::ObjectId) -> Vec<Glyph> {
1204 page_glyphs_cached(doc, page_id, &mut DocCaches::default())
1205}
1206
1207fn page_glyphs_cached(
1210 doc: &Document,
1211 page_id: lopdf::ObjectId,
1212 caches: &mut DocCaches,
1213) -> Vec<Glyph> {
1214 let mut out = Vec::new();
1215 let content_bytes = doc.get_page_content(page_id);
1218 let Ok(content) = lopdf::content::Content::decode(&content_bytes) else {
1219 return out;
1220 };
1221 if let Some(res) = page_res(doc, page_id) {
1222 run_content(
1223 doc,
1224 res,
1225 &content,
1226 Mat::ID,
1227 TextState::INIT,
1228 0,
1229 caches,
1230 &mut out,
1231 );
1232 }
1233 out
1234}
1235
1236#[allow(clippy::too_many_arguments)]
1241fn run_content(
1242 doc: &Document,
1243 res: &Dictionary,
1244 content: &lopdf::content::Content,
1245 base_ctm: Mat,
1246 init: TextState,
1247 depth: u32,
1248 caches: &mut DocCaches,
1249 out: &mut Vec<Glyph>,
1250) {
1251 let fonts = fonts_from_res(doc, res, caches);
1252 let xobjects = res
1253 .get(b"XObject")
1254 .ok()
1255 .and_then(|o| deref(doc, o))
1256 .and_then(|o| o.as_dict().ok());
1257
1258 #[allow(clippy::type_complexity)]
1263 let mut gstate_stack: Vec<(Mat, f64, f64, f64, f64, f64, f64, Option<&Rc<Font>>)> = Vec::new();
1264 let mut ctm = base_ctm;
1265 let mut tm = Mat::ID;
1266 let mut tlm = Mat::ID;
1267 let mut font: Option<&Rc<Font>> = None;
1268 let mut fsize = init.fsize;
1269 let mut tc = init.tc; let mut tw = init.tw; let mut th = init.th; let mut tl = init.tl; let mut trise = init.trise;
1274
1275 let op_f = |operands: &[Object], i: usize| operands.get(i).and_then(num).unwrap_or(0.0);
1276
1277 for op in &content.operations {
1278 let operands = &op.operands;
1279 match op.operator.as_str() {
1280 "q" => gstate_stack.push((ctm, tc, tw, th, tl, trise, fsize, font)),
1281 "Q" => {
1282 if let Some((c, a, b, h, l, r, fs, f)) = gstate_stack.pop() {
1283 ctm = c;
1284 tc = a;
1285 tw = b;
1286 th = h;
1287 tl = l;
1288 trise = r;
1289 fsize = fs;
1290 font = f;
1291 }
1292 }
1293 "cm" => {
1294 let m = Mat {
1295 a: op_f(operands, 0),
1296 b: op_f(operands, 1),
1297 c: op_f(operands, 2),
1298 d: op_f(operands, 3),
1299 e: op_f(operands, 4),
1300 f: op_f(operands, 5),
1301 };
1302 ctm = m.then(ctm);
1303 }
1304 "BT" => {
1305 tm = Mat::ID;
1306 tlm = Mat::ID;
1307 }
1308 "ET" => {}
1309 "Tf" => {
1310 if let Some(Object::Name(n)) = operands.first() {
1311 font = fonts.get(n.as_slice());
1312 }
1313 fsize = op_f(operands, 1);
1314 }
1315 "Td" => {
1316 tlm = Mat {
1317 a: 1.0,
1318 b: 0.0,
1319 c: 0.0,
1320 d: 1.0,
1321 e: op_f(operands, 0),
1322 f: op_f(operands, 1),
1323 }
1324 .then(tlm);
1325 tm = tlm;
1326 }
1327 "TD" => {
1328 tl = -op_f(operands, 1);
1329 tlm = Mat {
1330 a: 1.0,
1331 b: 0.0,
1332 c: 0.0,
1333 d: 1.0,
1334 e: op_f(operands, 0),
1335 f: op_f(operands, 1),
1336 }
1337 .then(tlm);
1338 tm = tlm;
1339 }
1340 "Tm" => {
1341 tlm = Mat {
1342 a: op_f(operands, 0),
1343 b: op_f(operands, 1),
1344 c: op_f(operands, 2),
1345 d: op_f(operands, 3),
1346 e: op_f(operands, 4),
1347 f: op_f(operands, 5),
1348 };
1349 tm = tlm;
1350 }
1351 "T*" => {
1352 tlm = Mat {
1353 a: 1.0,
1354 b: 0.0,
1355 c: 0.0,
1356 d: 1.0,
1357 e: 0.0,
1358 f: -tl,
1359 }
1360 .then(tlm);
1361 tm = tlm;
1362 }
1363 "Tc" => tc = op_f(operands, 0),
1364 "Tw" => tw = op_f(operands, 0),
1365 "Tz" => th = op_f(operands, 0) / 100.0,
1366 "TL" => tl = op_f(operands, 0),
1367 "Ts" => trise = op_f(operands, 0),
1368 "Tj" | "'" | "\"" => {
1369 if op.operator == "'" || op.operator == "\"" {
1370 tlm = Mat {
1372 a: 1.0,
1373 b: 0.0,
1374 c: 0.0,
1375 d: 1.0,
1376 e: 0.0,
1377 f: -tl,
1378 }
1379 .then(tlm);
1380 tm = tlm;
1381 }
1382 if op.operator == "\"" {
1383 tw = op_f(operands, 0);
1386 tc = op_f(operands, 1);
1387 }
1388 if let (Some(f), Some(Object::String(s, _))) = (font, operands.last()) {
1389 show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out);
1390 }
1391 }
1392 "TJ" => {
1393 if let (Some(f), Some(Object::Array(arr))) = (font, operands.first()) {
1394 for el in arr {
1395 match el {
1396 Object::String(s, _) => {
1397 show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out)
1398 }
1399 other => {
1400 if let Some(adj) = num(other) {
1401 let tx = -adj / 1000.0 * fsize * th;
1403 tm = Mat {
1404 a: 1.0,
1405 b: 0.0,
1406 c: 0.0,
1407 d: 1.0,
1408 e: tx,
1409 f: 0.0,
1410 }
1411 .then(tm);
1412 }
1413 }
1414 }
1415 }
1416 }
1417 }
1418 "Do" => {
1419 if depth >= 8 {
1422 continue;
1423 }
1424 let Some(Object::Name(n)) = operands.first() else {
1425 continue;
1426 };
1427 let obj = xobjects.and_then(|d| d.get(n.as_slice()).ok());
1428 let form_id = match obj {
1429 Some(Object::Reference(id)) => Some(*id),
1430 _ => None,
1431 };
1432 let stream = obj
1433 .and_then(|o| deref(doc, o))
1434 .and_then(|o| o.as_stream().ok());
1435 let Some(stream) = stream else { continue };
1436 let is_form = stream
1437 .dict
1438 .get(b"Subtype")
1439 .ok()
1440 .and_then(|o| o.as_name().ok())
1441 == Some(b"Form".as_slice());
1442 if !is_form {
1443 continue;
1444 }
1445 let cached = form_id.and_then(|id| caches.forms.get(&id).cloned());
1448 let form_content = match cached {
1449 Some(c) => c,
1450 None => {
1451 let Ok(data) = stream.decompressed_content() else {
1452 continue;
1453 };
1454 let Ok(c) = lopdf::content::Content::decode(&data) else {
1455 continue;
1456 };
1457 let c = Rc::new(c);
1458 if let Some(id) = form_id {
1459 caches.forms.insert(id, Rc::clone(&c));
1460 }
1461 c
1462 }
1463 };
1464 let form_mat = match stream.dict.get(b"Matrix").ok() {
1466 Some(Object::Array(a)) if a.len() == 6 => {
1467 let v: Vec<f64> = a.iter().filter_map(num).collect();
1468 if v.len() == 6 {
1469 Mat {
1470 a: v[0],
1471 b: v[1],
1472 c: v[2],
1473 d: v[3],
1474 e: v[4],
1475 f: v[5],
1476 }
1477 } else {
1478 Mat::ID
1479 }
1480 }
1481 _ => Mat::ID,
1482 };
1483 let form_res = stream
1485 .dict
1486 .get(b"Resources")
1487 .ok()
1488 .and_then(|o| deref(doc, o))
1489 .and_then(|o| o.as_dict().ok())
1490 .unwrap_or(res);
1491 let state = TextState {
1492 tc,
1493 tw,
1494 th,
1495 tl,
1496 trise,
1497 fsize,
1498 };
1499 run_content(
1500 doc,
1501 form_res,
1502 &form_content,
1503 form_mat.then(ctm),
1504 state,
1505 depth + 1,
1506 caches,
1507 out,
1508 );
1509 }
1510 _ => {}
1511 }
1512 }
1513}
1514
1515#[allow(clippy::too_many_arguments)]
1516fn show_text(
1517 font: &Font,
1518 bytes: &[u8],
1519 fsize: f64,
1520 tc: f64,
1521 tw: f64,
1522 th: f64,
1523 trise: f64,
1524 tm: &mut Mat,
1525 ctm: Mat,
1526 out: &mut Vec<Glyph>,
1527) {
1528 for code in codes(font, bytes) {
1529 let (text, w) = font.decode_code(code);
1530 let w0 = w / 1000.0; let scale = Mat {
1533 a: fsize * th,
1534 b: 0.0,
1535 c: 0.0,
1536 d: fsize,
1537 e: 0.0,
1538 f: trise,
1539 };
1540 let trm = scale.then(*tm).then(ctm);
1541 let (x0, y0) = trm.apply(0.0, font.descent / 1000.0);
1543 let (x1, _y1) = trm.apply(w0, font.descent / 1000.0);
1544 let (_x2, y2) = trm.apply(0.0, font.ascent / 1000.0);
1545 let (left, right) = (x0.min(x1), x0.max(x1));
1546 let (bot, top) = (y0.min(y2), y0.max(y2));
1547 if let Some(s) = text {
1548 for ch in s.chars() {
1550 if ch != '\u{0}' {
1551 out.push(Glyph {
1552 ch,
1553 l: left as f32,
1554 b: bot as f32,
1555 r: right as f32,
1556 t: top as f32,
1557 ll: left as f32,
1558 lb: bot as f32,
1559 lr: right as f32,
1560 lt: top as f32,
1561 font: font.hash,
1562 });
1563 }
1564 }
1565 }
1566 let is_space = !font.two_byte && code == 32;
1568 let tx = (w0 * fsize + tc + if is_space { tw } else { 0.0 }) * th;
1569 *tm = Mat {
1570 a: 1.0,
1571 b: 0.0,
1572 c: 0.0,
1573 d: 1.0,
1574 e: tx,
1575 f: 0.0,
1576 }
1577 .then(*tm);
1578 }
1579}
1580
1581fn simple_encoding_table(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
1585 let enc = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o));
1586 let base_name = match enc {
1587 Some(Object::Name(n)) => n.clone(),
1588 Some(Object::Dictionary(d)) => d
1589 .get(b"BaseEncoding")
1590 .ok()
1591 .and_then(|o| o.as_name().ok())
1592 .map(|n| n.to_vec())
1593 .unwrap_or_default(),
1594 _ => Vec::new(),
1595 };
1596 let mut m = if base_name == b"MacRomanEncoding" {
1597 macroman_table()
1598 } else {
1599 winansi_table()
1600 };
1601 if let Some(Object::Dictionary(d)) = enc {
1603 if let Some(Object::Array(diffs)) = d.get(b"Differences").ok().and_then(|o| deref(doc, o)) {
1604 let mut code = 0u8;
1605 for el in diffs {
1606 match el {
1607 Object::Integer(i) => code = *i as u8,
1608 Object::Name(name) => {
1609 if let Some(ch) = glyph_name_to_char(name) {
1610 m.insert(code, ch);
1611 }
1612 code = code.wrapping_add(1);
1613 }
1614 _ => {}
1615 }
1616 }
1617 }
1618 }
1619 m
1620}
1621
1622fn glyph_name_to_char(name: &[u8]) -> Option<char> {
1627 let s = std::str::from_utf8(name).ok()?;
1628 if let Some(hex) = s.strip_prefix("uni") {
1629 if let Ok(cp) = u32::from_str_radix(hex.get(0..4)?, 16) {
1630 return char::from_u32(cp);
1631 }
1632 }
1633 if s.len() == 1 {
1635 let b = s.as_bytes()[0];
1636 if b.is_ascii_alphabetic() {
1637 return Some(b as char);
1638 }
1639 }
1640 let resolved = match s {
1641 "space" => ' ',
1642 "exclam" => '!',
1643 "quotedbl" => '"',
1644 "numbersign" => '#',
1645 "dollar" => '$',
1646 "percent" => '%',
1647 "ampersand" => '&',
1648 "quotesingle" => '\'',
1649 "parenleft" => '(',
1650 "parenright" => ')',
1651 "asterisk" => '*',
1652 "plus" => '+',
1653 "comma" => ',',
1654 "hyphen" => '-',
1655 "period" => '.',
1656 "slash" => '/',
1657 "zero" => '0',
1658 "one" => '1',
1659 "two" => '2',
1660 "three" => '3',
1661 "four" => '4',
1662 "five" => '5',
1663 "six" => '6',
1664 "seven" => '7',
1665 "eight" => '8',
1666 "nine" => '9',
1667 "colon" => ':',
1668 "semicolon" => ';',
1669 "less" => '<',
1670 "equal" => '=',
1671 "greater" => '>',
1672 "question" => '?',
1673 "at" => '@',
1674 "bracketleft" => '[',
1675 "backslash" => '\\',
1676 "bracketright" => ']',
1677 "asciicircum" => '^',
1678 "underscore" => '_',
1679 "grave" => '`',
1680 "braceleft" => '{',
1681 "bar" => '|',
1682 "braceright" => '}',
1683 "asciitilde" => '~',
1684 "bullet" => '\u{2022}',
1685 "periodcentered" => '\u{00B7}',
1686 "endash" => '\u{2013}',
1687 "emdash" => '\u{2014}',
1688 "quoteright" => '\u{2019}',
1689 "quoteleft" => '\u{2018}',
1690 "quotedblleft" => '\u{201C}',
1691 "quotedblright" => '\u{201D}',
1692 "quotedblbase" => '\u{201E}',
1693 "quotesinglbase" => '\u{201A}',
1694 "ff" => '\u{FB00}',
1699 "fi" => '\u{FB01}',
1700 "fl" => '\u{FB02}',
1701 "ffi" => '\u{FB03}',
1702 "ffl" => '\u{FB04}',
1703 "ft" => '\u{FB05}',
1704 "st" => '\u{FB06}',
1705 "degree" => '\u{00B0}',
1706 "trademark" => '\u{2122}',
1707 "registered" => '\u{00AE}',
1708 "copyright" => '\u{00A9}',
1709 "ellipsis" => '\u{2026}',
1710 "minus" => '\u{2212}',
1711 "fraction" => '\u{2044}',
1712 "nbspace" => '\u{00A0}',
1713 "alpha" => '\u{03B1}',
1718 "beta" => '\u{03B2}',
1719 "gamma" => '\u{03B3}',
1720 "delta" => '\u{03B4}',
1721 "epsilon" | "epsilon1" => '\u{03B5}',
1722 "zeta" => '\u{03B6}',
1723 "eta" => '\u{03B7}',
1724 "theta" | "theta1" => '\u{03B8}',
1725 "iota" => '\u{03B9}',
1726 "kappa" => '\u{03BA}',
1727 "lambda" => '\u{03BB}',
1728 "mu" => '\u{03BC}',
1729 "nu" => '\u{03BD}',
1730 "xi" => '\u{03BE}',
1731 "omicron" => '\u{03BF}',
1732 "pi" | "pi1" => '\u{03C0}',
1733 "rho" | "rho1" => '\u{03C1}',
1734 "sigma" => '\u{03C3}',
1735 "sigma1" => '\u{03C2}',
1736 "tau" => '\u{03C4}',
1737 "upsilon" => '\u{03C5}',
1738 "phi" | "phi1" => '\u{03C6}',
1739 "chi" => '\u{03C7}',
1740 "psi" => '\u{03C8}',
1741 "omega" | "omega1" => '\u{03C9}',
1742 "Gamma" => '\u{0393}',
1743 "Delta" => '\u{0394}',
1744 "Theta" => '\u{0398}',
1745 "Lambda" => '\u{039B}',
1746 "Xi" => '\u{039E}',
1747 "Pi" => '\u{03A0}',
1748 "Sigma" => '\u{03A3}',
1749 "Upsilon" => '\u{03A5}',
1750 "Phi" => '\u{03A6}',
1751 "Psi" => '\u{03A8}',
1752 "Omega" => '\u{03A9}',
1753 "lessequal" => '\u{2264}',
1754 "greaterequal" => '\u{2265}',
1755 "notequal" => '\u{2260}',
1756 "approxequal" => '\u{2248}',
1757 "equivalence" => '\u{2261}',
1758 "element" => '\u{2208}',
1759 "plusminus" => '\u{00B1}',
1760 "multiply" => '\u{00D7}',
1761 "divide" => '\u{00F7}',
1762 "infinity" => '\u{221E}',
1763 "partialdiff" => '\u{2202}',
1764 "gradient" => '\u{2207}',
1765 "summation" => '\u{2211}',
1766 "product" => '\u{220F}',
1767 "integral" => '\u{222B}',
1768 "radical" => '\u{221A}',
1769 "proportional" => '\u{221D}',
1770 "arrowright" => '\u{2192}',
1771 "arrowleft" => '\u{2190}',
1772 "arrowup" => '\u{2191}',
1773 "arrowdown" => '\u{2193}',
1774 "arrowboth" => '\u{2194}',
1775 "arrowdblright" => '\u{21D2}',
1776 "logicaland" => '\u{2227}',
1777 "logicalor" => '\u{2228}',
1778 "intersection" => '\u{2229}',
1779 "union" => '\u{222A}',
1780 "similar" => '\u{223C}',
1781 "congruent" => '\u{2245}',
1782 "dotmath" => '\u{22C5}',
1783 "asteriskmath" => '\u{2217}',
1784 _ => {
1785 if let Some((base, _)) = s.split_once('.') {
1787 if !base.is_empty() {
1788 return glyph_name_to_char(base.as_bytes());
1789 }
1790 }
1791 return None;
1792 }
1793 };
1794 Some(resolved)
1795}
1796
1797fn winansi_table() -> HashMap<u8, char> {
1799 let mut m = HashMap::new();
1800 for b in 0x20u8..=0x7e {
1801 m.insert(b, b as char);
1802 }
1803 let extra: &[(u8, char)] = &[
1805 (0x91, '\u{2018}'),
1806 (0x92, '\u{2019}'),
1807 (0x93, '\u{201C}'),
1808 (0x94, '\u{201D}'),
1809 (0x95, '\u{2022}'),
1810 (0x96, '\u{2013}'),
1811 (0x97, '\u{2014}'),
1812 (0x85, '\u{2026}'),
1813 (0xA0, '\u{00A0}'),
1814 ];
1815 for &(b, c) in extra {
1816 m.insert(b, c);
1817 }
1818 for b in 0xA1u8..=0xFF {
1819 m.entry(b).or_insert(b as char);
1820 }
1821 m
1822}
1823
1824fn macroman_table() -> HashMap<u8, char> {
1827 let mut m = HashMap::new();
1828 for b in 0x20u8..=0x7e {
1829 m.insert(b, b as char);
1830 }
1831 let high: &[(u8, char)] = &[
1832 (0xA5, '\u{2022}'), (0xD0, '\u{2013}'), (0xD1, '\u{2014}'), (0xD2, '\u{201C}'),
1836 (0xD3, '\u{201D}'),
1837 (0xD4, '\u{2018}'),
1838 (0xD5, '\u{2019}'),
1839 (0xCA, '\u{00A0}'),
1840 (0xC9, '\u{2026}'),
1841 (0xDE, '\u{FB01}'),
1842 (0xDF, '\u{FB02}'),
1843 ];
1844 for &(b, c) in high {
1845 m.insert(b, c);
1846 }
1847 m
1848}
1849
1850#[cfg(test)]
1851mod xref_repair {
1852 fn pdf_with_xref(two_byte_eol: bool) -> Vec<u8> {
1856 let content = b"BT /F1 12 Tf 72 700 Td (Invoice 922769430725) Tj ET\n";
1857 let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
1858 let objs: Vec<Vec<u8>> = vec![
1859 b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
1860 b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
1861 b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
1862 /Resources<</Font<</F1 5 0 R>>>>>>"
1863 .to_vec(),
1864 [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
1865 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".to_vec(),
1866 ];
1867
1868 let mut out = b"%PDF-1.4\n".to_vec();
1869 let mut offsets = Vec::new();
1870 for (i, body) in objs.iter().enumerate() {
1871 offsets.push(out.len());
1872 out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
1873 out.extend_from_slice(body);
1874 out.extend_from_slice(b"endobj\n");
1875 }
1876 let xref_at = out.len();
1877 let eol: &[u8] = if two_byte_eol { b" \n" } else { b"\n" };
1878 out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
1879 out.extend_from_slice(b"0000000000 65535 f");
1880 out.extend_from_slice(eol);
1881 for off in &offsets {
1882 out.extend_from_slice(format!("{off:010} 00000 n").as_bytes());
1883 out.extend_from_slice(eol);
1884 }
1885 out.extend_from_slice(
1886 format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
1887 );
1888 out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
1889 out
1890 }
1891
1892 #[test]
1898 fn short_xref_entries_still_parse() {
1899 let good = pdf_with_xref(true);
1900 let broken = pdf_with_xref(false);
1901 assert!(
1902 broken.len() < good.len(),
1903 "the broken file is the shorter one"
1904 );
1905 assert!(
1906 lopdf::Document::load_mem(&good).is_ok(),
1907 "the control file must load unaided"
1908 );
1909 assert!(
1910 lopdf::Document::load_mem(&broken).is_err(),
1911 "lopdf rejects 19-byte entries — if this ever passes, drop the repair"
1912 );
1913
1914 let cells = |b: &[u8]| -> Vec<String> {
1915 super::pdf_textlines(b)
1916 .into_iter()
1917 .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
1918 .collect()
1919 };
1920 let from_good = cells(&good);
1921 assert!(
1922 from_good.iter().any(|t| t.contains("922769430725")),
1923 "control text: {from_good:?}"
1924 );
1925 assert_eq!(
1926 cells(&broken),
1927 from_good,
1928 "repair must match the good parse"
1929 );
1930 }
1931
1932 #[test]
1937 fn overstated_stream_length_still_yields_content() {
1938 let good = pdf_with_xref(true);
1939 let broken = {
1942 let at = good
1943 .windows(8)
1944 .position(|w| w == b"/Length ")
1945 .expect("a /Length")
1946 + 8;
1947 let digits = good[at..].iter().take_while(|c| c.is_ascii_digit()).count();
1948 let n: usize = std::str::from_utf8(&good[at..at + digits])
1949 .unwrap()
1950 .parse()
1951 .unwrap();
1952 let inflated = (n + 1).to_string();
1953 assert_eq!(inflated.len(), digits, "keep the digit count");
1954 let mut b = good.clone();
1955 b[at..at + digits].copy_from_slice(inflated.as_bytes());
1956 b
1957 };
1958 assert_eq!(broken.len(), good.len(), "the defect must not move bytes");
1959 let raw = lopdf::Document::load_mem(&broken).expect("still loads");
1961 assert!(
1962 raw.get_pages()
1963 .into_values()
1964 .all(|p| raw.get_page_content(p).is_empty()),
1965 "lopdf should drop the stream — if it stops, drop this repair"
1966 );
1967 let text = |b: &[u8]| -> Vec<String> {
1969 super::pdf_textlines(b)
1970 .into_iter()
1971 .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
1972 .collect()
1973 };
1974 let expected = text(&good);
1975 assert!(!expected.is_empty(), "control must produce text");
1976 assert_eq!(text(&broken), expected);
1977 }
1978
1979 #[test]
1983 fn repair_declines_when_padding_would_move_objects() {
1984 let mut incremental = pdf_with_xref(false);
1985 incremental.extend_from_slice(b"6 0 obj<</Type/Whatever>>endobj\n");
1986 let declined = super::pad_short_xref_entries(&incremental).unwrap_err();
1987 assert!(
1988 declined.contains("object follows the xref"),
1989 "reason: {declined}"
1990 );
1991 }
1992}
1993
1994#[cfg(test)]
2000mod base14_fonts {
2001 fn pdf_with_font(fontdict: &[u8], text: &[u8]) -> Vec<u8> {
2003 let content = [b"BT /F1 12 Tf 72 700 Td (".as_slice(), text, b") Tj ET\n"].concat();
2004 let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2005 let objs: Vec<Vec<u8>> = vec![
2006 b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2007 b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2008 b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2009 /Resources<</Font<</F1 5 0 R>>>>>>"
2010 .to_vec(),
2011 [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
2012 fontdict.to_vec(),
2013 ];
2014 let mut out = b"%PDF-1.4\n".to_vec();
2015 let mut offsets = Vec::new();
2016 for (i, body) in objs.iter().enumerate() {
2017 offsets.push(out.len());
2018 out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2019 out.extend_from_slice(body);
2020 out.extend_from_slice(b"endobj\n");
2021 }
2022 let xref_at = out.len();
2023 out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2024 out.extend_from_slice(b"0000000000 65535 f \n");
2025 for off in &offsets {
2026 out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2027 }
2028 out.extend_from_slice(
2029 format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2030 );
2031 out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2032 out
2033 }
2034
2035 fn cells(pdf: &[u8]) -> Vec<crate::pdfium_backend::TextCell> {
2037 super::pdf_textlines(pdf)
2038 .into_iter()
2039 .flat_map(|(_, _, c)| c)
2040 .collect()
2041 }
2042
2043 #[test]
2045 fn standard14_faces_get_builtin_widths() {
2046 for fontdict in [
2047 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>".as_slice(),
2049 b"<</Type/Font/Subtype/Type1/BaseFont/Times-BoldItalic>>",
2051 b"<</Type/Font/Subtype/TrueType/BaseFont/Arial,Bold>>",
2053 b"<</Type/Font/Subtype/Type1/BaseFont/ABCDEF+Courier-Oblique>>",
2054 ] {
2055 let pdf = pdf_with_font(fontdict, b"Words have width now");
2056 let cs = cells(&pdf);
2057 let text: String = cs
2058 .iter()
2059 .map(|c| c.text.as_str())
2060 .collect::<Vec<_>>()
2061 .join(" ");
2062 assert!(
2063 text.contains("Words have width now"),
2064 "{}: text lost: {text:?}",
2065 String::from_utf8_lossy(fontdict)
2066 );
2067 assert!(
2068 cs.iter().all(|c| c.r > c.l),
2069 "{}: zero-width cells: {cs:?}",
2070 String::from_utf8_lossy(fontdict)
2071 );
2072 }
2073 }
2074
2075 #[test]
2078 fn explicit_widths_win_and_unknown_faces_are_untouched() {
2079 let explicit = pdf_with_font(
2083 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/FirstChar 65\
2084 /Widths[100 100 100 100]/Encoding/WinAnsiEncoding>>",
2085 b"ABBA",
2086 );
2087 let builtin = pdf_with_font(
2088 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>",
2089 b"ABBA",
2090 );
2091 let w = |pdf: &[u8]| {
2092 let cs = cells(pdf);
2093 assert_eq!(cs.len(), 1, "one word cell: {cs:?}");
2094 cs[0].r - cs[0].l
2095 };
2096 let (we, wb) = (w(&explicit), w(&builtin));
2097 assert!(
2098 (we - 4.8).abs() < 0.1,
2099 "explicit widths must win: got {we}, want 4×100×12/1000"
2100 );
2101 assert!(
2102 wb > 2.0 * we,
2103 "built-in Helvetica is much wider: {wb} vs {we}"
2104 );
2105
2106 let unknown = pdf_with_font(
2110 b"<</Type/Font/Subtype/Type1/BaseFont/FancyCorp-Display>>",
2111 b"Mystery",
2112 );
2113 let cs = cells(&unknown);
2114 let text: String = cs.iter().map(|c| c.text.as_str()).collect();
2115 assert!(text.contains("Mystery"), "text still decodes: {cs:?}");
2116 }
2117}
2118
2119#[cfg(test)]
2120mod overpainted {
2121 use crate::pdfium_backend::TextCell;
2122
2123 fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
2124 TextCell {
2125 text: text.into(),
2126 l,
2127 t,
2128 r,
2129 b,
2130 }
2131 }
2132
2133 #[test]
2137 fn stacked_logo_glyphs_are_dropped() {
2138 let mut cells = vec![
2139 cell("\"", 72.7, 21.5, 86.4, 31.5),
2140 cell("==", 59.4, 21.5, 99.6, 31.5),
2141 cell("Herr", 65.2, 151.3, 81.7, 161.3),
2142 ];
2143 super::drop_overpainted_cells(&mut cells);
2144 assert_eq!(cells.len(), 1, "cells: {cells:?}");
2145 assert_eq!(cells[0].text, "Herr");
2146 }
2147
2148 #[test]
2152 fn prose_and_double_draw_are_kept() {
2153 let mut cells = vec![
2154 cell("Telefon", 354.3, 133.2, 381.5, 143.2),
2155 cell("0676/2000", 387.3, 133.2, 428.7, 143.2),
2156 cell("Bold", 100.0, 50.0, 130.0, 60.0),
2157 cell("Bold", 100.3, 50.0, 130.3, 60.0),
2158 ];
2159 super::drop_overpainted_cells(&mut cells);
2160 assert_eq!(cells.len(), 4);
2161 }
2162}
2163
2164#[cfg(test)]
2165mod vestigial_layer {
2166 use crate::pdfium_backend::{PdfPage, TextCell};
2167
2168 fn page_with(texts: &[&str]) -> PdfPage {
2169 let cells = texts
2170 .iter()
2171 .enumerate()
2172 .map(|(i, t)| TextCell {
2173 text: t.to_string(),
2174 l: 10.0,
2175 t: 10.0 + 12.0 * i as f32,
2176 r: 90.0,
2177 b: 20.0 + 12.0 * i as f32,
2178 })
2179 .collect();
2180 PdfPage::from_cells(595.0, 842.0, 1.0, cells)
2181 }
2182
2183 #[test]
2188 fn typed_in_form_fields_are_not_a_text_layer() {
2189 let pages = vec![
2190 page_with(&["03", "05", "2025"]),
2191 page_with(&[]),
2192 page_with(&[]),
2193 ];
2194 assert!(super::text_layer_is_vestigial(&pages));
2195 assert!(super::text_layer_is_vestigial(&[page_with(&[])]));
2196 }
2197
2198 #[test]
2201 fn sparse_but_real_documents_pass() {
2202 let one_pager = vec![page_with(&[
2203 "Confidential briefing",
2204 "Prepared for the board meeting",
2205 "Do not distribute",
2206 ])];
2207 assert!(!super::text_layer_is_vestigial(&one_pager));
2208 }
2209}