1use std::collections::HashMap;
17use std::sync::Arc;
18
19use lopdf::{Dictionary, Document, Object};
20
21use crate::pdfium_backend::Glyph;
22
23#[derive(Default)]
32struct DocCaches {
33 fonts: HashMap<(lopdf::ObjectId, Vec<u8>), Arc<Font>>,
34 forms: HashMap<lopdf::ObjectId, Arc<lopdf::content::Content>>,
35}
36
37#[derive(Clone, Copy)]
39struct Mat {
40 a: f64,
41 b: f64,
42 c: f64,
43 d: f64,
44 e: f64,
45 f: f64,
46}
47
48impl Mat {
49 const ID: Mat = Mat {
50 a: 1.0,
51 b: 0.0,
52 c: 0.0,
53 d: 1.0,
54 e: 0.0,
55 f: 0.0,
56 };
57
58 fn then(self, m: Mat) -> Mat {
60 Mat {
61 a: self.a * m.a + self.b * m.c,
62 b: self.a * m.b + self.b * m.d,
63 c: self.c * m.a + self.d * m.c,
64 d: self.c * m.b + self.d * m.d,
65 e: self.e * m.a + self.f * m.c + m.e,
66 f: self.e * m.b + self.f * m.d + m.f,
67 }
68 }
69
70 fn apply(self, x: f64, y: f64) -> (f64, f64) {
71 (
72 self.a * x + self.c * y + self.e,
73 self.b * x + self.d * y + self.f,
74 )
75 }
76}
77
78struct Font {
80 two_byte: bool,
82 to_unicode: HashMap<u32, String>,
84 widths: HashMap<u32, f64>,
86 default_width: f64,
87 simple_encoding: Option<HashMap<u8, char>>,
89 fallback_names: HashMap<u8, String>,
93 program_encoding: HashMap<u8, char>,
97 ascent: f64,
98 descent: f64,
99 hash: u64,
100 style: crate::font_style::FontStyle,
103}
104
105impl Font {
106 fn decode_code(&self, code: u32) -> (Option<String>, f64) {
107 let w = self
108 .widths
109 .get(&code)
110 .copied()
111 .unwrap_or(self.default_width);
112 if let Some(s) = self.to_unicode.get(&code) {
113 return (Some(decompose_ligatures(s)), w);
114 }
115 if !self.two_byte {
116 if let Some(name) = self.fallback_names.get(&(code as u8)) {
119 return (Some(format!("/{name}")), w);
120 }
121 if let Some(enc) = &self.simple_encoding {
122 if let Some(&ch) = enc.get(&(code as u8)) {
123 return (Some(decompose_ligatures(&ch.to_string())), w);
124 }
125 }
126 if let Some(&ch) = self.program_encoding.get(&(code as u8)) {
134 return (Some(decompose_ligatures(&ch.to_string())), w);
135 }
136 }
137 (None, w)
138 }
139}
140
141fn decompose_ligatures(s: &str) -> String {
145 if !s.chars().any(|c| ('\u{FB00}'..='\u{FB06}').contains(&c)) {
146 return s.to_string();
147 }
148 s.chars()
149 .map(|c| {
150 match c {
151 '\u{FB00}' => "ff",
152 '\u{FB01}' => "fi",
153 '\u{FB02}' => "fl",
154 '\u{FB03}' => "ffi",
155 '\u{FB04}' => "ffl",
156 '\u{FB05}' => "ft",
157 '\u{FB06}' => "st",
158 _ => return c.to_string(),
159 }
160 .to_string()
161 })
162 .collect()
163}
164
165fn hash_name(name: &[u8]) -> u64 {
166 use std::hash::{Hash, Hasher};
167 let mut h = std::collections::hash_map::DefaultHasher::new();
168 name.hash(&mut h);
169 h.finish()
170}
171
172fn as_dict<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Dictionary> {
174 match obj {
175 Object::Dictionary(d) => Some(d),
176 Object::Reference(id) => doc.get_object(*id).ok().and_then(|o| o.as_dict().ok()),
177 _ => None,
178 }
179}
180
181fn deref<'a>(doc: &'a Document, obj: &'a Object) -> Option<&'a Object> {
182 match obj {
183 Object::Reference(id) => doc.get_object(*id).ok(),
184 other => Some(other),
185 }
186}
187
188fn parse_font(doc: &Document, name: &[u8], fdict: &Dictionary) -> Font {
190 let subtype: &[u8] = fdict
191 .get(b"Subtype")
192 .ok()
193 .and_then(|o| o.as_name().ok())
194 .unwrap_or(&[]);
195 let two_byte = subtype == b"Type0".as_slice();
196
197 let to_unicode = fdict
198 .get(b"ToUnicode")
199 .ok()
200 .and_then(|o| deref(doc, o))
201 .and_then(|o| o.as_stream().ok())
202 .and_then(|s| s.decompressed_content().ok())
203 .map(|data| parse_tounicode(&data))
204 .unwrap_or_default();
205
206 let (mut widths, mut default_width) = if two_byte {
207 cid_widths(doc, fdict)
208 } else {
209 simple_widths(doc, fdict)
210 };
211
212 let simple_encoding = if two_byte {
213 None
214 } else {
215 Some(simple_encoding_table(doc, fdict))
216 };
217
218 if !two_byte && widths.is_empty() && default_width == 0.0 {
226 if let Some(std14) = base_font_name(fdict).and_then(|n| crate::std14::widths_for(&n)) {
227 if let Some(enc) = &simple_encoding {
228 for (&code, &ch) in enc {
229 if let Some(w) = std14.width(ch) {
230 widths.insert(u32::from(code), w);
231 }
232 }
233 }
234 default_width = 500.0;
237 }
238 }
239 let fallback_names = if two_byte {
240 HashMap::new()
241 } else {
242 differences_gid_names(doc, fdict)
243 };
244 let program_encoding = if two_byte {
245 HashMap::new()
246 } else {
247 type1_program_encoding(doc, fdict)
248 };
249
250 let (ascent, descent) = font_ascent_descent(doc, fdict, two_byte);
251
252 Font {
253 two_byte,
254 to_unicode,
255 widths,
256 default_width,
257 simple_encoding,
258 fallback_names,
259 program_encoding,
260 ascent,
261 descent,
262 hash: hash_name(name),
263 style: crate::font_style::parse_font_style(&String::from_utf8_lossy(
264 &base_font_name(fdict).unwrap_or_else(|| name.to_vec()),
265 )),
266 }
267}
268
269fn differences_gid_names(doc: &Document, fdict: &Dictionary) -> HashMap<u8, String> {
276 let mut map = HashMap::new();
277 let Some(Object::Dictionary(enc)) = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o))
278 else {
279 return map;
280 };
281 let Some(Object::Array(diffs)) = enc.get(b"Differences").ok().and_then(|o| deref(doc, o))
282 else {
283 return map;
284 };
285 let mut code = 0u8;
286 for el in diffs {
287 match el {
288 Object::Integer(i) => code = *i as u8,
289 Object::Name(name) => {
290 if glyph_name_to_char(name).is_none() && is_gid_name(name) {
291 map.insert(code, String::from_utf8_lossy(name).into_owned());
292 }
293 code = code.wrapping_add(1);
294 }
295 _ => {}
296 }
297 }
298 map
299}
300
301fn type1_program_encoding(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
309 let mut map = HashMap::new();
310 let Some(desc) = fdict
311 .get(b"FontDescriptor")
312 .ok()
313 .and_then(|o| deref(doc, o))
314 .and_then(|o| o.as_dict().ok())
315 else {
316 return map;
317 };
318 let Some(data) = desc
319 .get(b"FontFile")
320 .ok()
321 .and_then(|o| deref(doc, o))
322 .and_then(|o| o.as_stream().ok())
323 .and_then(|s| s.decompressed_content().ok())
324 else {
325 return map;
326 };
327 let head_end = data
329 .windows(5)
330 .position(|w| w == b"eexec")
331 .unwrap_or(data.len());
332 let head = String::from_utf8_lossy(&data[..head_end]);
333 let toks: Vec<&str> = head.split_whitespace().collect();
335 for w in toks.windows(4) {
336 if w[0] == "dup" && w[3] == "put" {
337 if let (Ok(code), Some(name)) = (w[1].parse::<u32>(), w[2].strip_prefix('/')) {
338 if code <= 255 {
339 if let Some(ch) = glyph_name_to_char(name.as_bytes()) {
340 map.insert(code as u8, ch);
341 }
342 }
343 }
344 }
345 }
346 map
347}
348
349fn is_gid_name(name: &[u8]) -> bool {
355 let Ok(s) = std::str::from_utf8(name) else {
356 return false;
357 };
358 if s.starts_with("afii") || s.starts_with("uni") {
359 return false;
360 }
361 for prefix in ["g", "G", "cid", "CID", "glyph", "index"] {
362 if let Some(rest) = s.strip_prefix(prefix) {
363 if !rest.is_empty() && rest.bytes().all(|b| b.is_ascii_digit()) {
364 return true;
365 }
366 }
367 }
368 let alpha = s.bytes().take_while(|b| b.is_ascii_alphabetic()).count();
372 let digits = s.len() - alpha;
373 (1..=3).contains(&alpha)
374 && digits >= 3
375 && s.as_bytes()[alpha..].iter().all(|b| b.is_ascii_digit())
376}
377
378fn font_ascent_descent(doc: &Document, fdict: &Dictionary, two_byte: bool) -> (f64, f64) {
379 let descr_owner = if two_byte {
381 fdict
382 .get(b"DescendantFonts")
383 .ok()
384 .and_then(|o| deref(doc, o))
385 .and_then(|o| match o {
386 Object::Array(a) => a.first(),
387 _ => None,
388 })
389 .and_then(|o| as_dict(doc, o))
390 } else {
391 Some(fdict)
392 };
393 let fd = descr_owner
394 .and_then(|d| d.get(b"FontDescriptor").ok())
395 .and_then(|o| as_dict(doc, o));
396 let asc = fd
397 .and_then(|d| d.get(b"Ascent").ok())
398 .and_then(|o| {
399 o.as_float()
400 .ok()
401 .or_else(|| o.as_i64().ok().map(|i| i as f32))
402 })
403 .unwrap_or(750.0) as f64;
404 let desc = fd
405 .and_then(|d| d.get(b"Descent").ok())
406 .and_then(|o| {
407 o.as_float()
408 .ok()
409 .or_else(|| o.as_i64().ok().map(|i| i as f32))
410 })
411 .unwrap_or(-250.0) as f64;
412 if asc - desc <= 1.0 {
419 return (750.0, -250.0);
420 }
421 (asc, desc)
422}
423
424fn base_font_name(fdict: &Dictionary) -> Option<Vec<u8>> {
426 let name = fdict.get(b"BaseFont").ok()?.as_name().ok()?;
427 let stripped = match name.iter().position(|&b| b == b'+') {
428 Some(i) if i == 6 => &name[i + 1..],
429 _ => name,
430 };
431 Some(stripped.to_vec())
432}
433
434fn simple_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
436 let mut map = HashMap::new();
437 let first = fdict
438 .get(b"FirstChar")
439 .ok()
440 .and_then(|o| o.as_i64().ok())
441 .unwrap_or(0) as u32;
442 if let Some(Object::Array(arr)) = fdict.get(b"Widths").ok().and_then(|o| deref(doc, o)) {
443 for (i, w) in arr.iter().enumerate() {
444 if let Some(w) = num(w) {
445 map.insert(first + i as u32, w);
446 }
447 }
448 }
449 let dw = fdict
450 .get(b"FontDescriptor")
451 .ok()
452 .and_then(|o| as_dict(doc, o))
453 .and_then(|d| d.get(b"MissingWidth").ok())
454 .and_then(num)
455 .unwrap_or(0.0);
456 (map, dw)
457}
458
459fn cid_widths(doc: &Document, fdict: &Dictionary) -> (HashMap<u32, f64>, f64) {
461 let mut map = HashMap::new();
462 let Some(desc) = fdict
463 .get(b"DescendantFonts")
464 .ok()
465 .and_then(|o| deref(doc, o))
466 .and_then(|o| match o {
467 Object::Array(a) => a.first(),
468 _ => None,
469 })
470 .and_then(|o| as_dict(doc, o))
471 else {
472 return (map, 1000.0);
473 };
474 let dw = desc.get(b"DW").ok().and_then(num).unwrap_or(1000.0);
475 if let Some(Object::Array(w)) = desc.get(b"W").ok().and_then(|o| deref(doc, o)) {
476 let mut i = 0;
477 while i < w.len() {
478 let c = w.get(i).and_then(num);
479 match (c, w.get(i + 1)) {
480 (Some(c), Some(Object::Array(list))) => {
482 for (k, wv) in list.iter().enumerate() {
483 if let Some(wv) = num(wv) {
484 map.insert(c as u32 + k as u32, wv);
485 }
486 }
487 i += 2;
488 }
489 (Some(c1), Some(o2)) => {
491 if let (Some(c2), Some(wv)) = (num(o2), w.get(i + 2).and_then(num)) {
492 for cid in c1 as u32..=c2 as u32 {
493 map.insert(cid, wv);
494 }
495 }
496 i += 3;
497 }
498 _ => break,
499 }
500 }
501 }
502 (map, dw)
503}
504
505fn num(o: &Object) -> Option<f64> {
506 match o {
507 Object::Integer(i) => Some(*i as f64),
508 Object::Real(r) => Some(*r as f64),
509 _ => None,
510 }
511}
512
513pub(crate) fn parse_tounicode(data: &[u8]) -> HashMap<u32, String> {
515 let text = String::from_utf8_lossy(data);
516 let mut map = HashMap::new();
517 let hex = |s: &str| -> Option<Vec<u16>> {
518 let s = s.trim();
519 if !s.starts_with('<') || !s.ends_with('>') {
520 return None;
521 }
522 let h = &s[1..s.len() - 1];
523 let bytes: Vec<u8> = (0..h.len())
524 .step_by(2)
525 .filter_map(|i| u8::from_str_radix(h.get(i..i + 2)?, 16).ok())
526 .collect();
527 Some(
528 bytes
529 .chunks(2)
530 .map(|c| {
531 if c.len() == 2 {
532 u16::from_be_bytes([c[0], c[1]])
533 } else {
534 c[0] as u16
535 }
536 })
537 .collect(),
538 )
539 };
540 let u16s_to_string = |u: &[u16]| String::from_utf16_lossy(u);
541 let code_of = |u: &[u16]| u.iter().fold(0u32, |acc, &x| (acc << 16) | x as u32);
542
543 let tokens: Vec<String> = {
547 let bytes = text.as_bytes();
548 let mut toks = Vec::new();
549 let mut i = 0;
550 while i < bytes.len() {
551 let c = bytes[i];
552 if c.is_ascii_whitespace() {
553 i += 1;
554 } else if c == b'<' {
555 let start = i;
556 while i < bytes.len() && bytes[i] != b'>' {
557 i += 1;
558 }
559 i += 1; toks.push(String::from_utf8_lossy(&bytes[start..i.min(bytes.len())]).into_owned());
561 } else if c == b'[' || c == b']' {
562 toks.push((c as char).to_string());
563 i += 1;
564 } else {
565 let start = i;
566 while i < bytes.len()
567 && !bytes[i].is_ascii_whitespace()
568 && bytes[i] != b'<'
569 && bytes[i] != b'['
570 && bytes[i] != b']'
571 {
572 i += 1;
573 }
574 toks.push(String::from_utf8_lossy(&bytes[start..i]).into_owned());
575 }
576 }
577 toks
578 };
579 let tokens: Vec<&str> = tokens.iter().map(|s| s.as_str()).collect();
580 let mut i = 0;
581 while i < tokens.len() {
582 match tokens[i] {
583 "beginbfchar" => {
584 i += 1;
585 while i + 1 < tokens.len() && tokens[i] != "endbfchar" {
586 if let (Some(src), Some(dst)) = (hex(tokens[i]), hex(tokens[i + 1])) {
587 map.insert(code_of(&src), u16s_to_string(&dst));
588 }
589 i += 2;
590 }
591 }
592 "beginbfrange" => {
593 i += 1;
594 while i + 2 < tokens.len() && tokens[i] != "endbfrange" {
595 let (Some(lo), Some(hi)) = (hex(tokens[i]), hex(tokens[i + 1])) else {
596 i += 1;
597 continue;
598 };
599 let lo = code_of(&lo);
600 let hi = code_of(&hi);
601 if tokens[i + 2] == "[" {
602 let mut j = i + 3;
604 let mut code = lo;
605 while j < tokens.len() && tokens[j] != "]" {
606 if let Some(dst) = hex(tokens[j]) {
607 map.insert(code, u16s_to_string(&dst));
608 }
609 code += 1;
610 j += 1;
611 }
612 i = j + 1;
613 } else if let Some(dst) = hex(tokens[i + 2]) {
614 let base = code_of(&dst);
616 for (k, code) in (lo..=hi).enumerate() {
617 if let Some(ch) = char::from_u32(base + k as u32) {
618 map.insert(code, ch.to_string());
619 }
620 }
621 i += 3;
622 } else {
623 i += 1;
624 }
625 }
626 }
627 _ => i += 1,
628 }
629 }
630 map
631}
632
633fn codes(font: &Font, bytes: &[u8]) -> Vec<u32> {
635 if font.two_byte {
636 bytes
637 .chunks(2)
638 .map(|c| {
639 if c.len() == 2 {
640 ((c[0] as u32) << 8) | c[1] as u32
641 } else {
642 c[0] as u32
643 }
644 })
645 .collect()
646 } else {
647 bytes.iter().map(|&b| b as u32).collect()
648 }
649}
650
651#[derive(Debug, Clone, Copy, PartialEq)]
666pub(crate) struct PageBox {
667 pub l: f32,
669 pub b: f32,
671 pub w: f32,
672 pub h: f32,
673}
674
675impl PageBox {
676 pub fn top(&self) -> f32 {
678 self.b + self.h
679 }
680}
681
682fn inherited_rect(
685 doc: &Document,
686 page_id: lopdf::ObjectId,
687 key: &[u8],
688) -> Option<(f32, f32, f32, f32)> {
689 let mut id = page_id;
690 for _ in 0..32 {
691 let dict = doc.get_object(id).ok()?.as_dict().ok()?;
692 if let Some(Object::Array(a)) = dict.get(key).ok().and_then(|o| deref(doc, o)) {
693 let v: Vec<f32> = a.iter().filter_map(|o| num(o).map(|x| x as f32)).collect();
694 if v.len() == 4 && v.iter().all(|x| x.is_finite()) {
695 return Some((
696 v[0].min(v[2]),
697 v[1].min(v[3]),
698 v[0].max(v[2]),
699 v[1].max(v[3]),
700 ));
701 }
702 return None;
703 }
704 id = dict.get(b"Parent").ok()?.as_reference().ok()?;
705 }
706 None
707}
708
709pub(crate) fn page_box(doc: &Document, page_id: lopdf::ObjectId) -> PageBox {
710 let nonempty = |r: &(f32, f32, f32, f32)| r.2 > r.0 && r.3 > r.1;
711 let media = inherited_rect(doc, page_id, b"MediaBox")
712 .filter(nonempty)
713 .unwrap_or((0.0, 0.0, 612.0, 792.0));
714 let crop = inherited_rect(doc, page_id, b"CropBox")
715 .map(|c| {
716 (
717 c.0.max(media.0),
718 c.1.max(media.1),
719 c.2.min(media.2),
720 c.3.min(media.3),
721 )
722 })
723 .filter(nonempty)
724 .unwrap_or(media);
725 PageBox {
726 l: crop.0,
727 b: crop.1,
728 w: crop.2 - crop.0,
729 h: crop.3 - crop.1,
730 }
731}
732
733fn page_size(doc: &Document, page_id: lopdf::ObjectId) -> (f32, f32) {
736 let pb = page_box(doc, page_id);
737 (pb.w, pb.h)
738}
739
740pub fn content_diagnosis(bytes: &[u8]) -> String {
746 let Some(doc) = load_document(bytes) else {
747 return "document does not load".into();
748 };
749 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
750 pages.sort_by_key(|(n, _)| *n);
751 let mut out = String::new();
752 let mut caches = DocCaches::default();
753 for (n, pid) in pages.into_iter().take(4) {
754 let content_bytes = doc.get_page_content(pid);
755 let ops = lopdf::content::Content::decode(&content_bytes)
756 .map(|c| c.operations.len())
757 .ok();
758 let res = page_res(&doc, pid);
759 let fonts = res.map(|r| fonts_from_res(&doc, r, &mut caches).len());
760 let glyphs = page_glyphs_cached(&doc, pid, &mut caches).len();
761 out.push_str(&format!(
762 "\n page {n}: content {} B, ops {}, resources {}, fonts {}, glyphs {}",
763 content_bytes.len(),
764 ops.map_or("UNDECODABLE".to_string(), |n| n.to_string()),
765 if res.is_some() { "ok" } else { "MISSING" },
766 fonts.map_or("-".to_string(), |n| n.to_string()),
767 glyphs,
768 ));
769 }
770 out
771}
772
773pub fn text_layer_is_vestigial(pages: &[crate::pdfium_backend::PdfPage]) -> bool {
786 let lines: usize = pages.iter().map(|p| p.cells.len()).sum();
787 if lines == 0 {
788 return true;
789 }
790 let chars: usize = pages
791 .iter()
792 .flat_map(|p| &p.cells)
793 .map(|c| c.text.chars().count())
794 .sum();
795 lines <= pages.len() && chars < 32
796}
797
798pub fn xref_repair_status(bytes: &[u8]) -> String {
802 if Document::load_mem(bytes).is_ok() {
803 return "loads unaided; no repair needed".into();
804 }
805 match pad_short_xref_entries(bytes) {
806 Ok(fixed) => match Document::load_mem(&fixed) {
807 Ok(_) => "repaired: cross-reference entries padded to 20 bytes".into(),
808 Err(e) => format!("padded the entries, but it still will not load: {e}"),
809 },
810 Err(why) => format!("repair declined — {why}"),
811 }
812}
813
814pub(crate) fn load_document(bytes: &[u8]) -> Option<Document> {
829 open_document(bytes, None).ok()
830}
831
832#[derive(Debug, Clone, Copy, PartialEq, Eq)]
834pub(crate) enum OpenError {
835 Unreadable,
837 Password,
839}
840
841pub(crate) fn open_document(bytes: &[u8], password: Option<&str>) -> Result<Document, OpenError> {
848 let Some(doc) = load_document_raw(bytes, password) else {
849 if password.is_some() && load_document_raw(bytes, None).is_some_and(|d| d.is_encrypted()) {
853 return Err(OpenError::Password);
854 }
855 return Err(OpenError::Unreadable);
856 };
857 if doc.is_encrypted() {
858 return Err(OpenError::Password);
859 }
860 Ok(doc)
861}
862
863fn load_options(password: Option<&str>) -> lopdf::LoadOptions {
864 lopdf::LoadOptions {
865 password: password.map(str::to_string),
866 ..lopdf::LoadOptions::default()
867 }
868}
869
870fn load_document_raw(bytes: &[u8], password: Option<&str>) -> Option<Document> {
871 let mut fallback = None;
876 if let Some(doc) = best_effort_load(bytes, password, &mut fallback) {
877 return Some(doc);
878 }
879 let xref_fixed = pad_short_xref_entries(bytes).ok();
880 if let Some(fixed) = &xref_fixed {
881 if let Some(doc) = best_effort_load(fixed, password, &mut fallback) {
882 return Some(doc);
883 }
884 }
885 let lengths_fixed = fix_stream_lengths(xref_fixed.as_deref().unwrap_or(bytes));
888 if let Some(doc) = best_effort_load(&lengths_fixed, password, &mut fallback) {
889 return Some(doc);
890 }
891 fallback
892}
893
894fn best_effort_load(
897 data: &[u8],
898 password: Option<&str>,
899 fallback: &mut Option<Document>,
900) -> Option<Document> {
901 match Document::load_mem_with_options(data, load_options(password)) {
902 Ok(doc) if has_page_content(&doc) => Some(doc),
903 Ok(doc) => {
904 fallback.get_or_insert(doc);
905 None
906 }
907 Err(_) => None,
908 }
909}
910
911fn has_page_content(doc: &Document) -> bool {
915 doc.get_pages()
916 .into_values()
917 .take(4)
918 .any(|pid| !doc.get_page_content(pid).is_empty())
919}
920
921fn fix_stream_lengths(bytes: &[u8]) -> Vec<u8> {
934 let mut out = bytes.to_vec();
935 let mut i = 0;
936 while let Some(rel) = find(&out[i..], b"stream") {
937 let kw = i + rel;
938 i = kw + 6;
939 if kw >= 3 && &out[kw - 3..kw] == b"end" {
941 continue;
942 }
943 let mut data = kw + 6;
945 if out.get(data..data + 2) == Some(b"\r\n".as_slice()) {
946 data += 2;
947 } else if matches!(out.get(data), Some(b'\n' | b'\r')) {
948 data += 1;
949 }
950 let Some(end) = find(&out[data..], b"endstream").map(|r| data + r) else {
951 continue;
952 };
953 let dict_start = out[..kw].iter().rposition(|&c| c == b'<').unwrap_or(0);
955 let Some(lrel) = find(&out[dict_start..kw], b"/Length") else {
956 continue;
957 };
958 let mut d = dict_start + lrel + 7;
959 while matches!(out.get(d), Some(b' ')) {
960 d += 1;
961 }
962 let digits = out[d..].iter().take_while(|c| c.is_ascii_digit()).count();
963 if digits == 0 {
964 continue;
965 }
966 let declared: usize = match std::str::from_utf8(&out[d..d + digits])
967 .ok()
968 .and_then(|s| s.parse().ok())
969 {
970 Some(v) => v,
971 None => continue,
972 };
973 let actual = end - data;
974 let replacement = actual.to_string();
977 if actual == declared || replacement.len() > digits {
978 continue;
979 }
980 out[d..d + digits].fill(b' ');
981 out[d..d + replacement.len()].copy_from_slice(replacement.as_bytes());
982 }
983 out
984}
985
986fn find(haystack: &[u8], needle: &[u8]) -> Option<usize> {
987 haystack.windows(needle.len()).position(|w| w == needle)
988}
989
990fn pad_short_xref_entries(bytes: &[u8]) -> Result<Vec<u8>, &'static str> {
993 let is_boundary = |i: usize| i == 0 || matches!(bytes[i - 1], b'\n' | b'\r');
996 let mut starts = (0..bytes.len().saturating_sub(4))
997 .filter(|&i| &bytes[i..i + 4] == b"xref" && is_boundary(i));
998 let xref_at = starts
999 .next()
1000 .ok_or("no classic `xref` section (an xref stream?)")?;
1001 if starts.next().is_some() {
1002 return Err("more than one xref section (incremental update)");
1003 }
1004 let last_obj = bytes
1005 .windows(3)
1006 .rposition(|w| w == b"obj")
1007 .ok_or("no objects found")?;
1008 if last_obj > xref_at {
1009 return Err("an object follows the xref — padding would move it");
1010 }
1011
1012 let mut out = bytes[..xref_at].to_vec();
1013 out.extend_from_slice(b"xref\n");
1014 let mut i = xref_at + 4;
1015 let skip_ws = |i: &mut usize| {
1016 while matches!(bytes.get(*i), Some(b'\r' | b'\n' | b' ')) {
1017 *i += 1;
1018 }
1019 };
1020 loop {
1021 skip_ws(&mut i);
1022 if bytes[i..].starts_with(b"trailer") {
1024 out.extend_from_slice(&bytes[i..]);
1025 return Ok(out);
1026 }
1027 let header_end = i + bytes[i..]
1028 .iter()
1029 .position(|c| matches!(c, b'\n' | b'\r'))
1030 .ok_or("subsection header runs off the end")?;
1031 let header = std::str::from_utf8(&bytes[i..header_end])
1032 .map_err(|_| "subsection header is not text")?
1033 .trim();
1034 let mut parts = header.split_whitespace();
1035 let count: usize = parts
1036 .nth(1)
1037 .and_then(|c| c.parse().ok())
1038 .ok_or("unparseable subsection header")?;
1039 if parts.next().is_some() || count == 0 {
1040 return Err("unexpected subsection header shape");
1041 }
1042 out.extend_from_slice(header.as_bytes());
1043 out.push(b'\n');
1044 i = header_end;
1045 for _ in 0..count {
1046 skip_ws(&mut i);
1047 let entry = bytes.get(i..i + 18).ok_or("xref entry runs off the end")?;
1049 let well_formed = entry[..10].iter().all(u8::is_ascii_digit)
1050 && entry[10] == b' '
1051 && entry[11..16].iter().all(u8::is_ascii_digit)
1052 && entry[16] == b' '
1053 && matches!(entry[17], b'n' | b'f');
1054 if !well_formed {
1055 return Err("xref entry is not `nnnnnnnnnn ggggg n`");
1056 }
1057 out.extend_from_slice(entry);
1058 out.extend_from_slice(b" \n"); i += 18;
1060 }
1061 }
1062}
1063
1064pub fn debug_glyphs(bytes: &[u8], index: usize) -> Vec<(char, f32, f32, f32, f32)> {
1067 let Some(doc) = load_document(bytes) else {
1068 return Vec::new();
1069 };
1070 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1071 pages.sort_by_key(|(n, _)| *n);
1072 let Some((_, pid)) = pages.get(index) else {
1073 return Vec::new();
1074 };
1075 page_glyphs(&doc, *pid)
1076 .into_iter()
1077 .map(|g| (g.ch, g.ll, g.lr, g.lb, g.lt))
1078 .collect()
1079}
1080
1081pub fn pdf_textlines(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
1085 let Some(doc) = load_document(bytes) else {
1086 return Vec::new();
1087 };
1088 let mut caches = DocCaches::default();
1089 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1090 pages.sort_by_key(|(n, _)| *n);
1091 pages
1092 .into_iter()
1093 .map(|(_, pid)| {
1094 let (w, h) = page_size(&doc, pid);
1095 let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1096 let cells = crate::dp_lines::line_cells(&glyphs, h, true);
1097 (w, h, cells)
1098 })
1099 .collect()
1100}
1101
1102pub fn pdf_words(bytes: &[u8]) -> Vec<(f32, f32, Vec<crate::pdfium_backend::TextCell>)> {
1107 let Some(doc) = load_document(bytes) else {
1108 return Vec::new();
1109 };
1110 let mut caches = DocCaches::default();
1111 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1112 pages.sort_by_key(|(n, _)| *n);
1113 pages
1114 .into_iter()
1115 .map(|(_, pid)| {
1116 let (w, h) = page_size(&doc, pid);
1117 let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1118 let cells = crate::dp_lines::word_cells(&glyphs, h, true);
1119 (w, h, cells)
1120 })
1121 .collect()
1122}
1123
1124#[derive(Default)]
1128pub struct PageParserCells {
1129 pub prose: Vec<crate::pdfium_backend::TextCell>,
1130 pub words: Vec<crate::pdfium_backend::TextCell>,
1131 pub code: Vec<crate::pdfium_backend::TextCell>,
1132}
1133
1134pub struct PageTextParser {
1147 doc: Document,
1148 caches: DocCaches,
1149 pages: Vec<lopdf::ObjectId>,
1151}
1152
1153impl PageTextParser {
1154 pub fn open(bytes: &[u8]) -> Option<Self> {
1156 Self::open_with_password(bytes, None)
1157 }
1158
1159 pub fn open_with_password(bytes: &[u8], password: Option<&str>) -> Option<Self> {
1163 let doc = open_document(bytes, password).ok()?;
1164 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1165 pages.sort_by_key(|(n, _)| *n);
1166 Some(Self {
1167 doc,
1168 caches: DocCaches::default(),
1169 pages: pages.into_iter().map(|(_, pid)| pid).collect(),
1170 })
1171 }
1172
1173 pub(crate) fn glyph_styles(
1181 &mut self,
1182 index: usize,
1183 ) -> Vec<crate::heading_hierarchy::GlyphStyle> {
1184 let Some(&pid) = self.pages.get(index) else {
1185 return Vec::new();
1186 };
1187 let (_w, h) = page_size(&self.doc, pid);
1188 let glyphs = page_glyphs_cached(&self.doc, pid, &mut self.caches);
1189 let styles: HashMap<u64, crate::font_style::FontStyle> = self
1192 .caches
1193 .fonts
1194 .values()
1195 .map(|f| (f.hash, f.style))
1196 .collect();
1197 glyphs
1198 .iter()
1199 .filter(|g| !g.ch.is_whitespace() && g.ll.is_finite())
1200 .map(|g| {
1201 let st = styles.get(&g.font).copied().unwrap_or_default();
1202 crate::heading_hierarchy::GlyphStyle {
1203 l: g.ll,
1204 t: h - g.lt,
1205 r: g.lr,
1206 b: h - g.lb,
1207 height: g.height(),
1208 weight_cls: crate::font_style::weight_class(st.weight),
1209 italic: st.italic,
1210 styled: st.known,
1211 }
1212 })
1213 .collect()
1214 }
1215
1216 pub fn cells(&mut self, index: usize) -> PageParserCells {
1220 let Some(&pid) = self.pages.get(index) else {
1221 return PageParserCells::default();
1222 };
1223 let (_w, h) = page_size(&self.doc, pid);
1224 let glyphs = page_glyphs_cached(&self.doc, pid, &mut self.caches);
1225 let (prose, words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1226 PageParserCells {
1227 prose,
1228 words,
1229 code: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1230 }
1231 }
1232}
1233
1234pub fn pdf_all_cells(bytes: &[u8]) -> Vec<PageParserCells> {
1239 let Some(mut parser) = PageTextParser::open(bytes) else {
1240 return Vec::new();
1241 };
1242 (0..parser.pages.len()).map(|i| parser.cells(i)).collect()
1243}
1244
1245pub fn pdf_text_pages(bytes: &[u8]) -> Vec<crate::pdfium_backend::PdfPage> {
1252 let Some(doc) = load_document(bytes) else {
1253 return Vec::new();
1254 };
1255 let mut caches = DocCaches::default();
1256 let mut pages: Vec<_> = doc.get_pages().into_iter().collect();
1257 pages.sort_by_key(|(n, _)| *n);
1258 pages
1259 .into_iter()
1260 .map(|(_, pid)| {
1261 let (w, h) = page_size(&doc, pid);
1262 let glyphs = page_glyphs_cached(&doc, pid, &mut caches);
1263 let (mut prose, mut words) = crate::dp_lines::line_and_word_cells(&glyphs, h, true);
1264 drop_overpainted_cells(&mut prose);
1265 drop_overpainted_cells(&mut words);
1266 crate::pdfium_backend::PdfPage {
1267 #[cfg(feature = "ocr-prep")]
1268 image_layout: None,
1269 width: w,
1270 height: h,
1271 scale: 1.0,
1273 cells: prose,
1274 code_cells: crate::pdfium_backend::code_cells_from_glyphs(&glyphs, h),
1275 word_cells: words,
1276 #[cfg(feature = "ocr-prep")]
1277 image: image::RgbImage::new(1, 1),
1278 links: Vec::new(),
1279 rotation: 0,
1280 }
1281 })
1282 .collect()
1283}
1284
1285fn drop_overpainted_cells(cells: &mut Vec<crate::pdfium_backend::TextCell>) {
1304 let mut paint = vec![false; cells.len()];
1305 for i in 0..cells.len() {
1306 for j in 0..cells.len() {
1307 if i == j || cells[i].text == cells[j].text {
1308 continue;
1309 }
1310 let (a, b) = (&cells[i], &cells[j]);
1311 let vo = (a.b.min(b.b) - a.t.max(b.t)).max(0.0);
1313 if vo < 0.6 * (a.b - a.t).min(b.b - b.t) {
1314 continue;
1315 }
1316 let ho = (a.r.min(b.r) - a.l.max(b.l)).max(0.0);
1318 if ho >= 0.8 * (a.r - a.l) && (a.r - a.l) <= (b.r - b.l) {
1319 paint[i] = true;
1320 paint[j] = true;
1321 }
1322 }
1323 }
1324 let mut keep = paint.iter().map(|p| !p);
1325 cells.retain(|_| keep.next().unwrap());
1326}
1327
1328#[derive(Clone, Copy)]
1332struct TextState {
1333 tc: f64,
1334 tw: f64,
1335 th: f64,
1336 tl: f64,
1337 trise: f64,
1338 fsize: f64,
1339}
1340
1341impl TextState {
1342 const INIT: TextState = TextState {
1343 tc: 0.0,
1344 tw: 0.0,
1345 th: 1.0,
1346 tl: 0.0,
1347 trise: 0.0,
1348 fsize: 0.0,
1349 };
1350}
1351
1352fn page_res(doc: &Document, page_id: lopdf::ObjectId) -> Option<&Dictionary> {
1355 let (inline, ids) = doc.get_page_resources(page_id).ok()?;
1356 if let Some(d) = inline {
1357 return Some(d);
1358 }
1359 ids.into_iter().find_map(|id| doc.get_dictionary(id).ok())
1360}
1361
1362fn fonts_from_res(
1366 doc: &Document,
1367 res: &Dictionary,
1368 caches: &mut DocCaches,
1369) -> HashMap<Vec<u8>, Arc<Font>> {
1370 let mut map = HashMap::new();
1371 let font_dict = res
1372 .get(b"Font")
1373 .ok()
1374 .and_then(|o| deref(doc, o))
1375 .and_then(|o| o.as_dict().ok());
1376 if let Some(fd) = font_dict {
1377 for (name, value) in fd.iter() {
1378 let font = match value {
1379 Object::Reference(id) => {
1380 let key = (*id, name.clone());
1381 if let Some(f) = caches.fonts.get(&key) {
1382 Arc::clone(f)
1383 } else if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1384 let f = Arc::new(parse_font(doc, name, fdict));
1385 caches.fonts.insert(key, Arc::clone(&f));
1386 f
1387 } else {
1388 continue;
1389 }
1390 }
1391 _ => {
1392 if let Some(fdict) = deref(doc, value).and_then(|o| o.as_dict().ok()) {
1393 Arc::new(parse_font(doc, name, fdict))
1394 } else {
1395 continue;
1396 }
1397 }
1398 };
1399 map.insert(name.clone(), font);
1400 }
1401 }
1402 map
1403}
1404
1405pub(crate) fn glyph_styles(
1411 bytes: &[u8],
1412 pages: &[usize],
1413) -> HashMap<usize, Vec<crate::heading_hierarchy::GlyphStyle>> {
1414 let mut out = HashMap::new();
1415 let Some(mut parser) = PageTextParser::open(bytes) else {
1416 return out;
1417 };
1418 for &page_no in pages {
1419 if page_no == 0 {
1420 continue;
1421 }
1422 let styles = parser.glyph_styles(page_no - 1);
1423 if !styles.is_empty() {
1424 out.insert(page_no, styles);
1425 }
1426 }
1427 out
1428}
1429
1430pub(crate) fn page_glyphs(doc: &Document, page_id: lopdf::ObjectId) -> Vec<Glyph> {
1431 page_glyphs_cached(doc, page_id, &mut DocCaches::default())
1432}
1433
1434fn page_glyphs_cached(
1437 doc: &Document,
1438 page_id: lopdf::ObjectId,
1439 caches: &mut DocCaches,
1440) -> Vec<Glyph> {
1441 let mut out = Vec::new();
1442 let content_bytes = doc.get_page_content(page_id);
1445 let Ok(content) = lopdf::content::Content::decode(&content_bytes) else {
1446 return out;
1447 };
1448 if let Some(res) = page_res(doc, page_id) {
1449 let pb = page_box(doc, page_id);
1452 let base = Mat {
1453 e: -(pb.l as f64),
1454 f: -(pb.b as f64),
1455 ..Mat::ID
1456 };
1457 run_content(
1458 doc,
1459 res,
1460 &content,
1461 base,
1462 TextState::INIT,
1463 0,
1464 caches,
1465 &mut out,
1466 );
1467 }
1468 out
1469}
1470
1471#[allow(clippy::too_many_arguments)]
1476fn run_content(
1477 doc: &Document,
1478 res: &Dictionary,
1479 content: &lopdf::content::Content,
1480 base_ctm: Mat,
1481 init: TextState,
1482 depth: u32,
1483 caches: &mut DocCaches,
1484 out: &mut Vec<Glyph>,
1485) {
1486 let fonts = fonts_from_res(doc, res, caches);
1487 let xobjects = res
1488 .get(b"XObject")
1489 .ok()
1490 .and_then(|o| deref(doc, o))
1491 .and_then(|o| o.as_dict().ok());
1492
1493 #[allow(clippy::type_complexity)]
1498 let mut gstate_stack: Vec<(Mat, f64, f64, f64, f64, f64, f64, Option<&Arc<Font>>)> = Vec::new();
1499 let mut ctm = base_ctm;
1500 let mut tm = Mat::ID;
1501 let mut tlm = Mat::ID;
1502 let mut font: Option<&Arc<Font>> = None;
1503 let mut fsize = init.fsize;
1504 let mut tc = init.tc; let mut tw = init.tw; let mut th = init.th; let mut tl = init.tl; let mut trise = init.trise;
1509
1510 let op_f = |operands: &[Object], i: usize| operands.get(i).and_then(num).unwrap_or(0.0);
1511
1512 for op in &content.operations {
1513 let operands = &op.operands;
1514 match op.operator.as_str() {
1515 "q" => gstate_stack.push((ctm, tc, tw, th, tl, trise, fsize, font)),
1516 "Q" => {
1517 if let Some((c, a, b, h, l, r, fs, f)) = gstate_stack.pop() {
1518 ctm = c;
1519 tc = a;
1520 tw = b;
1521 th = h;
1522 tl = l;
1523 trise = r;
1524 fsize = fs;
1525 font = f;
1526 }
1527 }
1528 "cm" => {
1529 let m = Mat {
1530 a: op_f(operands, 0),
1531 b: op_f(operands, 1),
1532 c: op_f(operands, 2),
1533 d: op_f(operands, 3),
1534 e: op_f(operands, 4),
1535 f: op_f(operands, 5),
1536 };
1537 ctm = m.then(ctm);
1538 }
1539 "BT" => {
1540 tm = Mat::ID;
1541 tlm = Mat::ID;
1542 }
1543 "ET" => {}
1544 "Tf" => {
1545 if let Some(Object::Name(n)) = operands.first() {
1546 font = fonts.get(n.as_slice());
1547 }
1548 fsize = op_f(operands, 1);
1549 }
1550 "Td" => {
1551 tlm = Mat {
1552 a: 1.0,
1553 b: 0.0,
1554 c: 0.0,
1555 d: 1.0,
1556 e: op_f(operands, 0),
1557 f: op_f(operands, 1),
1558 }
1559 .then(tlm);
1560 tm = tlm;
1561 }
1562 "TD" => {
1563 tl = -op_f(operands, 1);
1564 tlm = Mat {
1565 a: 1.0,
1566 b: 0.0,
1567 c: 0.0,
1568 d: 1.0,
1569 e: op_f(operands, 0),
1570 f: op_f(operands, 1),
1571 }
1572 .then(tlm);
1573 tm = tlm;
1574 }
1575 "Tm" => {
1576 tlm = Mat {
1577 a: op_f(operands, 0),
1578 b: op_f(operands, 1),
1579 c: op_f(operands, 2),
1580 d: op_f(operands, 3),
1581 e: op_f(operands, 4),
1582 f: op_f(operands, 5),
1583 };
1584 tm = tlm;
1585 }
1586 "T*" => {
1587 tlm = Mat {
1588 a: 1.0,
1589 b: 0.0,
1590 c: 0.0,
1591 d: 1.0,
1592 e: 0.0,
1593 f: -tl,
1594 }
1595 .then(tlm);
1596 tm = tlm;
1597 }
1598 "Tc" => tc = op_f(operands, 0),
1599 "Tw" => tw = op_f(operands, 0),
1600 "Tz" => th = op_f(operands, 0) / 100.0,
1601 "TL" => tl = op_f(operands, 0),
1602 "Ts" => trise = op_f(operands, 0),
1603 "Tj" | "'" | "\"" => {
1604 if op.operator == "'" || op.operator == "\"" {
1605 tlm = Mat {
1607 a: 1.0,
1608 b: 0.0,
1609 c: 0.0,
1610 d: 1.0,
1611 e: 0.0,
1612 f: -tl,
1613 }
1614 .then(tlm);
1615 tm = tlm;
1616 }
1617 if op.operator == "\"" {
1618 tw = op_f(operands, 0);
1621 tc = op_f(operands, 1);
1622 }
1623 if let (Some(f), Some(Object::String(s, _))) = (font, operands.last()) {
1624 show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out);
1625 }
1626 }
1627 "TJ" => {
1628 if let (Some(f), Some(Object::Array(arr))) = (font, operands.first()) {
1629 for el in arr {
1630 match el {
1631 Object::String(s, _) => {
1632 show_text(f, s, fsize, tc, tw, th, trise, &mut tm, ctm, out)
1633 }
1634 other => {
1635 if let Some(adj) = num(other) {
1636 let tx = -adj / 1000.0 * fsize * th;
1638 tm = Mat {
1639 a: 1.0,
1640 b: 0.0,
1641 c: 0.0,
1642 d: 1.0,
1643 e: tx,
1644 f: 0.0,
1645 }
1646 .then(tm);
1647 }
1648 }
1649 }
1650 }
1651 }
1652 }
1653 "Do" => {
1654 if depth >= 8 {
1657 continue;
1658 }
1659 let Some(Object::Name(n)) = operands.first() else {
1660 continue;
1661 };
1662 let obj = xobjects.and_then(|d| d.get(n.as_slice()).ok());
1663 let form_id = match obj {
1664 Some(Object::Reference(id)) => Some(*id),
1665 _ => None,
1666 };
1667 let stream = obj
1668 .and_then(|o| deref(doc, o))
1669 .and_then(|o| o.as_stream().ok());
1670 let Some(stream) = stream else { continue };
1671 let is_form = stream
1672 .dict
1673 .get(b"Subtype")
1674 .ok()
1675 .and_then(|o| o.as_name().ok())
1676 == Some(b"Form".as_slice());
1677 if !is_form {
1678 continue;
1679 }
1680 let cached = form_id.and_then(|id| caches.forms.get(&id).cloned());
1683 let form_content = match cached {
1684 Some(c) => c,
1685 None => {
1686 let Ok(data) = stream.decompressed_content() else {
1687 continue;
1688 };
1689 let Ok(c) = lopdf::content::Content::decode(&data) else {
1690 continue;
1691 };
1692 let c = Arc::new(c);
1693 if let Some(id) = form_id {
1694 caches.forms.insert(id, Arc::clone(&c));
1695 }
1696 c
1697 }
1698 };
1699 let form_mat = match stream.dict.get(b"Matrix").ok() {
1701 Some(Object::Array(a)) if a.len() == 6 => {
1702 let v: Vec<f64> = a.iter().filter_map(num).collect();
1703 if v.len() == 6 {
1704 Mat {
1705 a: v[0],
1706 b: v[1],
1707 c: v[2],
1708 d: v[3],
1709 e: v[4],
1710 f: v[5],
1711 }
1712 } else {
1713 Mat::ID
1714 }
1715 }
1716 _ => Mat::ID,
1717 };
1718 let form_res = stream
1720 .dict
1721 .get(b"Resources")
1722 .ok()
1723 .and_then(|o| deref(doc, o))
1724 .and_then(|o| o.as_dict().ok())
1725 .unwrap_or(res);
1726 let state = TextState {
1727 tc,
1728 tw,
1729 th,
1730 tl,
1731 trise,
1732 fsize,
1733 };
1734 run_content(
1735 doc,
1736 form_res,
1737 &form_content,
1738 form_mat.then(ctm),
1739 state,
1740 depth + 1,
1741 caches,
1742 out,
1743 );
1744 }
1745 _ => {}
1746 }
1747 }
1748}
1749
1750#[allow(clippy::too_many_arguments)]
1751fn show_text(
1752 font: &Font,
1753 bytes: &[u8],
1754 fsize: f64,
1755 tc: f64,
1756 tw: f64,
1757 th: f64,
1758 trise: f64,
1759 tm: &mut Mat,
1760 ctm: Mat,
1761 out: &mut Vec<Glyph>,
1762) {
1763 for code in codes(font, bytes) {
1764 let (text, w) = font.decode_code(code);
1765 let w0 = w / 1000.0; let scale = Mat {
1768 a: fsize * th,
1769 b: 0.0,
1770 c: 0.0,
1771 d: fsize,
1772 e: 0.0,
1773 f: trise,
1774 };
1775 let trm = scale.then(*tm).then(ctm);
1776 let (desc, asc) = (font.descent / 1000.0, font.ascent / 1000.0);
1778 let (x0, y0) = trm.apply(0.0, desc);
1779 let (x1, y1) = trm.apply(w0, desc);
1780 let (x2, y2) = trm.apply(w0, asc);
1781 let (x3, y3) = trm.apply(0.0, asc);
1782 let upright = trm.a > 0.0 && trm.b.abs() <= 1e-6 * trm.a;
1790 let (left, bot, right, top, quad) = if upright {
1791 (x0.min(x1), y0.min(y3), x0.max(x1), y0.max(y3), None)
1792 } else {
1793 let (xs, ys) = ([x0, x1, x2, x3], [y0, y1, y2, y3]);
1794 let fold = |v: [f64; 4], f: fn(f64, f64) -> f64| v.into_iter().reduce(f).unwrap();
1795 let q = [x0, y0, x1, y1, x2, y2, x3, y3].map(|v| v as f32);
1796 (
1797 fold(xs, f64::min),
1798 fold(ys, f64::min),
1799 fold(xs, f64::max),
1800 fold(ys, f64::max),
1801 Some(q),
1802 )
1803 };
1804 if let Some(s) = text {
1805 for ch in s.chars() {
1807 if ch != '\u{0}' {
1808 out.push(Glyph {
1809 ch,
1810 l: left as f32,
1811 b: bot as f32,
1812 r: right as f32,
1813 t: top as f32,
1814 ll: left as f32,
1815 lb: bot as f32,
1816 lr: right as f32,
1817 lt: top as f32,
1818 font: font.hash,
1819 quad,
1820 });
1821 }
1822 }
1823 }
1824 let is_space = !font.two_byte && code == 32;
1826 let tx = (w0 * fsize + tc + if is_space { tw } else { 0.0 }) * th;
1827 *tm = Mat {
1828 a: 1.0,
1829 b: 0.0,
1830 c: 0.0,
1831 d: 1.0,
1832 e: tx,
1833 f: 0.0,
1834 }
1835 .then(*tm);
1836 }
1837}
1838
1839fn simple_encoding_table(doc: &Document, fdict: &Dictionary) -> HashMap<u8, char> {
1843 let enc = fdict.get(b"Encoding").ok().and_then(|o| deref(doc, o));
1844 let base_name = match enc {
1845 Some(Object::Name(n)) => n.clone(),
1846 Some(Object::Dictionary(d)) => d
1847 .get(b"BaseEncoding")
1848 .ok()
1849 .and_then(|o| o.as_name().ok())
1850 .map(|n| n.to_vec())
1851 .unwrap_or_default(),
1852 _ => Vec::new(),
1853 };
1854 let mut m = if base_name == b"MacRomanEncoding" {
1855 macroman_table()
1856 } else if base_name.is_empty() {
1857 tex_math_builtin(fdict).unwrap_or_else(winansi_table)
1864 } else {
1865 winansi_table()
1866 };
1867 if let Some(Object::Dictionary(d)) = enc {
1869 if let Some(Object::Array(diffs)) = d.get(b"Differences").ok().and_then(|o| deref(doc, o)) {
1870 let mut code = 0u8;
1871 for el in diffs {
1872 match el {
1873 Object::Integer(i) => code = *i as u8,
1874 Object::Name(name) => {
1875 if let Some(ch) = glyph_name_to_char(name) {
1876 m.insert(code, ch);
1877 }
1878 code = code.wrapping_add(1);
1879 }
1880 _ => {}
1881 }
1882 }
1883 }
1884 }
1885 m
1886}
1887
1888fn tex_math_builtin(fdict: &Dictionary) -> Option<HashMap<u8, char>> {
1894 const CMSY: [char; 128] = [
1895 '−', '·', '×', '∗', '÷', '⋄', '±', '∓', '⊕', '⊖', '⊗', '⊘', '⊙', '◯', '∘', '•', '≍', '≡',
1896 '⊆', '⊇', '≤', '≥', '≼', '≽', '∼', '≈', '⊂', '⊃', '≪', '≫', '≺', '≻', '←', '→', '↑', '↓',
1897 '↔', '↗', '↘', '≃', '⇐', '⇒', '⇑', '⇓', '⇔', '↖', '↙', '∝', '′', '∞', '∈', '∋', '△', '▽',
1898 '\u{338}', '↦', '∀', '∃', '¬', '∅', 'ℜ', 'ℑ', '⊤', '⊥', 'ℵ', 'A', 'B', 'C', 'D', 'E', 'F',
1899 'G', 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X',
1900 'Y', 'Z', '∪', '∩', '⊎', '∧', '∨', '⊢', '⊣', '⌊', '⌋', '⌈', '⌉', '{', '}', '⟨', '⟩', '|',
1901 '∥', '↕', '⇕', '\\', '≀', '√', '∐', '∇', '∫', '⊔', '⊓', '⊑', '⊒', '§', '†', '‡', '¶', '♣',
1902 '♢', '♡', '♠',
1903 ];
1904 const CMMI: [char; 128] = [
1905 'Γ', 'Δ', 'Θ', 'Λ', 'Ξ', 'Π', 'Σ', 'Υ', 'Φ', 'Ψ', 'Ω', 'α', 'β', 'γ', 'δ', 'ε', 'ζ', 'η',
1906 'θ', 'ι', 'κ', 'λ', 'μ', 'ν', 'ξ', 'π', 'ρ', 'σ', 'τ', 'υ', 'φ', 'χ', 'ψ', 'ω', 'ϵ', 'ϑ',
1907 'ϖ', 'ϱ', 'ς', 'ϕ', '↼', '↽', '⇀', '⇁', '↩', '↪', '▷', '◁', '0', '1', '2', '3', '4', '5',
1908 '6', '7', '8', '9', '.', ',', '<', '/', '>', '⋆', '∂', 'A', 'B', 'C', 'D', 'E', 'F', 'G',
1909 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y',
1910 'Z', '♭', '♮', '♯', '⌣', '⌢', 'ℓ', 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k',
1911 'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', 'ı', 'ȷ', '℘',
1912 '\u{20d7}', '⁀',
1913 ];
1914 let name = base_font_name(fdict)?;
1915 let up = name.to_ascii_uppercase();
1916 let table: &[char; 128] = if up.starts_with(b"CMSY") || up.starts_with(b"CMBSY") {
1917 &CMSY
1918 } else if up.starts_with(b"CMMI") {
1919 &CMMI
1920 } else {
1921 return None;
1922 };
1923 Some(
1924 table
1925 .iter()
1926 .enumerate()
1927 .map(|(i, &c)| (i as u8, c))
1928 .collect(),
1929 )
1930}
1931
1932pub(crate) fn glyph_name_to_char(name: &[u8]) -> Option<char> {
1937 let s = std::str::from_utf8(name).ok()?;
1938 if let Some(hex) = s.strip_prefix("uni") {
1939 if let Ok(cp) = u32::from_str_radix(hex.get(0..4)?, 16) {
1940 return char::from_u32(cp);
1941 }
1942 }
1943 if s.len() == 1 {
1945 let b = s.as_bytes()[0];
1946 if b.is_ascii_alphabetic() {
1947 return Some(b as char);
1948 }
1949 }
1950 let resolved = match s {
1951 "space" => ' ',
1952 "exclam" => '!',
1953 "quotedbl" => '"',
1954 "numbersign" => '#',
1955 "dollar" => '$',
1956 "percent" => '%',
1957 "ampersand" => '&',
1958 "quotesingle" => '\'',
1959 "parenleft" => '(',
1960 "parenright" => ')',
1961 "asterisk" => '*',
1962 "plus" => '+',
1963 "comma" => ',',
1964 "hyphen" => '-',
1965 "period" => '.',
1966 "slash" => '/',
1967 "zero" => '0',
1968 "one" => '1',
1969 "two" => '2',
1970 "three" => '3',
1971 "four" => '4',
1972 "five" => '5',
1973 "six" => '6',
1974 "seven" => '7',
1975 "eight" => '8',
1976 "nine" => '9',
1977 "colon" => ':',
1978 "semicolon" => ';',
1979 "less" => '<',
1980 "equal" => '=',
1981 "greater" => '>',
1982 "question" => '?',
1983 "at" => '@',
1984 "bracketleft" => '[',
1985 "backslash" => '\\',
1986 "bracketright" => ']',
1987 "asciicircum" => '^',
1988 "underscore" => '_',
1989 "grave" => '`',
1990 "braceleft" => '{',
1991 "bar" => '|',
1992 "braceright" => '}',
1993 "asciitilde" => '~',
1994 "bullet" => '\u{2022}',
1995 "periodcentered" => '\u{00B7}',
1996 "endash" => '\u{2013}',
1997 "emdash" => '\u{2014}',
1998 "quoteright" => '\u{2019}',
1999 "quoteleft" => '\u{2018}',
2000 "quotedblleft" => '\u{201C}',
2001 "quotedblright" => '\u{201D}',
2002 "quotedblbase" => '\u{201E}',
2003 "quotesinglbase" => '\u{201A}',
2004 "ff" => '\u{FB00}',
2009 "fi" => '\u{FB01}',
2010 "fl" => '\u{FB02}',
2011 "ffi" => '\u{FB03}',
2012 "ffl" => '\u{FB04}',
2013 "ft" => '\u{FB05}',
2014 "st" => '\u{FB06}',
2015 "degree" => '\u{00B0}',
2016 "trademark" => '\u{2122}',
2017 "registered" => '\u{00AE}',
2018 "copyright" => '\u{00A9}',
2019 "ellipsis" => '\u{2026}',
2020 "minus" => '\u{2212}',
2021 "fraction" => '\u{2044}',
2022 "nbspace" => '\u{00A0}',
2023 "alpha" => '\u{03B1}',
2028 "beta" => '\u{03B2}',
2029 "gamma" => '\u{03B3}',
2030 "delta" => '\u{03B4}',
2031 "epsilon" | "epsilon1" => '\u{03B5}',
2032 "zeta" => '\u{03B6}',
2033 "eta" => '\u{03B7}',
2034 "theta" | "theta1" => '\u{03B8}',
2035 "iota" => '\u{03B9}',
2036 "kappa" => '\u{03BA}',
2037 "lambda" => '\u{03BB}',
2038 "mu" => '\u{03BC}',
2039 "nu" => '\u{03BD}',
2040 "xi" => '\u{03BE}',
2041 "omicron" => '\u{03BF}',
2042 "pi" | "pi1" => '\u{03C0}',
2043 "rho" | "rho1" => '\u{03C1}',
2044 "sigma" => '\u{03C3}',
2045 "sigma1" => '\u{03C2}',
2046 "tau" => '\u{03C4}',
2047 "upsilon" => '\u{03C5}',
2048 "phi" | "phi1" => '\u{03C6}',
2049 "chi" => '\u{03C7}',
2050 "psi" => '\u{03C8}',
2051 "omega" | "omega1" => '\u{03C9}',
2052 "Gamma" => '\u{0393}',
2053 "Delta" => '\u{0394}',
2054 "Theta" => '\u{0398}',
2055 "Lambda" => '\u{039B}',
2056 "Xi" => '\u{039E}',
2057 "Pi" => '\u{03A0}',
2058 "Sigma" => '\u{03A3}',
2059 "Upsilon" => '\u{03A5}',
2060 "Phi" => '\u{03A6}',
2061 "Psi" => '\u{03A8}',
2062 "Omega" => '\u{03A9}',
2063 "lessequal" => '\u{2264}',
2064 "greaterequal" => '\u{2265}',
2065 "notequal" => '\u{2260}',
2066 "approxequal" => '\u{2248}',
2067 "equivalence" => '\u{2261}',
2068 "element" => '\u{2208}',
2069 "plusminus" => '\u{00B1}',
2070 "multiply" => '\u{00D7}',
2071 "divide" => '\u{00F7}',
2072 "infinity" => '\u{221E}',
2073 "partialdiff" => '\u{2202}',
2074 "gradient" => '\u{2207}',
2075 "summation" => '\u{2211}',
2076 "product" => '\u{220F}',
2077 "integral" => '\u{222B}',
2078 "radical" => '\u{221A}',
2079 "proportional" => '\u{221D}',
2080 "arrowright" => '\u{2192}',
2081 "arrowleft" => '\u{2190}',
2082 "arrowup" => '\u{2191}',
2083 "arrowdown" => '\u{2193}',
2084 "arrowboth" => '\u{2194}',
2085 "arrowdblright" => '\u{21D2}',
2086 "logicaland" => '\u{2227}',
2087 "logicalor" => '\u{2228}',
2088 "intersection" => '\u{2229}',
2089 "union" => '\u{222A}',
2090 "similar" => '\u{223C}',
2091 "congruent" => '\u{2245}',
2092 "dotmath" => '\u{22C5}',
2093 "asteriskmath" => '\u{2217}',
2094 _ => {
2095 if let Some((base, _)) = s.split_once('.') {
2097 if !base.is_empty() {
2098 return glyph_name_to_char(base.as_bytes());
2099 }
2100 }
2101 return None;
2102 }
2103 };
2104 Some(resolved)
2105}
2106
2107fn winansi_table() -> HashMap<u8, char> {
2109 let mut m = HashMap::new();
2110 for b in 0x20u8..=0x7e {
2111 m.insert(b, b as char);
2112 }
2113 let extra: &[(u8, char)] = &[
2115 (0x91, '\u{2018}'),
2116 (0x92, '\u{2019}'),
2117 (0x93, '\u{201C}'),
2118 (0x94, '\u{201D}'),
2119 (0x95, '\u{2022}'),
2120 (0x96, '\u{2013}'),
2121 (0x97, '\u{2014}'),
2122 (0x85, '\u{2026}'),
2123 (0xA0, '\u{00A0}'),
2124 ];
2125 for &(b, c) in extra {
2126 m.insert(b, c);
2127 }
2128 for b in 0xA1u8..=0xFF {
2129 m.entry(b).or_insert(b as char);
2130 }
2131 m
2132}
2133
2134fn macroman_table() -> HashMap<u8, char> {
2137 let mut m = HashMap::new();
2138 for b in 0x20u8..=0x7e {
2139 m.insert(b, b as char);
2140 }
2141 let high: &[(u8, char)] = &[
2142 (0xA5, '\u{2022}'), (0xD0, '\u{2013}'), (0xD1, '\u{2014}'), (0xD2, '\u{201C}'),
2146 (0xD3, '\u{201D}'),
2147 (0xD4, '\u{2018}'),
2148 (0xD5, '\u{2019}'),
2149 (0xCA, '\u{00A0}'),
2150 (0xC9, '\u{2026}'),
2151 (0xDE, '\u{FB01}'),
2152 (0xDF, '\u{FB02}'),
2153 ];
2154 for &(b, c) in high {
2155 m.insert(b, c);
2156 }
2157 m
2158}
2159
2160#[cfg(test)]
2161mod page_box_frame {
2162 use super::*;
2163
2164 fn pdf(boxes: &str, x: f32, y: f32) -> Vec<u8> {
2167 let content = format!("BT /F1 12 Tf {x} {y} Td (First printing) Tj ET\n");
2168 let objs: Vec<String> = vec![
2169 "<</Type/Catalog/Pages 2 0 R>>".into(),
2170 format!("<</Type/Pages/Kids[3 0 R]/Count 1{boxes}>>"),
2171 "<</Type/Page/Parent 2 0 R/Contents 4 0 R/Resources<</Font<</F1 5 0 R>>>>>>".into(),
2172 format!("<</Length {}>>stream\n{content}endstream", content.len()),
2173 "<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".into(),
2174 ];
2175 let mut out = b"%PDF-1.4\n".to_vec();
2176 let mut offsets = Vec::new();
2177 for (i, body) in objs.iter().enumerate() {
2178 offsets.push(out.len());
2179 out.extend_from_slice(format!("{} 0 obj{body}endobj\n", i + 1).as_bytes());
2180 }
2181 let xref_at = out.len();
2182 out.extend_from_slice(
2183 format!("xref\n0 {}\n0000000000 65535 f \n", objs.len() + 1).as_bytes(),
2184 );
2185 for off in &offsets {
2186 out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2187 }
2188 out.extend_from_slice(
2189 format!(
2190 "trailer<</Size {}/Root 1 0 R>>\nstartxref\n{xref_at}\n%%EOF\n",
2191 objs.len() + 1
2192 )
2193 .as_bytes(),
2194 );
2195 out
2196 }
2197
2198 fn only_page(bytes: &[u8]) -> (PageBox, Vec<Glyph>) {
2199 let doc = load_document(bytes).expect("loads");
2200 let pid = *doc.get_pages().values().next().expect("one page");
2201 (page_box(&doc, pid), page_glyphs(&doc, pid))
2202 }
2203
2204 #[test]
2209 fn glyphs_count_from_the_cropbox_corner_like_pdfium() {
2210 let (pb, shifted) = only_page(&pdf(
2211 "/MediaBox[-56.505 -58.25 576.303 723.31]/CropBox[1.095 -0.65 518.703 665.71]",
2212 37.0 + 1.095,
2213 58.0 - 0.65,
2214 ));
2215 assert!((pb.l - 1.095).abs() < 1e-3 && (pb.b + 0.65).abs() < 1e-3);
2216 assert!(
2217 (pb.w - 517.608).abs() < 1e-3 && (pb.h - 666.36).abs() < 1e-3,
2218 "{pb:?}"
2219 );
2220 let (pb0, plain) = only_page(&pdf("/MediaBox[0 0 517.608 666.36]", 37.0, 58.0));
2221 assert!((pb0.w - pb.w).abs() < 1e-3 && (pb0.h - pb.h).abs() < 1e-3);
2222 assert_eq!(shifted.len(), plain.len());
2223 assert!(!plain.is_empty());
2224 for (a, b) in shifted.iter().zip(&plain) {
2225 assert!(
2226 (a.l - b.l).abs() < 1e-3 && (a.b - b.b).abs() < 1e-3,
2227 "{:?} vs {:?}",
2228 (a.l, a.b),
2229 (b.l, b.b)
2230 );
2231 }
2232 assert!((plain[0].l - 37.0).abs() < 1e-3, "{}", plain[0].l);
2233 }
2234
2235 #[test]
2238 fn page_box_follows_pdfium_fallbacks() {
2239 let (pb, _) = only_page(&pdf("", 10.0, 10.0));
2240 assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 612.0, 792.0));
2241 let (pb, _) = only_page(&pdf(
2242 "/MediaBox[0 0 500 700]/CropBox[-100 100 600 900]",
2243 10.0,
2244 10.0,
2245 ));
2246 assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 100.0, 500.0, 600.0));
2247 let (pb, _) = only_page(&pdf(
2248 "/MediaBox[0 0 500 700]/CropBox[800 800 900 900]",
2249 10.0,
2250 10.0,
2251 ));
2252 assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 500.0, 700.0));
2253 let (pb, _) = only_page(&pdf("/MediaBox[500 700 0 0]", 10.0, 10.0));
2255 assert_eq!((pb.l, pb.b, pb.w, pb.h), (0.0, 0.0, 500.0, 700.0));
2256 }
2257}
2258
2259#[cfg(test)]
2260mod xref_repair {
2261 fn pdf_with_xref(two_byte_eol: bool) -> Vec<u8> {
2265 let content = b"BT /F1 12 Tf 72 700 Td (Invoice 922769430725) Tj ET\n";
2266 let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2267 let objs: Vec<Vec<u8>> = vec![
2268 b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2269 b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2270 b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2271 /Resources<</Font<</F1 5 0 R>>>>>>"
2272 .to_vec(),
2273 [stream.as_slice(), content.as_slice(), b"endstream"].concat(),
2274 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>".to_vec(),
2275 ];
2276
2277 let mut out = b"%PDF-1.4\n".to_vec();
2278 let mut offsets = Vec::new();
2279 for (i, body) in objs.iter().enumerate() {
2280 offsets.push(out.len());
2281 out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2282 out.extend_from_slice(body);
2283 out.extend_from_slice(b"endobj\n");
2284 }
2285 let xref_at = out.len();
2286 let eol: &[u8] = if two_byte_eol { b" \n" } else { b"\n" };
2287 out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2288 out.extend_from_slice(b"0000000000 65535 f");
2289 out.extend_from_slice(eol);
2290 for off in &offsets {
2291 out.extend_from_slice(format!("{off:010} 00000 n").as_bytes());
2292 out.extend_from_slice(eol);
2293 }
2294 out.extend_from_slice(
2295 format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2296 );
2297 out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2298 out
2299 }
2300
2301 #[test]
2307 fn short_xref_entries_still_parse() {
2308 let good = pdf_with_xref(true);
2309 let broken = pdf_with_xref(false);
2310 assert!(
2311 broken.len() < good.len(),
2312 "the broken file is the shorter one"
2313 );
2314 assert!(
2315 lopdf::Document::load_mem(&good).is_ok(),
2316 "the control file must load unaided"
2317 );
2318 assert!(
2319 lopdf::Document::load_mem(&broken).is_err(),
2320 "lopdf rejects 19-byte entries — if this ever passes, drop the repair"
2321 );
2322
2323 let cells = |b: &[u8]| -> Vec<String> {
2324 super::pdf_textlines(b)
2325 .into_iter()
2326 .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2327 .collect()
2328 };
2329 let from_good = cells(&good);
2330 assert!(
2331 from_good.iter().any(|t| t.contains("922769430725")),
2332 "control text: {from_good:?}"
2333 );
2334 assert_eq!(
2335 cells(&broken),
2336 from_good,
2337 "repair must match the good parse"
2338 );
2339 }
2340
2341 #[test]
2346 fn overstated_stream_length_still_yields_content() {
2347 let good = pdf_with_xref(true);
2348 let broken = {
2351 let at = good
2352 .windows(8)
2353 .position(|w| w == b"/Length ")
2354 .expect("a /Length")
2355 + 8;
2356 let digits = good[at..].iter().take_while(|c| c.is_ascii_digit()).count();
2357 let n: usize = std::str::from_utf8(&good[at..at + digits])
2358 .unwrap()
2359 .parse()
2360 .unwrap();
2361 let inflated = (n + 1).to_string();
2362 assert_eq!(inflated.len(), digits, "keep the digit count");
2363 let mut b = good.clone();
2364 b[at..at + digits].copy_from_slice(inflated.as_bytes());
2365 b
2366 };
2367 assert_eq!(broken.len(), good.len(), "the defect must not move bytes");
2368 let raw = lopdf::Document::load_mem(&broken).expect("still loads");
2370 assert!(
2371 raw.get_pages()
2372 .into_values()
2373 .all(|p| raw.get_page_content(p).is_empty()),
2374 "lopdf should drop the stream — if it stops, drop this repair"
2375 );
2376 let text = |b: &[u8]| -> Vec<String> {
2378 super::pdf_textlines(b)
2379 .into_iter()
2380 .flat_map(|(_, _, c)| c.into_iter().map(|c| c.text))
2381 .collect()
2382 };
2383 let expected = text(&good);
2384 assert!(!expected.is_empty(), "control must produce text");
2385 assert_eq!(text(&broken), expected);
2386 }
2387
2388 #[test]
2392 fn repair_declines_when_padding_would_move_objects() {
2393 let mut incremental = pdf_with_xref(false);
2394 incremental.extend_from_slice(b"6 0 obj<</Type/Whatever>>endobj\n");
2395 let declined = super::pad_short_xref_entries(&incremental).unwrap_err();
2396 assert!(
2397 declined.contains("object follows the xref"),
2398 "reason: {declined}"
2399 );
2400 }
2401}
2402
2403#[cfg(test)]
2409mod base14_fonts {
2410 fn pdf_with_font(fontdict: &[u8], text: &[u8]) -> Vec<u8> {
2412 let content = [b"BT /F1 12 Tf 72 700 Td (".as_slice(), text, b") Tj ET\n"].concat();
2413 pdf_with_content(fontdict, &content)
2414 }
2415
2416 fn pdf_with_content(fontdict: &[u8], content: &[u8]) -> Vec<u8> {
2418 let stream = format!("<</Length {}>>stream\n", content.len()).into_bytes();
2419 let objs: Vec<Vec<u8>> = vec![
2420 b"<</Type/Catalog/Pages 2 0 R>>".to_vec(),
2421 b"<</Type/Pages/Kids[3 0 R]/Count 1>>".to_vec(),
2422 b"<</Type/Page/Parent 2 0 R/MediaBox[0 0 595 842]/Contents 4 0 R\
2423 /Resources<</Font<</F1 5 0 R>>>>>>"
2424 .to_vec(),
2425 [stream.as_slice(), content, b"endstream"].concat(),
2426 fontdict.to_vec(),
2427 ];
2428 let mut out = b"%PDF-1.4\n".to_vec();
2429 let mut offsets = Vec::new();
2430 for (i, body) in objs.iter().enumerate() {
2431 offsets.push(out.len());
2432 out.extend_from_slice(format!("{} 0 obj", i + 1).as_bytes());
2433 out.extend_from_slice(body);
2434 out.extend_from_slice(b"endobj\n");
2435 }
2436 let xref_at = out.len();
2437 out.extend_from_slice(format!("xref\n0 {}\n", objs.len() + 1).as_bytes());
2438 out.extend_from_slice(b"0000000000 65535 f \n");
2439 for off in &offsets {
2440 out.extend_from_slice(format!("{off:010} 00000 n \n").as_bytes());
2441 }
2442 out.extend_from_slice(
2443 format!("trailer<</Size {}/Root 1 0 R>>\n", objs.len() + 1).as_bytes(),
2444 );
2445 out.extend_from_slice(format!("startxref\n{xref_at}\n%%EOF\n").as_bytes());
2446 out
2447 }
2448
2449 fn cells(pdf: &[u8]) -> Vec<crate::pdfium_backend::TextCell> {
2451 super::pdf_textlines(pdf)
2452 .into_iter()
2453 .flat_map(|(_, _, c)| c)
2454 .collect()
2455 }
2456
2457 #[test]
2464 fn rotated_text_matrix_reads_in_order() {
2465 const HELV: &[u8] = b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>";
2466 const TEXT: &str = "Upright text on a rotated page.";
2467 for (tm, [l, b, r, t]) in [
2468 ("0 14 -14 0 200 100", [189.9, 100.0, 202.9, 289.1]), ("-14 0 0 -14 350 300", [160.9, 289.9, 350.0, 302.9]), ("0 -14 14 0 200 500", [197.1, 310.9, 210.1, 500.0]), (
2472 "9.8995 9.8995 -9.8995 9.8995 100 100",
2473 [92.9, 98.0, 235.8, 240.8],
2474 ), ] {
2476 let content = format!("BT /F1 1 Tf {tm} Tm ({TEXT}) Tj ET\n");
2477 let pdf = pdf_with_content(HELV, content.as_bytes());
2478 let cs = cells(&pdf);
2479 assert_eq!(cs.len(), 1, "{tm}: {cs:?}");
2480 let c = &cs[0];
2481 assert_eq!(c.text, TEXT, "{tm}");
2482 let got = [c.l, 842.0 - c.b, c.r, 842.0 - c.t];
2487 for (g, w) in got.iter().zip([l, b, r, t]) {
2488 assert!(
2489 (g - w).abs() < 1.0,
2490 "{tm}: box {got:?} vs docling-parse {:?}",
2491 [l, b, r, t]
2492 );
2493 }
2494 let words: Vec<String> = super::pdf_words(&pdf)
2495 .into_iter()
2496 .flat_map(|(_, _, c)| c)
2497 .map(|c| c.text)
2498 .collect();
2499 assert_eq!(words, TEXT.split(' ').collect::<Vec<_>>(), "{tm}");
2500 }
2501 }
2502
2503 #[test]
2506 fn upright_glyphs_carry_no_quad() {
2507 let only_page = |pdf: &[u8]| {
2508 let doc = super::load_document(pdf).expect("loads");
2509 let pid = *doc.get_pages().values().next().expect("one page");
2510 super::page_glyphs(&doc, pid)
2511 };
2512 let pdf = pdf_with_font(b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>", b"Plain");
2513 let glyphs = only_page(&pdf);
2514 assert!(!glyphs.is_empty());
2515 assert!(glyphs.iter().all(|g| g.quad.is_none() && g.lr > g.ll));
2516 let pdf = pdf_with_content(
2518 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>",
2519 b"BT /F1 1 Tf 0 14 -14 0 200 100 Tm (W) Tj ET\n",
2520 );
2521 let glyphs = only_page(&pdf);
2522 let g = &glyphs[0];
2523 assert!(g.quad.is_some());
2524 assert!((g.height() - 14.0).abs() < 0.05, "{}", g.height());
2526 }
2527
2528 #[test]
2530 fn standard14_faces_get_builtin_widths() {
2531 for fontdict in [
2532 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>".as_slice(),
2534 b"<</Type/Font/Subtype/Type1/BaseFont/Times-BoldItalic>>",
2536 b"<</Type/Font/Subtype/TrueType/BaseFont/Arial,Bold>>",
2538 b"<</Type/Font/Subtype/Type1/BaseFont/ABCDEF+Courier-Oblique>>",
2539 ] {
2540 let pdf = pdf_with_font(fontdict, b"Words have width now");
2541 let cs = cells(&pdf);
2542 let text: String = cs
2543 .iter()
2544 .map(|c| c.text.as_str())
2545 .collect::<Vec<_>>()
2546 .join(" ");
2547 assert!(
2548 text.contains("Words have width now"),
2549 "{}: text lost: {text:?}",
2550 String::from_utf8_lossy(fontdict)
2551 );
2552 assert!(
2553 cs.iter().all(|c| c.r > c.l),
2554 "{}: zero-width cells: {cs:?}",
2555 String::from_utf8_lossy(fontdict)
2556 );
2557 }
2558 }
2559
2560 #[test]
2563 fn explicit_widths_win_and_unknown_faces_are_untouched() {
2564 let explicit = pdf_with_font(
2568 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/FirstChar 65\
2569 /Widths[100 100 100 100]/Encoding/WinAnsiEncoding>>",
2570 b"ABBA",
2571 );
2572 let builtin = pdf_with_font(
2573 b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica/Encoding/WinAnsiEncoding>>",
2574 b"ABBA",
2575 );
2576 let w = |pdf: &[u8]| {
2577 let cs = cells(pdf);
2578 assert_eq!(cs.len(), 1, "one word cell: {cs:?}");
2579 cs[0].r - cs[0].l
2580 };
2581 let (we, wb) = (w(&explicit), w(&builtin));
2582 assert!(
2583 (we - 4.8).abs() < 0.1,
2584 "explicit widths must win: got {we}, want 4×100×12/1000"
2585 );
2586 assert!(
2587 wb > 2.0 * we,
2588 "built-in Helvetica is much wider: {wb} vs {we}"
2589 );
2590
2591 let unknown = pdf_with_font(
2595 b"<</Type/Font/Subtype/Type1/BaseFont/FancyCorp-Display>>",
2596 b"Mystery",
2597 );
2598 let cs = cells(&unknown);
2599 let text: String = cs.iter().map(|c| c.text.as_str()).collect();
2600 assert!(text.contains("Mystery"), "text still decodes: {cs:?}");
2601 }
2602}
2603
2604#[cfg(test)]
2605mod overpainted {
2606 use crate::pdfium_backend::TextCell;
2607
2608 fn cell(text: &str, l: f32, t: f32, r: f32, b: f32) -> TextCell {
2609 TextCell {
2610 text: text.into(),
2611 l,
2612 t,
2613 r,
2614 b,
2615 }
2616 }
2617
2618 #[test]
2622 fn stacked_logo_glyphs_are_dropped() {
2623 let mut cells = vec![
2624 cell("\"", 72.7, 21.5, 86.4, 31.5),
2625 cell("==", 59.4, 21.5, 99.6, 31.5),
2626 cell("Herr", 65.2, 151.3, 81.7, 161.3),
2627 ];
2628 super::drop_overpainted_cells(&mut cells);
2629 assert_eq!(cells.len(), 1, "cells: {cells:?}");
2630 assert_eq!(cells[0].text, "Herr");
2631 }
2632
2633 #[test]
2637 fn prose_and_double_draw_are_kept() {
2638 let mut cells = vec![
2639 cell("Telefon", 354.3, 133.2, 381.5, 143.2),
2640 cell("0676/2000", 387.3, 133.2, 428.7, 143.2),
2641 cell("Bold", 100.0, 50.0, 130.0, 60.0),
2642 cell("Bold", 100.3, 50.0, 130.3, 60.0),
2643 ];
2644 super::drop_overpainted_cells(&mut cells);
2645 assert_eq!(cells.len(), 4);
2646 }
2647}
2648
2649#[cfg(test)]
2650mod vestigial_layer {
2651 use crate::pdfium_backend::{PdfPage, TextCell};
2652
2653 fn page_with(texts: &[&str]) -> PdfPage {
2654 let cells = texts
2655 .iter()
2656 .enumerate()
2657 .map(|(i, t)| TextCell {
2658 text: t.to_string(),
2659 l: 10.0,
2660 t: 10.0 + 12.0 * i as f32,
2661 r: 90.0,
2662 b: 20.0 + 12.0 * i as f32,
2663 })
2664 .collect();
2665 PdfPage::from_cells(595.0, 842.0, 1.0, cells)
2666 }
2667
2668 #[test]
2673 fn typed_in_form_fields_are_not_a_text_layer() {
2674 let pages = vec![
2675 page_with(&["03", "05", "2025"]),
2676 page_with(&[]),
2677 page_with(&[]),
2678 ];
2679 assert!(super::text_layer_is_vestigial(&pages));
2680 assert!(super::text_layer_is_vestigial(&[page_with(&[])]));
2681 }
2682
2683 #[test]
2686 fn sparse_but_real_documents_pass() {
2687 let one_pager = vec![page_with(&[
2688 "Confidential briefing",
2689 "Prepared for the board meeting",
2690 "Do not distribute",
2691 ])];
2692 assert!(!super::text_layer_is_vestigial(&one_pager));
2693 }
2694}