1use crate::{GedcomError, GedcomErrorKind, Limits, fault};
18use std::fmt;
19
20const HEADER_SCAN_BYTES: usize = 8 * 1024;
23
24#[derive(Clone, Copy, Debug, Eq, PartialEq)]
26#[non_exhaustive]
27#[cfg_attr(feature = "serde", derive(serde::Serialize))]
28pub enum GedcomEncoding {
29 Ascii,
31 Utf8,
33 Utf16Be,
35 Utf16Le,
37 Ansel,
39 Ansi,
41}
42
43impl GedcomEncoding {
44 #[must_use]
46 pub const fn label(self) -> &'static str {
47 match self {
48 Self::Ascii => "ASCII",
49 Self::Utf8 => "UTF-8",
50 Self::Utf16Be => "UTF-16 (big-endian)",
51 Self::Utf16Le => "UTF-16 (little-endian)",
52 Self::Ansel => "ANSEL",
53 Self::Ansi => "ANSI (Windows-1252)",
54 }
55 }
56}
57
58impl fmt::Display for GedcomEncoding {
59 fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
60 formatter.write_str(self.label())
61 }
62}
63
64#[derive(Clone, Debug, Default, Eq, PartialEq)]
66#[non_exhaustive]
67#[cfg_attr(feature = "serde", derive(serde::Serialize))]
68#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))]
69pub struct EncodingReport {
70 pub used: Option<GedcomEncoding>,
72 pub declared: Option<String>,
74 pub byte_order_mark: bool,
76 pub undecodable_bytes: usize,
78 pub warnings: Vec<String>,
80}
81
82impl EncodingReport {
83 #[must_use]
87 pub fn for_encoding(encoding: GedcomEncoding) -> Self {
88 Self {
89 used: Some(encoding),
90 ..Self::default()
91 }
92 }
93
94 #[must_use]
96 pub fn summary(&self) -> String {
97 let used = self.used.map_or("unknown", GedcomEncoding::label);
98 let mut text = format!("Read as {used}");
99 if self.byte_order_mark {
100 text.push_str(" (byte-order mark)");
101 }
102 match &self.declared {
103 Some(declared) => {
104 let _ = fmt::Write::write_fmt(&mut text, format_args!(", declared {declared}"));
105 }
106 None => text.push_str(", no CHAR declared"),
107 }
108 if self.undecodable_bytes > 0 {
109 let _ = fmt::Write::write_fmt(
110 &mut text,
111 format_args!(", {} byte(s) undecodable", self.undecodable_bytes),
112 );
113 }
114 text
115 }
116}
117
118pub fn decode_gedcom(bytes: &[u8]) -> Result<(String, EncodingReport), GedcomError> {
131 decode_gedcom_with(bytes, Limits::DEFAULT)
132}
133
134pub fn decode_gedcom_with(
140 bytes: &[u8],
141 limits: Limits,
142) -> Result<(String, EncodingReport), GedcomError> {
143 if bytes.len() > limits.input_bytes {
144 return Err(fault(
145 0,
146 GedcomErrorKind::Limit,
147 format!(
148 "the file is {} MB and the limit is {} MB",
149 bytes.len() / (1024 * 1024),
150 limits.input_bytes / (1024 * 1024)
151 ),
152 ));
153 }
154
155 let mut report = EncodingReport::default();
156
157 let (body, marked) = strip_byte_order_mark(bytes);
158 report.byte_order_mark = marked.is_some();
159
160 if let Some(encoding) = marked {
162 report.declared = declared_charset(body, encoding);
163 let text = decode_as(body, encoding, &mut report);
164 finish(encoding, &mut report);
165 return Ok((text, report));
166 }
167
168 if let Some(encoding) = unmarked_utf16(body) {
171 report.declared = declared_charset(body, encoding);
172 report.warnings.push(format!(
173 "File is {encoding} but carries no byte-order mark."
174 ));
175 let text = decode_as(body, encoding, &mut report);
176 finish(encoding, &mut report);
177 return Ok((text, report));
178 }
179
180 report.declared = declared_charset(body, GedcomEncoding::Utf8);
182 let declared = report
183 .declared
184 .as_ref()
185 .map(|value| value.trim().to_owned());
186 let encoding = match declared.as_deref() {
187 Some(declared) => match_declared(declared, body, &mut report),
188 None => {
189 if std::str::from_utf8(body).is_ok() {
193 GedcomEncoding::Utf8
194 } else {
195 report.warnings.push(
196 "File declares no CHAR and is not valid UTF-8; read as ANSEL, the GEDCOM 5.5 default."
197 .to_owned(),
198 );
199 GedcomEncoding::Ansel
200 }
201 }
202 };
203
204 let text = decode_as(body, encoding, &mut report);
205 finish(encoding, &mut report);
206 Ok((text, report))
207}
208
209fn finish(encoding: GedcomEncoding, report: &mut EncodingReport) {
210 report.used = Some(encoding);
211 if report.undecodable_bytes > 0 {
212 report.warnings.push(format!(
213 "{} byte(s) had no mapping in {encoding} and were replaced with U+FFFD.",
214 report.undecodable_bytes
215 ));
216 }
217}
218
219fn match_declared(declared: &str, body: &[u8], report: &mut EncodingReport) -> GedcomEncoding {
222 let normalized = declared
223 .chars()
224 .filter(char::is_ascii_alphanumeric)
225 .collect::<String>()
226 .to_ascii_uppercase();
227
228 match normalized.as_str() {
229 "UTF8" => {
230 if std::str::from_utf8(body).is_ok() {
231 GedcomEncoding::Utf8
232 } else {
233 report.warnings.push(
236 "File declares UTF-8 but contains invalid UTF-8; read as Windows-1252 instead."
237 .to_owned(),
238 );
239 GedcomEncoding::Ansi
240 }
241 }
242 "ANSEL" => GedcomEncoding::Ansel,
243 "ANSI" | "WINDOWS1252" | "CP1252" | "IBMWINDOWS" => GedcomEncoding::Ansi,
244 "ASCII" | "USASCII" | "ANSIZ3947" => GedcomEncoding::Ascii,
245 "UNICODE" | "UTF16" => {
246 report.warnings.push(
249 "File declares UNICODE but has no byte-order mark and no UTF-16 structure; read as UTF-8."
250 .to_owned(),
251 );
252 GedcomEncoding::Utf8
253 }
254 _ => {
255 report.warnings.push(format!(
256 "Unrecognized CHAR value {declared:?}; read as UTF-8."
257 ));
258 GedcomEncoding::Utf8
259 }
260 }
261}
262
263fn strip_byte_order_mark(bytes: &[u8]) -> (&[u8], Option<GedcomEncoding>) {
264 if let Some(rest) = bytes.strip_prefix(&[0xEF, 0xBB, 0xBF]) {
265 return (rest, Some(GedcomEncoding::Utf8));
266 }
267 if let Some(rest) = bytes.strip_prefix(&[0xFE, 0xFF]) {
268 return (rest, Some(GedcomEncoding::Utf16Be));
269 }
270 if let Some(rest) = bytes.strip_prefix(&[0xFF, 0xFE]) {
271 return (rest, Some(GedcomEncoding::Utf16Le));
272 }
273 (bytes, None)
274}
275
276const fn unmarked_utf16(bytes: &[u8]) -> Option<GedcomEncoding> {
277 match bytes {
278 [0x00, second, ..] if *second != 0x00 => Some(GedcomEncoding::Utf16Be),
279 [first, 0x00, ..] if *first != 0x00 => Some(GedcomEncoding::Utf16Le),
280 _ => None,
281 }
282}
283
284fn declared_charset(bytes: &[u8], encoding: GedcomEncoding) -> Option<String> {
289 let head = &bytes[..bytes.len().min(HEADER_SCAN_BYTES)];
290 let text = match encoding {
291 GedcomEncoding::Utf16Be | GedcomEncoding::Utf16Le => {
292 let mut scratch = EncodingReport::default();
293 decode_utf16(head, encoding == GedcomEncoding::Utf16Be, &mut scratch)
294 }
295 _ => head.iter().map(|byte| char::from(*byte)).collect(),
296 };
297
298 text.lines()
299 .map(str::trim_end)
300 .find_map(|line| line.strip_prefix("1 CHAR "))
301 .map(str::trim)
302 .filter(|value| !value.is_empty())
303 .map(str::to_owned)
304}
305
306fn decode_as(bytes: &[u8], encoding: GedcomEncoding, report: &mut EncodingReport) -> String {
307 match encoding {
308 GedcomEncoding::Utf16Be => decode_utf16(bytes, true, report),
309 GedcomEncoding::Utf16Le => decode_utf16(bytes, false, report),
310 GedcomEncoding::Ansel => decode_ansel(bytes, report),
311 GedcomEncoding::Ansi => decode_windows_1252(bytes),
312 GedcomEncoding::Ascii => decode_ascii(bytes, report),
313 GedcomEncoding::Utf8 => std::str::from_utf8(bytes).map_or_else(
314 |_| {
315 let text = String::from_utf8_lossy(bytes).into_owned();
316 report.undecodable_bytes += text.matches('\u{FFFD}').count();
317 text
318 },
319 str::to_owned,
320 ),
321 }
322}
323
324fn decode_ascii(bytes: &[u8], report: &mut EncodingReport) -> String {
325 bytes
326 .iter()
327 .map(|byte| {
328 if byte.is_ascii() {
329 char::from(*byte)
330 } else {
331 report.undecodable_bytes += 1;
332 '\u{FFFD}'
333 }
334 })
335 .collect()
336}
337
338fn decode_utf16(bytes: &[u8], big_endian: bool, report: &mut EncodingReport) -> String {
339 if !bytes.len().is_multiple_of(2) {
340 report
341 .warnings
342 .push("UTF-16 file has an odd number of bytes; the final byte was ignored.".to_owned());
343 }
344 let units = bytes
345 .as_chunks::<2>()
346 .0
347 .iter()
348 .map(|pair| {
349 if big_endian {
350 u16::from_be_bytes([pair[0], pair[1]])
351 } else {
352 u16::from_le_bytes([pair[0], pair[1]])
353 }
354 })
355 .collect::<Vec<_>>();
356
357 let mut text = String::with_capacity(units.len());
358 for unit in char::decode_utf16(units) {
359 text.push(unit.unwrap_or_else(|_| {
360 report.undecodable_bytes += 1;
361 '\u{FFFD}'
362 }));
363 }
364 text
365}
366
367fn decode_windows_1252(bytes: &[u8]) -> String {
369 bytes
370 .iter()
371 .map(|byte| match byte {
372 0x80..=0x9F => WINDOWS_1252_HIGH[usize::from(byte - 0x80)],
373 other => char::from(*other),
374 })
375 .collect()
376}
377
378pub const WINDOWS_1252_HIGH: [char; 32] = [
381 '\u{20AC}', '\u{0081}', '\u{201A}', '\u{0192}', '\u{201E}', '\u{2026}', '\u{2020}', '\u{2021}',
382 '\u{02C6}', '\u{2030}', '\u{0160}', '\u{2039}', '\u{0152}', '\u{008D}', '\u{017D}', '\u{008F}',
383 '\u{0090}', '\u{2018}', '\u{2019}', '\u{201C}', '\u{201D}', '\u{2022}', '\u{2013}', '\u{2014}',
384 '\u{02DC}', '\u{2122}', '\u{0161}', '\u{203A}', '\u{0153}', '\u{009D}', '\u{017E}', '\u{0178}',
385];
386
387fn decode_ansel(bytes: &[u8], report: &mut EncodingReport) -> String {
394 let mut text = String::with_capacity(bytes.len());
395 let mut marks = Vec::new();
396 let mut index = 0;
397
398 while index < bytes.len() {
399 let byte = bytes[index];
400 index += 1;
401
402 if let Some(mark) = ansel_combining(byte) {
403 marks.push(mark);
404 continue;
405 }
406
407 let base = if byte.is_ascii() {
408 char::from(byte)
409 } else if let Some(character) = ansel_graphic(byte) {
410 character
411 } else {
412 report.undecodable_bytes += 1;
413 '\u{FFFD}'
414 };
415
416 text.push(base);
417 text.extend(marks.iter().copied());
420 marks.clear();
421 }
422
423 text.extend(marks);
424 text
425}
426
427const fn ansel_combining(byte: u8) -> Option<char> {
429 Some(match byte {
430 0xE0 => '\u{0309}', 0xE1 => '\u{0300}', 0xE2 => '\u{0301}', 0xE3 => '\u{0302}', 0xE4 => '\u{0303}', 0xE5 => '\u{0304}', 0xE6 => '\u{0306}', 0xE7 => '\u{0307}', 0xE8 => '\u{0308}', 0xE9 => '\u{030C}', 0xEA => '\u{030A}', 0xEB => '\u{FE20}', 0xEC => '\u{FE21}', 0xED => '\u{0315}', 0xEE => '\u{030B}', 0xEF => '\u{0310}', 0xF0 => '\u{0327}', 0xF1 => '\u{0328}', 0xF2 => '\u{0323}', 0xF3 => '\u{0324}', 0xF4 => '\u{0325}', 0xF5 => '\u{0333}', 0xF6 => '\u{0332}', 0xF7 => '\u{0326}', 0xF8 => '\u{031C}', 0xF9 => '\u{032E}', 0xFA => '\u{FE22}', 0xFB => '\u{FE23}', 0xFE => '\u{0313}', _ => return None,
460 })
461}
462
463const fn ansel_graphic(byte: u8) -> Option<char> {
466 Some(match byte {
467 0xA1 => 'Ł',
468 0xA2 => 'Ø',
469 0xA3 => 'Đ',
470 0xA4 => 'Þ',
471 0xA5 => 'Æ',
472 0xA6 => 'Œ',
473 0xA7 => '\u{02B9}', 0xA8 => '·',
475 0xA9 => '\u{266D}', 0xAA => '®',
477 0xAB => '±',
478 0xAC => 'Ơ',
479 0xAD => 'Ư',
480 0xAE => '\u{02BB}', 0xB0 => '\u{02BC}', 0xB1 => 'ł',
483 0xB2 => 'ø',
484 0xB3 => 'đ',
485 0xB4 => 'þ',
486 0xB5 => 'æ',
487 0xB6 => 'œ',
488 0xB7 => '\u{02BA}', 0xB8 => 'ı',
490 0xB9 => '£',
491 0xBA => 'ð',
492 0xBC => 'ơ',
493 0xBD => 'ư',
494 0xBE => '\u{25A1}', 0xBF => '\u{25A0}', 0xC0 => '°',
497 0xC1 => '\u{2113}', 0xC2 => '\u{2117}', 0xC3 => '©',
500 0xC4 => '\u{266F}', 0xC5 => '¿',
502 0xC6 => '¡',
503 0xCD => 'e', 0xCE => 'o', 0xCF => 'ß',
506 _ => return None,
507 })
508}
509
510#[cfg(test)]
511mod tests {
512 use super::*;
513
514 #[test]
515 fn a_utf8_byte_order_mark_is_stripped_and_reported() {
516 let bytes = b"\xEF\xBB\xBF0 HEAD\n0 TRLR\n";
517
518 let (text, report) = decode_gedcom(bytes).expect("decode");
519
520 assert!(
521 text.starts_with("0 HEAD"),
522 "mark must not survive: {text:?}"
523 );
524 assert!(report.byte_order_mark);
525 assert_eq!(report.used, Some(GedcomEncoding::Utf8));
526 }
527
528 #[test]
529 fn utf16_is_read_with_or_without_a_byte_order_mark() {
530 let marked = b"\xFF\xFE0\x00 \x00H\x00E\x00A\x00D\x00";
531 let (text, report) = decode_gedcom(marked).expect("decode marked");
532 assert_eq!(text, "0 HEAD");
533 assert_eq!(report.used, Some(GedcomEncoding::Utf16Le));
534 assert!(report.byte_order_mark);
535
536 let bare = b"\x000\x00 \x00H\x00E\x00A\x00D";
537 let (text, report) = decode_gedcom(bare).expect("decode bare");
538 assert_eq!(text, "0 HEAD");
539 assert_eq!(report.used, Some(GedcomEncoding::Utf16Be));
540 assert!(!report.byte_order_mark);
541 assert!(
542 !report.warnings.is_empty(),
543 "a missing mark is worth saying"
544 );
545 }
546
547 #[test]
548 fn ansel_puts_a_diacritic_after_the_letter_it_modifies() {
549 let bytes = b"0 HEAD\n1 CHAR ANSEL\n0 @I1@ INDI\n1 NAME Jos\xE2e\n";
551
552 let (text, report) = decode_gedcom(bytes).expect("decode");
553
554 assert_eq!(report.used, Some(GedcomEncoding::Ansel));
555 assert_eq!(report.declared.as_deref(), Some("ANSEL"));
556 assert!(text.contains("Jose\u{0301}"), "got {text:?}");
557 assert_eq!(report.undecodable_bytes, 0);
558 }
559
560 #[test]
561 fn ansel_graphic_characters_decode() {
562 let bytes = b"0 HEAD\n1 CHAR ANSEL\n1 NOTE \xA2 \xB2 \xCF\n";
563
564 let (text, _) = decode_gedcom(bytes).expect("decode");
565
566 assert!(text.contains("Ø ø ß"), "got {text:?}");
567 }
568
569 #[test]
570 fn a_declared_charset_that_the_bytes_contradict_is_reported_not_obeyed() {
571 let bytes = b"0 HEAD\n1 CHAR UTF-8\n1 NOTE caf\xE9\n";
573
574 let (text, report) = decode_gedcom(bytes).expect("decode");
575
576 assert_eq!(report.used, Some(GedcomEncoding::Ansi));
577 assert_eq!(report.declared.as_deref(), Some("UTF-8"));
578 assert!(text.contains("café"), "got {text:?}");
579 assert!(
580 report
581 .warnings
582 .iter()
583 .any(|warning| warning.contains("declares UTF-8")),
584 "{:?}",
585 report.warnings
586 );
587 }
588
589 #[test]
590 fn an_undeclared_file_prefers_utf8_and_falls_back_to_ansel() {
591 let utf8 = "0 HEAD\n1 NOTE café\n".as_bytes();
592 let (text, report) = decode_gedcom(utf8).expect("decode utf8");
593 assert_eq!(report.used, Some(GedcomEncoding::Utf8));
594 assert!(text.contains("café"));
595 assert!(report.declared.is_none());
596
597 let ansel = b"0 HEAD\n1 NOTE caf\xE2e\n";
598 let (text, report) = decode_gedcom(ansel).expect("decode ansel");
599 assert_eq!(report.used, Some(GedcomEncoding::Ansel));
600 assert!(text.contains("cafe\u{0301}"), "got {text:?}");
601 }
602
603 #[test]
604 fn utf16_surrogate_pairs_decode_to_the_characters_they_spell() {
605 let mut bytes = vec![0xFFu8, 0xFE];
608 for unit in "0 HEAD\n1 NOTE \u{1D11E}\n".encode_utf16() {
609 bytes.extend_from_slice(&unit.to_le_bytes());
610 }
611
612 let (text, report) = decode_gedcom(&bytes).expect("decode");
613
614 assert_eq!(report.undecodable_bytes, 0);
615 assert!(text.contains('\u{1D11E}'), "got {text:?}");
616 }
617
618 #[test]
619 fn a_char_line_beyond_the_header_scan_window_falls_back_honestly() {
620 use std::fmt::Write as _;
624 let mut text = String::from("0 HEAD\n");
625 for index in 0..600 {
626 let _ = writeln!(
627 text,
628 "1 NOTE padding line number {index} to push the declaration far down"
629 );
630 }
631 text.push_str("1 CHAR ANSEL\n0 TRLR\n");
632
633 let (_, report) = decode_gedcom(text.as_bytes()).expect("decode");
634
635 assert_eq!(report.declared, None, "the declaration is out of reach");
636 assert_eq!(report.used, Some(GedcomEncoding::Utf8));
637 }
638
639 #[test]
640 fn an_oversized_file_is_refused_with_the_limit_named() {
641 let limits = Limits {
644 input_bytes: 2 * 1024 * 1024,
645 ..Limits::DEFAULT
646 };
647 let bytes = vec![b'0'; limits.input_bytes + 1];
648
649 let error = decode_gedcom_with(&bytes, limits).expect_err("must refuse");
650
651 let message = error.to_string();
652 assert!(message.contains("2 MB"), "limit must be named: {message}");
653 }
654
655 #[test]
656 fn undecodable_bytes_are_counted_rather_than_failing_the_import() {
657 let bytes = b"0 HEAD\n1 CHAR ANSEL\n1 NOTE \xD0\n";
658
659 let (text, report) = decode_gedcom(bytes).expect("decode");
660
661 assert_eq!(report.undecodable_bytes, 1);
662 assert!(text.contains('\u{FFFD}'));
663 assert!(!report.warnings.is_empty());
664 }
665}