1use crate::{GedcomError, GedcomErrorKind, Limits, fault};
18use std::fmt;
19
20const HEADER_SCAN_BYTES: usize = 8 * 1024;
23
24#[derive(Clone, Copy, Debug, Eq, PartialEq)]
26#[non_exhaustive]
27#[cfg_attr(feature = "serde", derive(serde::Serialize))]
28#[cfg_attr(feature = "ts", derive(ts_rs::TS))]
29pub enum GedcomEncoding {
30 Ascii,
32 Utf8,
34 Utf16Be,
36 Utf16Le,
38 Ansel,
40 Ansi,
42}
43
44impl GedcomEncoding {
45 #[must_use]
47 pub const fn label(self) -> &'static str {
48 match self {
49 Self::Ascii => "ASCII",
50 Self::Utf8 => "UTF-8",
51 Self::Utf16Be => "UTF-16 (big-endian)",
52 Self::Utf16Le => "UTF-16 (little-endian)",
53 Self::Ansel => "ANSEL",
54 Self::Ansi => "ANSI (Windows-1252)",
55 }
56 }
57}
58
59impl fmt::Display for GedcomEncoding {
60 fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
61 formatter.write_str(self.label())
62 }
63}
64
65#[derive(Clone, Debug, Default, Eq, PartialEq)]
67#[non_exhaustive]
68#[cfg_attr(feature = "serde", derive(serde::Serialize))]
69#[cfg_attr(feature = "ts", derive(ts_rs::TS))]
70#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))]
71pub struct EncodingReport {
72 pub used: Option<GedcomEncoding>,
74 pub declared: Option<String>,
76 pub byte_order_mark: bool,
78 pub undecodable_bytes: usize,
80 pub warnings: Vec<String>,
82}
83
84impl EncodingReport {
85 #[must_use]
89 pub fn for_encoding(encoding: GedcomEncoding) -> Self {
90 Self {
91 used: Some(encoding),
92 ..Self::default()
93 }
94 }
95
96 #[must_use]
98 pub fn summary(&self) -> String {
99 let used = self.used.map_or("unknown", GedcomEncoding::label);
100 let mut text = format!("Read as {used}");
101 if self.byte_order_mark {
102 text.push_str(" (byte-order mark)");
103 }
104 match &self.declared {
105 Some(declared) => {
106 let _ = fmt::Write::write_fmt(&mut text, format_args!(", declared {declared}"));
107 }
108 None => text.push_str(", no CHAR declared"),
109 }
110 if self.undecodable_bytes > 0 {
111 let _ = fmt::Write::write_fmt(
112 &mut text,
113 format_args!(", {} byte(s) undecodable", self.undecodable_bytes),
114 );
115 }
116 text
117 }
118}
119
120pub fn decode_gedcom(bytes: &[u8]) -> Result<(String, EncodingReport), GedcomError> {
133 decode_gedcom_with(bytes, Limits::DEFAULT)
134}
135
136pub fn decode_gedcom_with(
142 bytes: &[u8],
143 limits: Limits,
144) -> Result<(String, EncodingReport), GedcomError> {
145 if bytes.len() > limits.input_bytes {
146 return Err(fault(
147 0,
148 GedcomErrorKind::Limit,
149 format!(
150 "the file is {} MB and the limit is {} MB",
151 bytes.len() / (1024 * 1024),
152 limits.input_bytes / (1024 * 1024)
153 ),
154 ));
155 }
156
157 let mut report = EncodingReport::default();
158
159 let (body, marked) = strip_byte_order_mark(bytes);
160 report.byte_order_mark = marked.is_some();
161
162 if let Some(encoding) = marked {
164 report.declared = declared_charset(body, encoding);
165 let text = decode_as(body, encoding, &mut report);
166 finish(encoding, &mut report);
167 return Ok((text, report));
168 }
169
170 if let Some(encoding) = unmarked_utf16(body) {
173 report.declared = declared_charset(body, encoding);
174 report.warnings.push(format!(
175 "File is {encoding} but carries no byte-order mark."
176 ));
177 let text = decode_as(body, encoding, &mut report);
178 finish(encoding, &mut report);
179 return Ok((text, report));
180 }
181
182 report.declared = declared_charset(body, GedcomEncoding::Utf8);
184 let declared = report
185 .declared
186 .as_ref()
187 .map(|value| value.trim().to_owned());
188 let encoding = match declared.as_deref() {
189 Some(declared) => match_declared(declared, body, &mut report),
190 None => {
191 if std::str::from_utf8(body).is_ok() {
195 GedcomEncoding::Utf8
196 } else {
197 report.warnings.push(
198 "File declares no CHAR and is not valid UTF-8; read as ANSEL, the GEDCOM 5.5 default."
199 .to_owned(),
200 );
201 GedcomEncoding::Ansel
202 }
203 }
204 };
205
206 let text = decode_as(body, encoding, &mut report);
207 finish(encoding, &mut report);
208 Ok((text, report))
209}
210
211fn finish(encoding: GedcomEncoding, report: &mut EncodingReport) {
212 report.used = Some(encoding);
213 if report.undecodable_bytes > 0 {
214 report.warnings.push(format!(
215 "{} byte(s) had no mapping in {encoding} and were replaced with U+FFFD.",
216 report.undecodable_bytes
217 ));
218 }
219}
220
221fn match_declared(declared: &str, body: &[u8], report: &mut EncodingReport) -> GedcomEncoding {
224 let normalized = declared
225 .chars()
226 .filter(char::is_ascii_alphanumeric)
227 .collect::<String>()
228 .to_ascii_uppercase();
229
230 match normalized.as_str() {
231 "UTF8" => {
232 if std::str::from_utf8(body).is_ok() {
233 GedcomEncoding::Utf8
234 } else {
235 report.warnings.push(
238 "File declares UTF-8 but contains invalid UTF-8; read as Windows-1252 instead."
239 .to_owned(),
240 );
241 GedcomEncoding::Ansi
242 }
243 }
244 "ANSEL" => GedcomEncoding::Ansel,
245 "ANSI" | "WINDOWS1252" | "CP1252" | "IBMWINDOWS" => GedcomEncoding::Ansi,
246 "ASCII" | "USASCII" | "ANSIZ3947" => GedcomEncoding::Ascii,
247 "UNICODE" | "UTF16" => {
248 report.warnings.push(
251 "File declares UNICODE but has no byte-order mark and no UTF-16 structure; read as UTF-8."
252 .to_owned(),
253 );
254 GedcomEncoding::Utf8
255 }
256 _ => {
257 report.warnings.push(format!(
258 "Unrecognized CHAR value {declared:?}; read as UTF-8."
259 ));
260 GedcomEncoding::Utf8
261 }
262 }
263}
264
265fn strip_byte_order_mark(bytes: &[u8]) -> (&[u8], Option<GedcomEncoding>) {
266 if let Some(rest) = bytes.strip_prefix(&[0xEF, 0xBB, 0xBF]) {
267 return (rest, Some(GedcomEncoding::Utf8));
268 }
269 if let Some(rest) = bytes.strip_prefix(&[0xFE, 0xFF]) {
270 return (rest, Some(GedcomEncoding::Utf16Be));
271 }
272 if let Some(rest) = bytes.strip_prefix(&[0xFF, 0xFE]) {
273 return (rest, Some(GedcomEncoding::Utf16Le));
274 }
275 (bytes, None)
276}
277
278const fn unmarked_utf16(bytes: &[u8]) -> Option<GedcomEncoding> {
279 match bytes {
280 [0x00, second, ..] if *second != 0x00 => Some(GedcomEncoding::Utf16Be),
281 [first, 0x00, ..] if *first != 0x00 => Some(GedcomEncoding::Utf16Le),
282 _ => None,
283 }
284}
285
286fn declared_charset(bytes: &[u8], encoding: GedcomEncoding) -> Option<String> {
291 let head = &bytes[..bytes.len().min(HEADER_SCAN_BYTES)];
292 let text = match encoding {
293 GedcomEncoding::Utf16Be | GedcomEncoding::Utf16Le => {
294 let mut scratch = EncodingReport::default();
295 decode_utf16(head, encoding == GedcomEncoding::Utf16Be, &mut scratch)
296 }
297 _ => head.iter().map(|byte| char::from(*byte)).collect(),
298 };
299
300 text.lines()
301 .map(str::trim_end)
302 .find_map(|line| line.strip_prefix("1 CHAR "))
303 .map(str::trim)
304 .filter(|value| !value.is_empty())
305 .map(str::to_owned)
306}
307
308fn decode_as(bytes: &[u8], encoding: GedcomEncoding, report: &mut EncodingReport) -> String {
309 match encoding {
310 GedcomEncoding::Utf16Be => decode_utf16(bytes, true, report),
311 GedcomEncoding::Utf16Le => decode_utf16(bytes, false, report),
312 GedcomEncoding::Ansel => decode_ansel(bytes, report),
313 GedcomEncoding::Ansi => decode_windows_1252(bytes),
314 GedcomEncoding::Ascii => decode_ascii(bytes, report),
315 GedcomEncoding::Utf8 => std::str::from_utf8(bytes).map_or_else(
316 |_| {
317 let text = String::from_utf8_lossy(bytes).into_owned();
318 report.undecodable_bytes += text.matches('\u{FFFD}').count();
319 text
320 },
321 str::to_owned,
322 ),
323 }
324}
325
326fn decode_ascii(bytes: &[u8], report: &mut EncodingReport) -> String {
327 bytes
328 .iter()
329 .map(|byte| {
330 if byte.is_ascii() {
331 char::from(*byte)
332 } else {
333 report.undecodable_bytes += 1;
334 '\u{FFFD}'
335 }
336 })
337 .collect()
338}
339
340fn decode_utf16(bytes: &[u8], big_endian: bool, report: &mut EncodingReport) -> String {
341 if !bytes.len().is_multiple_of(2) {
342 report
343 .warnings
344 .push("UTF-16 file has an odd number of bytes; the final byte was ignored.".to_owned());
345 }
346 let units = bytes
347 .as_chunks::<2>()
348 .0
349 .iter()
350 .map(|pair| {
351 if big_endian {
352 u16::from_be_bytes([pair[0], pair[1]])
353 } else {
354 u16::from_le_bytes([pair[0], pair[1]])
355 }
356 })
357 .collect::<Vec<_>>();
358
359 let mut text = String::with_capacity(units.len());
360 for unit in char::decode_utf16(units) {
361 text.push(unit.unwrap_or_else(|_| {
362 report.undecodable_bytes += 1;
363 '\u{FFFD}'
364 }));
365 }
366 text
367}
368
369fn decode_windows_1252(bytes: &[u8]) -> String {
371 bytes
372 .iter()
373 .map(|byte| match byte {
374 0x80..=0x9F => WINDOWS_1252_HIGH[usize::from(byte - 0x80)],
375 other => char::from(*other),
376 })
377 .collect()
378}
379
380pub const WINDOWS_1252_HIGH: [char; 32] = [
383 '\u{20AC}', '\u{0081}', '\u{201A}', '\u{0192}', '\u{201E}', '\u{2026}', '\u{2020}', '\u{2021}',
384 '\u{02C6}', '\u{2030}', '\u{0160}', '\u{2039}', '\u{0152}', '\u{008D}', '\u{017D}', '\u{008F}',
385 '\u{0090}', '\u{2018}', '\u{2019}', '\u{201C}', '\u{201D}', '\u{2022}', '\u{2013}', '\u{2014}',
386 '\u{02DC}', '\u{2122}', '\u{0161}', '\u{203A}', '\u{0153}', '\u{009D}', '\u{017E}', '\u{0178}',
387];
388
389fn decode_ansel(bytes: &[u8], report: &mut EncodingReport) -> String {
396 let mut text = String::with_capacity(bytes.len());
397 let mut marks = Vec::new();
398 let mut index = 0;
399
400 while index < bytes.len() {
401 let byte = bytes[index];
402 index += 1;
403
404 if let Some(mark) = ansel_combining(byte) {
405 marks.push(mark);
406 continue;
407 }
408
409 let base = if byte.is_ascii() {
410 char::from(byte)
411 } else if let Some(character) = ansel_graphic(byte) {
412 character
413 } else {
414 report.undecodable_bytes += 1;
415 '\u{FFFD}'
416 };
417
418 text.push(base);
419 text.extend(marks.iter().copied());
422 marks.clear();
423 }
424
425 text.extend(marks);
426 text
427}
428
429const fn ansel_combining(byte: u8) -> Option<char> {
431 Some(match byte {
432 0xE0 => '\u{0309}', 0xE1 => '\u{0300}', 0xE2 => '\u{0301}', 0xE3 => '\u{0302}', 0xE4 => '\u{0303}', 0xE5 => '\u{0304}', 0xE6 => '\u{0306}', 0xE7 => '\u{0307}', 0xE8 => '\u{0308}', 0xE9 => '\u{030C}', 0xEA => '\u{030A}', 0xEB => '\u{FE20}', 0xEC => '\u{FE21}', 0xED => '\u{0315}', 0xEE => '\u{030B}', 0xEF => '\u{0310}', 0xF0 => '\u{0327}', 0xF1 => '\u{0328}', 0xF2 => '\u{0323}', 0xF3 => '\u{0324}', 0xF4 => '\u{0325}', 0xF5 => '\u{0333}', 0xF6 => '\u{0332}', 0xF7 => '\u{0326}', 0xF8 => '\u{031C}', 0xF9 => '\u{032E}', 0xFA => '\u{FE22}', 0xFB => '\u{FE23}', 0xFE => '\u{0313}', _ => return None,
462 })
463}
464
465const fn ansel_graphic(byte: u8) -> Option<char> {
468 Some(match byte {
469 0xA1 => 'Ł',
470 0xA2 => 'Ø',
471 0xA3 => 'Đ',
472 0xA4 => 'Þ',
473 0xA5 => 'Æ',
474 0xA6 => 'Œ',
475 0xA7 => '\u{02B9}', 0xA8 => '·',
477 0xA9 => '\u{266D}', 0xAA => '®',
479 0xAB => '±',
480 0xAC => 'Ơ',
481 0xAD => 'Ư',
482 0xAE => '\u{02BB}', 0xB0 => '\u{02BC}', 0xB1 => 'ł',
485 0xB2 => 'ø',
486 0xB3 => 'đ',
487 0xB4 => 'þ',
488 0xB5 => 'æ',
489 0xB6 => 'œ',
490 0xB7 => '\u{02BA}', 0xB8 => 'ı',
492 0xB9 => '£',
493 0xBA => 'ð',
494 0xBC => 'ơ',
495 0xBD => 'ư',
496 0xBE => '\u{25A1}', 0xBF => '\u{25A0}', 0xC0 => '°',
499 0xC1 => '\u{2113}', 0xC2 => '\u{2117}', 0xC3 => '©',
502 0xC4 => '\u{266F}', 0xC5 => '¿',
504 0xC6 => '¡',
505 0xCD => 'e', 0xCE => 'o', 0xCF => 'ß',
508 _ => return None,
509 })
510}
511
512#[cfg(test)]
513mod tests {
514 use super::*;
515
516 #[test]
517 fn a_utf8_byte_order_mark_is_stripped_and_reported() {
518 let bytes = b"\xEF\xBB\xBF0 HEAD\n0 TRLR\n";
519
520 let (text, report) = decode_gedcom(bytes).expect("decode");
521
522 assert!(
523 text.starts_with("0 HEAD"),
524 "mark must not survive: {text:?}"
525 );
526 assert!(report.byte_order_mark);
527 assert_eq!(report.used, Some(GedcomEncoding::Utf8));
528 }
529
530 #[test]
531 fn utf16_is_read_with_or_without_a_byte_order_mark() {
532 let marked = b"\xFF\xFE0\x00 \x00H\x00E\x00A\x00D\x00";
533 let (text, report) = decode_gedcom(marked).expect("decode marked");
534 assert_eq!(text, "0 HEAD");
535 assert_eq!(report.used, Some(GedcomEncoding::Utf16Le));
536 assert!(report.byte_order_mark);
537
538 let bare = b"\x000\x00 \x00H\x00E\x00A\x00D";
539 let (text, report) = decode_gedcom(bare).expect("decode bare");
540 assert_eq!(text, "0 HEAD");
541 assert_eq!(report.used, Some(GedcomEncoding::Utf16Be));
542 assert!(!report.byte_order_mark);
543 assert!(
544 !report.warnings.is_empty(),
545 "a missing mark is worth saying"
546 );
547 }
548
549 #[test]
550 fn ansel_puts_a_diacritic_after_the_letter_it_modifies() {
551 let bytes = b"0 HEAD\n1 CHAR ANSEL\n0 @I1@ INDI\n1 NAME Jos\xE2e\n";
553
554 let (text, report) = decode_gedcom(bytes).expect("decode");
555
556 assert_eq!(report.used, Some(GedcomEncoding::Ansel));
557 assert_eq!(report.declared.as_deref(), Some("ANSEL"));
558 assert!(text.contains("Jose\u{0301}"), "got {text:?}");
559 assert_eq!(report.undecodable_bytes, 0);
560 }
561
562 #[test]
563 fn ansel_graphic_characters_decode() {
564 let bytes = b"0 HEAD\n1 CHAR ANSEL\n1 NOTE \xA2 \xB2 \xCF\n";
565
566 let (text, _) = decode_gedcom(bytes).expect("decode");
567
568 assert!(text.contains("Ø ø ß"), "got {text:?}");
569 }
570
571 #[test]
572 fn a_declared_charset_that_the_bytes_contradict_is_reported_not_obeyed() {
573 let bytes = b"0 HEAD\n1 CHAR UTF-8\n1 NOTE caf\xE9\n";
575
576 let (text, report) = decode_gedcom(bytes).expect("decode");
577
578 assert_eq!(report.used, Some(GedcomEncoding::Ansi));
579 assert_eq!(report.declared.as_deref(), Some("UTF-8"));
580 assert!(text.contains("café"), "got {text:?}");
581 assert!(
582 report
583 .warnings
584 .iter()
585 .any(|warning| warning.contains("declares UTF-8")),
586 "{:?}",
587 report.warnings
588 );
589 }
590
591 #[test]
592 fn an_undeclared_file_prefers_utf8_and_falls_back_to_ansel() {
593 let utf8 = "0 HEAD\n1 NOTE café\n".as_bytes();
594 let (text, report) = decode_gedcom(utf8).expect("decode utf8");
595 assert_eq!(report.used, Some(GedcomEncoding::Utf8));
596 assert!(text.contains("café"));
597 assert!(report.declared.is_none());
598
599 let ansel = b"0 HEAD\n1 NOTE caf\xE2e\n";
600 let (text, report) = decode_gedcom(ansel).expect("decode ansel");
601 assert_eq!(report.used, Some(GedcomEncoding::Ansel));
602 assert!(text.contains("cafe\u{0301}"), "got {text:?}");
603 }
604
605 #[test]
606 fn utf16_surrogate_pairs_decode_to_the_characters_they_spell() {
607 let mut bytes = vec![0xFFu8, 0xFE];
610 for unit in "0 HEAD\n1 NOTE \u{1D11E}\n".encode_utf16() {
611 bytes.extend_from_slice(&unit.to_le_bytes());
612 }
613
614 let (text, report) = decode_gedcom(&bytes).expect("decode");
615
616 assert_eq!(report.undecodable_bytes, 0);
617 assert!(text.contains('\u{1D11E}'), "got {text:?}");
618 }
619
620 #[test]
621 fn a_char_line_beyond_the_header_scan_window_falls_back_honestly() {
622 use std::fmt::Write as _;
626 let mut text = String::from("0 HEAD\n");
627 for index in 0..600 {
628 let _ = writeln!(
629 text,
630 "1 NOTE padding line number {index} to push the declaration far down"
631 );
632 }
633 text.push_str("1 CHAR ANSEL\n0 TRLR\n");
634
635 let (_, report) = decode_gedcom(text.as_bytes()).expect("decode");
636
637 assert_eq!(report.declared, None, "the declaration is out of reach");
638 assert_eq!(report.used, Some(GedcomEncoding::Utf8));
639 }
640
641 #[test]
642 fn an_oversized_file_is_refused_with_the_limit_named() {
643 let limits = Limits {
646 input_bytes: 2 * 1024 * 1024,
647 ..Limits::DEFAULT
648 };
649 let bytes = vec![b'0'; limits.input_bytes + 1];
650
651 let error = decode_gedcom_with(&bytes, limits).expect_err("must refuse");
652
653 let message = error.to_string();
654 assert!(message.contains("2 MB"), "limit must be named: {message}");
655 }
656
657 #[test]
658 fn undecodable_bytes_are_counted_rather_than_failing_the_import() {
659 let bytes = b"0 HEAD\n1 CHAR ANSEL\n1 NOTE \xD0\n";
660
661 let (text, report) = decode_gedcom(bytes).expect("decode");
662
663 assert_eq!(report.undecodable_bytes, 1);
664 assert!(text.contains('\u{FFFD}'));
665 assert!(!report.warnings.is_empty());
666 }
667}