use crate::{GedcomError, GedcomErrorKind, Limits, fault};
use std::fmt;
const HEADER_SCAN_BYTES: usize = 8 * 1024;
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
#[non_exhaustive]
#[cfg_attr(feature = "serde", derive(serde::Serialize))]
pub enum GedcomEncoding {
Ascii,
Utf8,
Utf16Be,
Utf16Le,
Ansel,
Ansi,
}
impl GedcomEncoding {
#[must_use]
pub const fn label(self) -> &'static str {
match self {
Self::Ascii => "ASCII",
Self::Utf8 => "UTF-8",
Self::Utf16Be => "UTF-16 (big-endian)",
Self::Utf16Le => "UTF-16 (little-endian)",
Self::Ansel => "ANSEL",
Self::Ansi => "ANSI (Windows-1252)",
}
}
}
impl fmt::Display for GedcomEncoding {
fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
formatter.write_str(self.label())
}
}
#[derive(Clone, Debug, Default, Eq, PartialEq)]
#[non_exhaustive]
#[cfg_attr(feature = "serde", derive(serde::Serialize))]
#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))]
pub struct EncodingReport {
pub used: Option<GedcomEncoding>,
pub declared: Option<String>,
pub byte_order_mark: bool,
pub undecodable_bytes: usize,
pub warnings: Vec<String>,
}
impl EncodingReport {
#[must_use]
pub fn for_encoding(encoding: GedcomEncoding) -> Self {
Self {
used: Some(encoding),
..Self::default()
}
}
#[must_use]
pub fn summary(&self) -> String {
let used = self.used.map_or("unknown", GedcomEncoding::label);
let mut text = format!("Read as {used}");
if self.byte_order_mark {
text.push_str(" (byte-order mark)");
}
match &self.declared {
Some(declared) => {
let _ = fmt::Write::write_fmt(&mut text, format_args!(", declared {declared}"));
}
None => text.push_str(", no CHAR declared"),
}
if self.undecodable_bytes > 0 {
let _ = fmt::Write::write_fmt(
&mut text,
format_args!(", {} byte(s) undecodable", self.undecodable_bytes),
);
}
text
}
}
pub fn decode_gedcom(bytes: &[u8]) -> Result<(String, EncodingReport), GedcomError> {
decode_gedcom_with(bytes, Limits::DEFAULT)
}
pub fn decode_gedcom_with(
bytes: &[u8],
limits: Limits,
) -> Result<(String, EncodingReport), GedcomError> {
if bytes.len() > limits.input_bytes {
return Err(fault(
0,
GedcomErrorKind::Limit,
format!(
"the file is {} MB and the limit is {} MB",
bytes.len() / (1024 * 1024),
limits.input_bytes / (1024 * 1024)
),
));
}
let mut report = EncodingReport::default();
let (body, marked) = strip_byte_order_mark(bytes);
report.byte_order_mark = marked.is_some();
if let Some(encoding) = marked {
report.declared = declared_charset(body, encoding);
let text = decode_as(body, encoding, &mut report);
finish(encoding, &mut report);
return Ok((text, report));
}
if let Some(encoding) = unmarked_utf16(body) {
report.declared = declared_charset(body, encoding);
report.warnings.push(format!(
"File is {encoding} but carries no byte-order mark."
));
let text = decode_as(body, encoding, &mut report);
finish(encoding, &mut report);
return Ok((text, report));
}
report.declared = declared_charset(body, GedcomEncoding::Utf8);
let declared = report
.declared
.as_ref()
.map(|value| value.trim().to_owned());
let encoding = match declared.as_deref() {
Some(declared) => match_declared(declared, body, &mut report),
None => {
if std::str::from_utf8(body).is_ok() {
GedcomEncoding::Utf8
} else {
report.warnings.push(
"File declares no CHAR and is not valid UTF-8; read as ANSEL, the GEDCOM 5.5 default."
.to_owned(),
);
GedcomEncoding::Ansel
}
}
};
let text = decode_as(body, encoding, &mut report);
finish(encoding, &mut report);
Ok((text, report))
}
fn finish(encoding: GedcomEncoding, report: &mut EncodingReport) {
report.used = Some(encoding);
if report.undecodable_bytes > 0 {
report.warnings.push(format!(
"{} byte(s) had no mapping in {encoding} and were replaced with U+FFFD.",
report.undecodable_bytes
));
}
}
fn match_declared(declared: &str, body: &[u8], report: &mut EncodingReport) -> GedcomEncoding {
let normalized = declared
.chars()
.filter(char::is_ascii_alphanumeric)
.collect::<String>()
.to_ascii_uppercase();
match normalized.as_str() {
"UTF8" => {
if std::str::from_utf8(body).is_ok() {
GedcomEncoding::Utf8
} else {
report.warnings.push(
"File declares UTF-8 but contains invalid UTF-8; read as Windows-1252 instead."
.to_owned(),
);
GedcomEncoding::Ansi
}
}
"ANSEL" => GedcomEncoding::Ansel,
"ANSI" | "WINDOWS1252" | "CP1252" | "IBMWINDOWS" => GedcomEncoding::Ansi,
"ASCII" | "USASCII" | "ANSIZ3947" => GedcomEncoding::Ascii,
"UNICODE" | "UTF16" => {
report.warnings.push(
"File declares UNICODE but has no byte-order mark and no UTF-16 structure; read as UTF-8."
.to_owned(),
);
GedcomEncoding::Utf8
}
_ => {
report.warnings.push(format!(
"Unrecognized CHAR value {declared:?}; read as UTF-8."
));
GedcomEncoding::Utf8
}
}
}
fn strip_byte_order_mark(bytes: &[u8]) -> (&[u8], Option<GedcomEncoding>) {
if let Some(rest) = bytes.strip_prefix(&[0xEF, 0xBB, 0xBF]) {
return (rest, Some(GedcomEncoding::Utf8));
}
if let Some(rest) = bytes.strip_prefix(&[0xFE, 0xFF]) {
return (rest, Some(GedcomEncoding::Utf16Be));
}
if let Some(rest) = bytes.strip_prefix(&[0xFF, 0xFE]) {
return (rest, Some(GedcomEncoding::Utf16Le));
}
(bytes, None)
}
const fn unmarked_utf16(bytes: &[u8]) -> Option<GedcomEncoding> {
match bytes {
[0x00, second, ..] if *second != 0x00 => Some(GedcomEncoding::Utf16Be),
[first, 0x00, ..] if *first != 0x00 => Some(GedcomEncoding::Utf16Le),
_ => None,
}
}
fn declared_charset(bytes: &[u8], encoding: GedcomEncoding) -> Option<String> {
let head = &bytes[..bytes.len().min(HEADER_SCAN_BYTES)];
let text = match encoding {
GedcomEncoding::Utf16Be | GedcomEncoding::Utf16Le => {
let mut scratch = EncodingReport::default();
decode_utf16(head, encoding == GedcomEncoding::Utf16Be, &mut scratch)
}
_ => head.iter().map(|byte| char::from(*byte)).collect(),
};
text.lines()
.map(str::trim_end)
.find_map(|line| line.strip_prefix("1 CHAR "))
.map(str::trim)
.filter(|value| !value.is_empty())
.map(str::to_owned)
}
fn decode_as(bytes: &[u8], encoding: GedcomEncoding, report: &mut EncodingReport) -> String {
match encoding {
GedcomEncoding::Utf16Be => decode_utf16(bytes, true, report),
GedcomEncoding::Utf16Le => decode_utf16(bytes, false, report),
GedcomEncoding::Ansel => decode_ansel(bytes, report),
GedcomEncoding::Ansi => decode_windows_1252(bytes),
GedcomEncoding::Ascii => decode_ascii(bytes, report),
GedcomEncoding::Utf8 => std::str::from_utf8(bytes).map_or_else(
|_| {
let text = String::from_utf8_lossy(bytes).into_owned();
report.undecodable_bytes += text.matches('\u{FFFD}').count();
text
},
str::to_owned,
),
}
}
fn decode_ascii(bytes: &[u8], report: &mut EncodingReport) -> String {
bytes
.iter()
.map(|byte| {
if byte.is_ascii() {
char::from(*byte)
} else {
report.undecodable_bytes += 1;
'\u{FFFD}'
}
})
.collect()
}
fn decode_utf16(bytes: &[u8], big_endian: bool, report: &mut EncodingReport) -> String {
if !bytes.len().is_multiple_of(2) {
report
.warnings
.push("UTF-16 file has an odd number of bytes; the final byte was ignored.".to_owned());
}
let units = bytes
.as_chunks::<2>()
.0
.iter()
.map(|pair| {
if big_endian {
u16::from_be_bytes([pair[0], pair[1]])
} else {
u16::from_le_bytes([pair[0], pair[1]])
}
})
.collect::<Vec<_>>();
let mut text = String::with_capacity(units.len());
for unit in char::decode_utf16(units) {
text.push(unit.unwrap_or_else(|_| {
report.undecodable_bytes += 1;
'\u{FFFD}'
}));
}
text
}
fn decode_windows_1252(bytes: &[u8]) -> String {
bytes
.iter()
.map(|byte| match byte {
0x80..=0x9F => WINDOWS_1252_HIGH[usize::from(byte - 0x80)],
other => char::from(*other),
})
.collect()
}
pub const WINDOWS_1252_HIGH: [char; 32] = [
'\u{20AC}', '\u{0081}', '\u{201A}', '\u{0192}', '\u{201E}', '\u{2026}', '\u{2020}', '\u{2021}',
'\u{02C6}', '\u{2030}', '\u{0160}', '\u{2039}', '\u{0152}', '\u{008D}', '\u{017D}', '\u{008F}',
'\u{0090}', '\u{2018}', '\u{2019}', '\u{201C}', '\u{201D}', '\u{2022}', '\u{2013}', '\u{2014}',
'\u{02DC}', '\u{2122}', '\u{0161}', '\u{203A}', '\u{0153}', '\u{009D}', '\u{017E}', '\u{0178}',
];
fn decode_ansel(bytes: &[u8], report: &mut EncodingReport) -> String {
let mut text = String::with_capacity(bytes.len());
let mut marks = Vec::new();
let mut index = 0;
while index < bytes.len() {
let byte = bytes[index];
index += 1;
if let Some(mark) = ansel_combining(byte) {
marks.push(mark);
continue;
}
let base = if byte.is_ascii() {
char::from(byte)
} else if let Some(character) = ansel_graphic(byte) {
character
} else {
report.undecodable_bytes += 1;
'\u{FFFD}'
};
text.push(base);
text.extend(marks.iter().copied());
marks.clear();
}
text.extend(marks);
text
}
const fn ansel_combining(byte: u8) -> Option<char> {
Some(match byte {
0xE0 => '\u{0309}', 0xE1 => '\u{0300}', 0xE2 => '\u{0301}', 0xE3 => '\u{0302}', 0xE4 => '\u{0303}', 0xE5 => '\u{0304}', 0xE6 => '\u{0306}', 0xE7 => '\u{0307}', 0xE8 => '\u{0308}', 0xE9 => '\u{030C}', 0xEA => '\u{030A}', 0xEB => '\u{FE20}', 0xEC => '\u{FE21}', 0xED => '\u{0315}', 0xEE => '\u{030B}', 0xEF => '\u{0310}', 0xF0 => '\u{0327}', 0xF1 => '\u{0328}', 0xF2 => '\u{0323}', 0xF3 => '\u{0324}', 0xF4 => '\u{0325}', 0xF5 => '\u{0333}', 0xF6 => '\u{0332}', 0xF7 => '\u{0326}', 0xF8 => '\u{031C}', 0xF9 => '\u{032E}', 0xFA => '\u{FE22}', 0xFB => '\u{FE23}', 0xFE => '\u{0313}', _ => return None,
})
}
const fn ansel_graphic(byte: u8) -> Option<char> {
Some(match byte {
0xA1 => 'Ł',
0xA2 => 'Ø',
0xA3 => 'Đ',
0xA4 => 'Þ',
0xA5 => 'Æ',
0xA6 => 'Œ',
0xA7 => '\u{02B9}', 0xA8 => '·',
0xA9 => '\u{266D}', 0xAA => '®',
0xAB => '±',
0xAC => 'Ơ',
0xAD => 'Ư',
0xAE => '\u{02BB}', 0xB0 => '\u{02BC}', 0xB1 => 'ł',
0xB2 => 'ø',
0xB3 => 'đ',
0xB4 => 'þ',
0xB5 => 'æ',
0xB6 => 'œ',
0xB7 => '\u{02BA}', 0xB8 => 'ı',
0xB9 => '£',
0xBA => 'ð',
0xBC => 'ơ',
0xBD => 'ư',
0xBE => '\u{25A1}', 0xBF => '\u{25A0}', 0xC0 => '°',
0xC1 => '\u{2113}', 0xC2 => '\u{2117}', 0xC3 => '©',
0xC4 => '\u{266F}', 0xC5 => '¿',
0xC6 => '¡',
0xCD => 'e', 0xCE => 'o', 0xCF => 'ß',
_ => return None,
})
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn a_utf8_byte_order_mark_is_stripped_and_reported() {
let bytes = b"\xEF\xBB\xBF0 HEAD\n0 TRLR\n";
let (text, report) = decode_gedcom(bytes).expect("decode");
assert!(
text.starts_with("0 HEAD"),
"mark must not survive: {text:?}"
);
assert!(report.byte_order_mark);
assert_eq!(report.used, Some(GedcomEncoding::Utf8));
}
#[test]
fn utf16_is_read_with_or_without_a_byte_order_mark() {
let marked = b"\xFF\xFE0\x00 \x00H\x00E\x00A\x00D\x00";
let (text, report) = decode_gedcom(marked).expect("decode marked");
assert_eq!(text, "0 HEAD");
assert_eq!(report.used, Some(GedcomEncoding::Utf16Le));
assert!(report.byte_order_mark);
let bare = b"\x000\x00 \x00H\x00E\x00A\x00D";
let (text, report) = decode_gedcom(bare).expect("decode bare");
assert_eq!(text, "0 HEAD");
assert_eq!(report.used, Some(GedcomEncoding::Utf16Be));
assert!(!report.byte_order_mark);
assert!(
!report.warnings.is_empty(),
"a missing mark is worth saying"
);
}
#[test]
fn ansel_puts_a_diacritic_after_the_letter_it_modifies() {
let bytes = b"0 HEAD\n1 CHAR ANSEL\n0 @I1@ INDI\n1 NAME Jos\xE2e\n";
let (text, report) = decode_gedcom(bytes).expect("decode");
assert_eq!(report.used, Some(GedcomEncoding::Ansel));
assert_eq!(report.declared.as_deref(), Some("ANSEL"));
assert!(text.contains("Jose\u{0301}"), "got {text:?}");
assert_eq!(report.undecodable_bytes, 0);
}
#[test]
fn ansel_graphic_characters_decode() {
let bytes = b"0 HEAD\n1 CHAR ANSEL\n1 NOTE \xA2 \xB2 \xCF\n";
let (text, _) = decode_gedcom(bytes).expect("decode");
assert!(text.contains("Ø ø ß"), "got {text:?}");
}
#[test]
fn a_declared_charset_that_the_bytes_contradict_is_reported_not_obeyed() {
let bytes = b"0 HEAD\n1 CHAR UTF-8\n1 NOTE caf\xE9\n";
let (text, report) = decode_gedcom(bytes).expect("decode");
assert_eq!(report.used, Some(GedcomEncoding::Ansi));
assert_eq!(report.declared.as_deref(), Some("UTF-8"));
assert!(text.contains("café"), "got {text:?}");
assert!(
report
.warnings
.iter()
.any(|warning| warning.contains("declares UTF-8")),
"{:?}",
report.warnings
);
}
#[test]
fn an_undeclared_file_prefers_utf8_and_falls_back_to_ansel() {
let utf8 = "0 HEAD\n1 NOTE café\n".as_bytes();
let (text, report) = decode_gedcom(utf8).expect("decode utf8");
assert_eq!(report.used, Some(GedcomEncoding::Utf8));
assert!(text.contains("café"));
assert!(report.declared.is_none());
let ansel = b"0 HEAD\n1 NOTE caf\xE2e\n";
let (text, report) = decode_gedcom(ansel).expect("decode ansel");
assert_eq!(report.used, Some(GedcomEncoding::Ansel));
assert!(text.contains("cafe\u{0301}"), "got {text:?}");
}
#[test]
fn utf16_surrogate_pairs_decode_to_the_characters_they_spell() {
let mut bytes = vec![0xFFu8, 0xFE];
for unit in "0 HEAD\n1 NOTE \u{1D11E}\n".encode_utf16() {
bytes.extend_from_slice(&unit.to_le_bytes());
}
let (text, report) = decode_gedcom(&bytes).expect("decode");
assert_eq!(report.undecodable_bytes, 0);
assert!(text.contains('\u{1D11E}'), "got {text:?}");
}
#[test]
fn a_char_line_beyond_the_header_scan_window_falls_back_honestly() {
use std::fmt::Write as _;
let mut text = String::from("0 HEAD\n");
for index in 0..600 {
let _ = writeln!(
text,
"1 NOTE padding line number {index} to push the declaration far down"
);
}
text.push_str("1 CHAR ANSEL\n0 TRLR\n");
let (_, report) = decode_gedcom(text.as_bytes()).expect("decode");
assert_eq!(report.declared, None, "the declaration is out of reach");
assert_eq!(report.used, Some(GedcomEncoding::Utf8));
}
#[test]
fn an_oversized_file_is_refused_with_the_limit_named() {
let limits = Limits {
input_bytes: 2 * 1024 * 1024,
..Limits::DEFAULT
};
let bytes = vec![b'0'; limits.input_bytes + 1];
let error = decode_gedcom_with(&bytes, limits).expect_err("must refuse");
let message = error.to_string();
assert!(message.contains("2 MB"), "limit must be named: {message}");
}
#[test]
fn undecodable_bytes_are_counted_rather_than_failing_the_import() {
let bytes = b"0 HEAD\n1 CHAR ANSEL\n1 NOTE \xD0\n";
let (text, report) = decode_gedcom(bytes).expect("decode");
assert_eq!(report.undecodable_bytes, 1);
assert!(text.contains('\u{FFFD}'));
assert!(!report.warnings.is_empty());
}
}