Skip to main content

gedcomkit/
decode.rs

1//! Turning the bytes of a GEDCOM file into text.
2//!
3//! This is a layer of its own because getting it wrong silently rejects most
4//! real files. A byte-order mark is the common case — every GEDCOM 7 file
5//! gedcom.io publishes carries one, as does every export from Family Tree
6//! Maker 24, Legacy 10, `RootsMagic` 8, and Ancestris 11 — and an unstripped mark
7//! makes the first line unparseable for a reason that names the wrong thing.
8//!
9//! Older files are not UTF-8 at all. GEDCOM 5.5 defaults to ANSEL, a 1980s
10//! library-cataloguing set whose diacritics *precede* the letter they modify,
11//! and files declaring `ANSI` are Windows-1252 in practice.
12//!
13//! Nothing here guesses in silence. Every decision, and every disagreement
14//! between what a file declares and what it contains, is reported so the import
15//! preview can show it before anything is written.
16
17use crate::{GedcomError, GedcomErrorKind, Limits, fault};
18use std::fmt;
19
20/// How many leading bytes are searched for the `CHAR` declaration. The header
21/// is the first record, and no real one approaches this.
22const HEADER_SCAN_BYTES: usize = 8 * 1024;
23
24/// The character encoding a GEDCOM file was read as.
25#[derive(Clone, Copy, Debug, Eq, PartialEq)]
26#[non_exhaustive]
27#[cfg_attr(feature = "serde", derive(serde::Serialize))]
28#[cfg_attr(feature = "ts", derive(ts_rs::TS))]
29pub enum GedcomEncoding {
30    /// Seven-bit ASCII, which every other encoding here agrees with.
31    Ascii,
32    /// UTF-8, with or without a byte-order mark.
33    Utf8,
34    /// UTF-16, big-endian.
35    Utf16Be,
36    /// UTF-16, little-endian.
37    Utf16Le,
38    /// ANSEL, the GEDCOM 5.5 default.
39    Ansel,
40    /// Windows-1252, which GEDCOM files call `ANSI`.
41    Ansi,
42}
43
44impl GedcomEncoding {
45    /// The name to show a user.
46    #[must_use]
47    pub const fn label(self) -> &'static str {
48        match self {
49            Self::Ascii => "ASCII",
50            Self::Utf8 => "UTF-8",
51            Self::Utf16Be => "UTF-16 (big-endian)",
52            Self::Utf16Le => "UTF-16 (little-endian)",
53            Self::Ansel => "ANSEL",
54            Self::Ansi => "ANSI (Windows-1252)",
55        }
56    }
57}
58
59impl fmt::Display for GedcomEncoding {
60    fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
61        formatter.write_str(self.label())
62    }
63}
64
65/// What decoding did, and what it had to decide.
66#[derive(Clone, Debug, Default, Eq, PartialEq)]
67#[non_exhaustive]
68#[cfg_attr(feature = "serde", derive(serde::Serialize))]
69#[cfg_attr(feature = "ts", derive(ts_rs::TS))]
70#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))]
71pub struct EncodingReport {
72    /// The encoding the bytes were read as.
73    pub used: Option<GedcomEncoding>,
74    /// The `1 CHAR` payload exactly as the file wrote it, when it had one.
75    pub declared: Option<String>,
76    /// Whether a byte-order mark was found and removed.
77    pub byte_order_mark: bool,
78    /// Bytes that no mapping covered and that became U+FFFD.
79    pub undecodable_bytes: usize,
80    /// Everything a user should read before accepting the import.
81    pub warnings: Vec<String>,
82}
83
84impl EncodingReport {
85    /// A report that says only which encoding applies — what a caller that
86    /// serialized the document itself (always UTF-8) reports, since nothing
87    /// was decoded and there is nothing to warn about.
88    #[must_use]
89    pub fn for_encoding(encoding: GedcomEncoding) -> Self {
90        Self {
91            used: Some(encoding),
92            ..Self::default()
93        }
94    }
95
96    /// A one-line summary for the import preview.
97    #[must_use]
98    pub fn summary(&self) -> String {
99        let used = self.used.map_or("unknown", GedcomEncoding::label);
100        let mut text = format!("Read as {used}");
101        if self.byte_order_mark {
102            text.push_str(" (byte-order mark)");
103        }
104        match &self.declared {
105            Some(declared) => {
106                let _ = fmt::Write::write_fmt(&mut text, format_args!(", declared {declared}"));
107            }
108            None => text.push_str(", no CHAR declared"),
109        }
110        if self.undecodable_bytes > 0 {
111            let _ = fmt::Write::write_fmt(
112                &mut text,
113                format_args!(", {} byte(s) undecodable", self.undecodable_bytes),
114            );
115        }
116        text
117    }
118}
119
120/// Reads GEDCOM bytes as text, choosing an encoding and reporting the choice.
121///
122/// The order of authority is: a byte-order mark, then UTF-16 recognized from
123/// its null bytes, then the `1 CHAR` declaration, then a fallback that prefers
124/// UTF-8 when the bytes are valid UTF-8 and ANSEL when they are not.
125///
126/// # Errors
127///
128/// Returns [`GedcomError`] when the input exceeds the limit. A file that
129/// cannot be decoded cleanly is *not* an error: undecodable bytes become
130/// U+FFFD and are counted, so the user can see the damage and decide, rather
131/// than losing the whole file to a few bytes.
132pub fn decode_gedcom(bytes: &[u8]) -> Result<(String, EncodingReport), GedcomError> {
133    decode_gedcom_with(bytes, Limits::DEFAULT)
134}
135
136/// Reads GEDCOM bytes as text with explicit limits.
137///
138/// # Errors
139///
140/// As [`decode_gedcom`].
141pub fn decode_gedcom_with(
142    bytes: &[u8],
143    limits: Limits,
144) -> Result<(String, EncodingReport), GedcomError> {
145    if bytes.len() > limits.input_bytes {
146        return Err(fault(
147            0,
148            GedcomErrorKind::Limit,
149            format!(
150                "the file is {} MB and the limit is {} MB",
151                bytes.len() / (1024 * 1024),
152                limits.input_bytes / (1024 * 1024)
153            ),
154        ));
155    }
156
157    let mut report = EncodingReport::default();
158
159    let (body, marked) = strip_byte_order_mark(bytes);
160    report.byte_order_mark = marked.is_some();
161
162    // A mark settles it: the file says what it is in its own first bytes.
163    if let Some(encoding) = marked {
164        report.declared = declared_charset(body, encoding);
165        let text = decode_as(body, encoding, &mut report);
166        finish(encoding, &mut report);
167        return Ok((text, report));
168    }
169
170    // UTF-16 without a mark still announces itself: every GEDCOM begins with
171    // the ASCII digit `0`, so one half of that code unit is a null byte.
172    if let Some(encoding) = unmarked_utf16(body) {
173        report.declared = declared_charset(body, encoding);
174        report.warnings.push(format!(
175            "File is {encoding} but carries no byte-order mark."
176        ));
177        let text = decode_as(body, encoding, &mut report);
178        finish(encoding, &mut report);
179        return Ok((text, report));
180    }
181
182    // Everything below is byte-oriented, so the header can be read as ASCII.
183    report.declared = declared_charset(body, GedcomEncoding::Utf8);
184    let declared = report
185        .declared
186        .as_ref()
187        .map(|value| value.trim().to_owned());
188    let encoding = match declared.as_deref() {
189        Some(declared) => match_declared(declared, body, &mut report),
190        None => {
191            // No declaration. UTF-8 is a superset of ASCII, so trying it first
192            // never damages a plain file, and failing it is good evidence the
193            // file is one of the older single-byte sets.
194            if std::str::from_utf8(body).is_ok() {
195                GedcomEncoding::Utf8
196            } else {
197                report.warnings.push(
198                    "File declares no CHAR and is not valid UTF-8; read as ANSEL, the GEDCOM 5.5 default."
199                        .to_owned(),
200                );
201                GedcomEncoding::Ansel
202            }
203        }
204    };
205
206    let text = decode_as(body, encoding, &mut report);
207    finish(encoding, &mut report);
208    Ok((text, report))
209}
210
211fn finish(encoding: GedcomEncoding, report: &mut EncodingReport) {
212    report.used = Some(encoding);
213    if report.undecodable_bytes > 0 {
214        report.warnings.push(format!(
215            "{} byte(s) had no mapping in {encoding} and were replaced with U+FFFD.",
216            report.undecodable_bytes
217        ));
218    }
219}
220
221/// Resolves a `CHAR` payload to an encoding, warning when the file's contents
222/// contradict it.
223fn match_declared(declared: &str, body: &[u8], report: &mut EncodingReport) -> GedcomEncoding {
224    let normalized = declared
225        .chars()
226        .filter(char::is_ascii_alphanumeric)
227        .collect::<String>()
228        .to_ascii_uppercase();
229
230    match normalized.as_str() {
231        "UTF8" => {
232            if std::str::from_utf8(body).is_ok() {
233                GedcomEncoding::Utf8
234            } else {
235                // Windows-1252 maps every byte, so it recovers the text rather
236                // than losing the file. The mismatch is the story to tell.
237                report.warnings.push(
238                    "File declares UTF-8 but contains invalid UTF-8; read as Windows-1252 instead."
239                        .to_owned(),
240                );
241                GedcomEncoding::Ansi
242            }
243        }
244        "ANSEL" => GedcomEncoding::Ansel,
245        "ANSI" | "WINDOWS1252" | "CP1252" | "IBMWINDOWS" => GedcomEncoding::Ansi,
246        "ASCII" | "USASCII" | "ANSIZ3947" => GedcomEncoding::Ascii,
247        "UNICODE" | "UTF16" => {
248            // 5.5.1 requires a byte-order mark for UNICODE and this file has
249            // none, or an earlier branch would have taken it.
250            report.warnings.push(
251                "File declares UNICODE but has no byte-order mark and no UTF-16 structure; read as UTF-8."
252                    .to_owned(),
253            );
254            GedcomEncoding::Utf8
255        }
256        _ => {
257            report.warnings.push(format!(
258                "Unrecognized CHAR value {declared:?}; read as UTF-8."
259            ));
260            GedcomEncoding::Utf8
261        }
262    }
263}
264
265fn strip_byte_order_mark(bytes: &[u8]) -> (&[u8], Option<GedcomEncoding>) {
266    if let Some(rest) = bytes.strip_prefix(&[0xEF, 0xBB, 0xBF]) {
267        return (rest, Some(GedcomEncoding::Utf8));
268    }
269    if let Some(rest) = bytes.strip_prefix(&[0xFE, 0xFF]) {
270        return (rest, Some(GedcomEncoding::Utf16Be));
271    }
272    if let Some(rest) = bytes.strip_prefix(&[0xFF, 0xFE]) {
273        return (rest, Some(GedcomEncoding::Utf16Le));
274    }
275    (bytes, None)
276}
277
278const fn unmarked_utf16(bytes: &[u8]) -> Option<GedcomEncoding> {
279    match bytes {
280        [0x00, second, ..] if *second != 0x00 => Some(GedcomEncoding::Utf16Be),
281        [first, 0x00, ..] if *first != 0x00 => Some(GedcomEncoding::Utf16Le),
282        _ => None,
283    }
284}
285
286/// Finds the `1 CHAR` payload by reading the header as ASCII.
287///
288/// Every encoding here agrees with ASCII on the bytes a header is written in,
289/// so this works before the encoding is known — which is the point.
290fn declared_charset(bytes: &[u8], encoding: GedcomEncoding) -> Option<String> {
291    let head = &bytes[..bytes.len().min(HEADER_SCAN_BYTES)];
292    let text = match encoding {
293        GedcomEncoding::Utf16Be | GedcomEncoding::Utf16Le => {
294            let mut scratch = EncodingReport::default();
295            decode_utf16(head, encoding == GedcomEncoding::Utf16Be, &mut scratch)
296        }
297        _ => head.iter().map(|byte| char::from(*byte)).collect(),
298    };
299
300    text.lines()
301        .map(str::trim_end)
302        .find_map(|line| line.strip_prefix("1 CHAR "))
303        .map(str::trim)
304        .filter(|value| !value.is_empty())
305        .map(str::to_owned)
306}
307
308fn decode_as(bytes: &[u8], encoding: GedcomEncoding, report: &mut EncodingReport) -> String {
309    match encoding {
310        GedcomEncoding::Utf16Be => decode_utf16(bytes, true, report),
311        GedcomEncoding::Utf16Le => decode_utf16(bytes, false, report),
312        GedcomEncoding::Ansel => decode_ansel(bytes, report),
313        GedcomEncoding::Ansi => decode_windows_1252(bytes),
314        GedcomEncoding::Ascii => decode_ascii(bytes, report),
315        GedcomEncoding::Utf8 => std::str::from_utf8(bytes).map_or_else(
316            |_| {
317                let text = String::from_utf8_lossy(bytes).into_owned();
318                report.undecodable_bytes += text.matches('\u{FFFD}').count();
319                text
320            },
321            str::to_owned,
322        ),
323    }
324}
325
326fn decode_ascii(bytes: &[u8], report: &mut EncodingReport) -> String {
327    bytes
328        .iter()
329        .map(|byte| {
330            if byte.is_ascii() {
331                char::from(*byte)
332            } else {
333                report.undecodable_bytes += 1;
334                '\u{FFFD}'
335            }
336        })
337        .collect()
338}
339
340fn decode_utf16(bytes: &[u8], big_endian: bool, report: &mut EncodingReport) -> String {
341    if !bytes.len().is_multiple_of(2) {
342        report
343            .warnings
344            .push("UTF-16 file has an odd number of bytes; the final byte was ignored.".to_owned());
345    }
346    let units = bytes
347        .as_chunks::<2>()
348        .0
349        .iter()
350        .map(|pair| {
351            if big_endian {
352                u16::from_be_bytes([pair[0], pair[1]])
353            } else {
354                u16::from_le_bytes([pair[0], pair[1]])
355            }
356        })
357        .collect::<Vec<_>>();
358
359    let mut text = String::with_capacity(units.len());
360    for unit in char::decode_utf16(units) {
361        text.push(unit.unwrap_or_else(|_| {
362            report.undecodable_bytes += 1;
363            '\u{FFFD}'
364        }));
365    }
366    text
367}
368
369/// Windows-1252, which differs from Latin-1 only in `0x80`–`0x9F`.
370fn decode_windows_1252(bytes: &[u8]) -> String {
371    bytes
372        .iter()
373        .map(|byte| match byte {
374            0x80..=0x9F => WINDOWS_1252_HIGH[usize::from(byte - 0x80)],
375            other => char::from(*other),
376        })
377        .collect()
378}
379
380/// The five positions Windows-1252 leaves undefined map to the C1 control they
381/// sit on, which keeps the text the same length and loses nothing meaningful.
382pub const WINDOWS_1252_HIGH: [char; 32] = [
383    '\u{20AC}', '\u{0081}', '\u{201A}', '\u{0192}', '\u{201E}', '\u{2026}', '\u{2020}', '\u{2021}',
384    '\u{02C6}', '\u{2030}', '\u{0160}', '\u{2039}', '\u{0152}', '\u{008D}', '\u{017D}', '\u{008F}',
385    '\u{0090}', '\u{2018}', '\u{2019}', '\u{201C}', '\u{201D}', '\u{2022}', '\u{2013}', '\u{2014}',
386    '\u{02DC}', '\u{2122}', '\u{0161}', '\u{203A}', '\u{0153}', '\u{009D}', '\u{017E}', '\u{0178}',
387];
388
389/// Decodes ANSEL as GEDCOM 5.5 Appendix D defines it.
390///
391/// The one structural difference from every other encoding here: a diacritic
392/// comes *before* the letter it modifies, where Unicode puts it after. So a
393/// run of marks is buffered, the base character is emitted, and the marks
394/// follow it in the order they were written.
395fn decode_ansel(bytes: &[u8], report: &mut EncodingReport) -> String {
396    let mut text = String::with_capacity(bytes.len());
397    let mut marks = Vec::new();
398    let mut index = 0;
399
400    while index < bytes.len() {
401        let byte = bytes[index];
402        index += 1;
403
404        if let Some(mark) = ansel_combining(byte) {
405            marks.push(mark);
406            continue;
407        }
408
409        let base = if byte.is_ascii() {
410            char::from(byte)
411        } else if let Some(character) = ansel_graphic(byte) {
412            character
413        } else {
414            report.undecodable_bytes += 1;
415            '\u{FFFD}'
416        };
417
418        text.push(base);
419        // A mark with nothing after it is written anyway rather than dropped;
420        // losing evidence silently is worse than an odd-looking character.
421        text.extend(marks.iter().copied());
422        marks.clear();
423    }
424
425    text.extend(marks);
426    text
427}
428
429/// The combining diacritics, `0xE0`–`0xFB` and `0xFE`.
430const fn ansel_combining(byte: u8) -> Option<char> {
431    Some(match byte {
432        0xE0 => '\u{0309}', // hook above
433        0xE1 => '\u{0300}', // grave
434        0xE2 => '\u{0301}', // acute
435        0xE3 => '\u{0302}', // circumflex
436        0xE4 => '\u{0303}', // tilde
437        0xE5 => '\u{0304}', // macron
438        0xE6 => '\u{0306}', // breve
439        0xE7 => '\u{0307}', // dot above
440        0xE8 => '\u{0308}', // diaeresis
441        0xE9 => '\u{030C}', // caron
442        0xEA => '\u{030A}', // ring above
443        0xEB => '\u{FE20}', // ligature, left half
444        0xEC => '\u{FE21}', // ligature, right half
445        0xED => '\u{0315}', // comma above right
446        0xEE => '\u{030B}', // double acute
447        0xEF => '\u{0310}', // candrabindu
448        0xF0 => '\u{0327}', // cedilla
449        0xF1 => '\u{0328}', // ogonek
450        0xF2 => '\u{0323}', // dot below
451        0xF3 => '\u{0324}', // diaeresis below
452        0xF4 => '\u{0325}', // ring below
453        0xF5 => '\u{0333}', // double low line
454        0xF6 => '\u{0332}', // line below
455        0xF7 => '\u{0326}', // comma below
456        0xF8 => '\u{031C}', // left half ring below
457        0xF9 => '\u{032E}', // breve below
458        0xFA => '\u{FE22}', // double tilde, left half
459        0xFB => '\u{FE23}', // double tilde, right half
460        0xFE => '\u{0313}', // comma above
461        _ => return None,
462    })
463}
464
465/// The non-combining graphic characters, `0xA1`–`0xCF`, including the two LDS
466/// box extensions and the two midline letters GEDCOM 5.5 adds.
467const fn ansel_graphic(byte: u8) -> Option<char> {
468    Some(match byte {
469        0xA1 => 'Ł',
470        0xA2 => 'Ø',
471        0xA3 => 'Đ',
472        0xA4 => 'Þ',
473        0xA5 => 'Æ',
474        0xA6 => 'Œ',
475        0xA7 => '\u{02B9}', // single prime
476        0xA8 => '·',
477        0xA9 => '\u{266D}', // musical flat
478        0xAA => '®',
479        0xAB => '±',
480        0xAC => 'Ơ',
481        0xAD => 'Ư',
482        0xAE => '\u{02BB}', // left half ring
483        0xB0 => '\u{02BC}', // right half ring
484        0xB1 => 'ł',
485        0xB2 => 'ø',
486        0xB3 => 'đ',
487        0xB4 => 'þ',
488        0xB5 => 'æ',
489        0xB6 => 'œ',
490        0xB7 => '\u{02BA}', // double prime
491        0xB8 => 'ı',
492        0xB9 => '£',
493        0xBA => 'ð',
494        0xBC => 'ơ',
495        0xBD => 'ư',
496        0xBE => '\u{25A1}', // empty box, an LDS extension
497        0xBF => '\u{25A0}', // black box, an LDS extension
498        0xC0 => '°',
499        0xC1 => '\u{2113}', // script l
500        0xC2 => '\u{2117}', // phonograph copyright
501        0xC3 => '©',
502        0xC4 => '\u{266F}', // musical sharp
503        0xC5 => '¿',
504        0xC6 => '¡',
505        0xCD => 'e', // midline e, an LDS extension with no Unicode of its own
506        0xCE => 'o', // midline o, likewise
507        0xCF => 'ß',
508        _ => return None,
509    })
510}
511
512#[cfg(test)]
513mod tests {
514    use super::*;
515
516    #[test]
517    fn a_utf8_byte_order_mark_is_stripped_and_reported() {
518        let bytes = b"\xEF\xBB\xBF0 HEAD\n0 TRLR\n";
519
520        let (text, report) = decode_gedcom(bytes).expect("decode");
521
522        assert!(
523            text.starts_with("0 HEAD"),
524            "mark must not survive: {text:?}"
525        );
526        assert!(report.byte_order_mark);
527        assert_eq!(report.used, Some(GedcomEncoding::Utf8));
528    }
529
530    #[test]
531    fn utf16_is_read_with_or_without_a_byte_order_mark() {
532        let marked = b"\xFF\xFE0\x00 \x00H\x00E\x00A\x00D\x00";
533        let (text, report) = decode_gedcom(marked).expect("decode marked");
534        assert_eq!(text, "0 HEAD");
535        assert_eq!(report.used, Some(GedcomEncoding::Utf16Le));
536        assert!(report.byte_order_mark);
537
538        let bare = b"\x000\x00 \x00H\x00E\x00A\x00D";
539        let (text, report) = decode_gedcom(bare).expect("decode bare");
540        assert_eq!(text, "0 HEAD");
541        assert_eq!(report.used, Some(GedcomEncoding::Utf16Be));
542        assert!(!report.byte_order_mark);
543        assert!(
544            !report.warnings.is_empty(),
545            "a missing mark is worth saying"
546        );
547    }
548
549    #[test]
550    fn ansel_puts_a_diacritic_after_the_letter_it_modifies() {
551        // "1 NAME Jos<acute>e" — ANSEL writes the accent first.
552        let bytes = b"0 HEAD\n1 CHAR ANSEL\n0 @I1@ INDI\n1 NAME Jos\xE2e\n";
553
554        let (text, report) = decode_gedcom(bytes).expect("decode");
555
556        assert_eq!(report.used, Some(GedcomEncoding::Ansel));
557        assert_eq!(report.declared.as_deref(), Some("ANSEL"));
558        assert!(text.contains("Jose\u{0301}"), "got {text:?}");
559        assert_eq!(report.undecodable_bytes, 0);
560    }
561
562    #[test]
563    fn ansel_graphic_characters_decode() {
564        let bytes = b"0 HEAD\n1 CHAR ANSEL\n1 NOTE \xA2 \xB2 \xCF\n";
565
566        let (text, _) = decode_gedcom(bytes).expect("decode");
567
568        assert!(text.contains("Ø ø ß"), "got {text:?}");
569    }
570
571    #[test]
572    fn a_declared_charset_that_the_bytes_contradict_is_reported_not_obeyed() {
573        // Declares UTF-8, contains a lone 0xE9 — Windows-1252 recovers it.
574        let bytes = b"0 HEAD\n1 CHAR UTF-8\n1 NOTE caf\xE9\n";
575
576        let (text, report) = decode_gedcom(bytes).expect("decode");
577
578        assert_eq!(report.used, Some(GedcomEncoding::Ansi));
579        assert_eq!(report.declared.as_deref(), Some("UTF-8"));
580        assert!(text.contains("café"), "got {text:?}");
581        assert!(
582            report
583                .warnings
584                .iter()
585                .any(|warning| warning.contains("declares UTF-8")),
586            "{:?}",
587            report.warnings
588        );
589    }
590
591    #[test]
592    fn an_undeclared_file_prefers_utf8_and_falls_back_to_ansel() {
593        let utf8 = "0 HEAD\n1 NOTE café\n".as_bytes();
594        let (text, report) = decode_gedcom(utf8).expect("decode utf8");
595        assert_eq!(report.used, Some(GedcomEncoding::Utf8));
596        assert!(text.contains("café"));
597        assert!(report.declared.is_none());
598
599        let ansel = b"0 HEAD\n1 NOTE caf\xE2e\n";
600        let (text, report) = decode_gedcom(ansel).expect("decode ansel");
601        assert_eq!(report.used, Some(GedcomEncoding::Ansel));
602        assert!(text.contains("cafe\u{0301}"), "got {text:?}");
603    }
604
605    #[test]
606    fn utf16_surrogate_pairs_decode_to_the_characters_they_spell() {
607        // "0 HEAD\n1 NOTE 𝄞" — the clef is U+1D11E, a surrogate pair in
608        // UTF-16, which the per-unit reader must join rather than replace.
609        let mut bytes = vec![0xFFu8, 0xFE];
610        for unit in "0 HEAD\n1 NOTE \u{1D11E}\n".encode_utf16() {
611            bytes.extend_from_slice(&unit.to_le_bytes());
612        }
613
614        let (text, report) = decode_gedcom(&bytes).expect("decode");
615
616        assert_eq!(report.undecodable_bytes, 0);
617        assert!(text.contains('\u{1D11E}'), "got {text:?}");
618    }
619
620    #[test]
621    fn a_char_line_beyond_the_header_scan_window_falls_back_honestly() {
622        // The declaration hunt reads a bounded prefix; a CHAR pushed past it
623        // by an absurd header is treated as undeclared — valid UTF-8 reads as
624        // UTF-8 — rather than making the scan unbounded.
625        use std::fmt::Write as _;
626        let mut text = String::from("0 HEAD\n");
627        for index in 0..600 {
628            let _ = writeln!(
629                text,
630                "1 NOTE padding line number {index} to push the declaration far down"
631            );
632        }
633        text.push_str("1 CHAR ANSEL\n0 TRLR\n");
634
635        let (_, report) = decode_gedcom(text.as_bytes()).expect("decode");
636
637        assert_eq!(report.declared, None, "the declaration is out of reach");
638        assert_eq!(report.used, Some(GedcomEncoding::Utf8));
639    }
640
641    #[test]
642    fn an_oversized_file_is_refused_with_the_limit_named() {
643        // The limit is passed in rather than allocating the default one, which
644        // is 128 MB and would make this test cost more than it proves.
645        let limits = Limits {
646            input_bytes: 2 * 1024 * 1024,
647            ..Limits::DEFAULT
648        };
649        let bytes = vec![b'0'; limits.input_bytes + 1];
650
651        let error = decode_gedcom_with(&bytes, limits).expect_err("must refuse");
652
653        let message = error.to_string();
654        assert!(message.contains("2 MB"), "limit must be named: {message}");
655    }
656
657    #[test]
658    fn undecodable_bytes_are_counted_rather_than_failing_the_import() {
659        let bytes = b"0 HEAD\n1 CHAR ANSEL\n1 NOTE \xD0\n";
660
661        let (text, report) = decode_gedcom(bytes).expect("decode");
662
663        assert_eq!(report.undecodable_bytes, 1);
664        assert!(text.contains('\u{FFFD}'));
665        assert!(!report.warnings.is_empty());
666    }
667}