Skip to main content

gedcomkit/
decode.rs

1//! Turning the bytes of a GEDCOM file into text.
2//!
3//! This is a layer of its own because getting it wrong silently rejects most
4//! real files. A byte-order mark is the common case — every GEDCOM 7 file
5//! gedcom.io publishes carries one, as does every export from Family Tree
6//! Maker 24, Legacy 10, `RootsMagic` 8, and Ancestris 11 — and an unstripped mark
7//! makes the first line unparseable for a reason that names the wrong thing.
8//!
9//! Older files are not UTF-8 at all. GEDCOM 5.5 defaults to ANSEL, a 1980s
10//! library-cataloguing set whose diacritics *precede* the letter they modify,
11//! and files declaring `ANSI` are Windows-1252 in practice.
12//!
13//! Nothing here guesses in silence. Every decision, and every disagreement
14//! between what a file declares and what it contains, is reported so the import
15//! preview can show it before anything is written.
16
17use crate::{GedcomError, GedcomErrorKind, Limits, fault};
18use std::fmt;
19
20/// How many leading bytes are searched for the `CHAR` declaration. The header
21/// is the first record, and no real one approaches this.
22const HEADER_SCAN_BYTES: usize = 8 * 1024;
23
24/// The character encoding a GEDCOM file was read as.
25#[derive(Clone, Copy, Debug, Eq, PartialEq)]
26#[non_exhaustive]
27#[cfg_attr(feature = "serde", derive(serde::Serialize))]
28pub enum GedcomEncoding {
29    /// Seven-bit ASCII, which every other encoding here agrees with.
30    Ascii,
31    /// UTF-8, with or without a byte-order mark.
32    Utf8,
33    /// UTF-16, big-endian.
34    Utf16Be,
35    /// UTF-16, little-endian.
36    Utf16Le,
37    /// ANSEL, the GEDCOM 5.5 default.
38    Ansel,
39    /// Windows-1252, which GEDCOM files call `ANSI`.
40    Ansi,
41}
42
43impl GedcomEncoding {
44    /// The name to show a user.
45    #[must_use]
46    pub const fn label(self) -> &'static str {
47        match self {
48            Self::Ascii => "ASCII",
49            Self::Utf8 => "UTF-8",
50            Self::Utf16Be => "UTF-16 (big-endian)",
51            Self::Utf16Le => "UTF-16 (little-endian)",
52            Self::Ansel => "ANSEL",
53            Self::Ansi => "ANSI (Windows-1252)",
54        }
55    }
56}
57
58impl fmt::Display for GedcomEncoding {
59    fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
60        formatter.write_str(self.label())
61    }
62}
63
64/// What decoding did, and what it had to decide.
65#[derive(Clone, Debug, Default, Eq, PartialEq)]
66#[non_exhaustive]
67#[cfg_attr(feature = "serde", derive(serde::Serialize))]
68#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))]
69pub struct EncodingReport {
70    /// The encoding the bytes were read as.
71    pub used: Option<GedcomEncoding>,
72    /// The `1 CHAR` payload exactly as the file wrote it, when it had one.
73    pub declared: Option<String>,
74    /// Whether a byte-order mark was found and removed.
75    pub byte_order_mark: bool,
76    /// Bytes that no mapping covered and that became U+FFFD.
77    pub undecodable_bytes: usize,
78    /// Everything a user should read before accepting the import.
79    pub warnings: Vec<String>,
80}
81
82impl EncodingReport {
83    /// A report that says only which encoding applies — what a caller that
84    /// serialized the document itself (always UTF-8) reports, since nothing
85    /// was decoded and there is nothing to warn about.
86    #[must_use]
87    pub fn for_encoding(encoding: GedcomEncoding) -> Self {
88        Self {
89            used: Some(encoding),
90            ..Self::default()
91        }
92    }
93
94    /// A one-line summary for the import preview.
95    #[must_use]
96    pub fn summary(&self) -> String {
97        let used = self.used.map_or("unknown", GedcomEncoding::label);
98        let mut text = format!("Read as {used}");
99        if self.byte_order_mark {
100            text.push_str(" (byte-order mark)");
101        }
102        match &self.declared {
103            Some(declared) => {
104                let _ = fmt::Write::write_fmt(&mut text, format_args!(", declared {declared}"));
105            }
106            None => text.push_str(", no CHAR declared"),
107        }
108        if self.undecodable_bytes > 0 {
109            let _ = fmt::Write::write_fmt(
110                &mut text,
111                format_args!(", {} byte(s) undecodable", self.undecodable_bytes),
112            );
113        }
114        text
115    }
116}
117
118/// Reads GEDCOM bytes as text, choosing an encoding and reporting the choice.
119///
120/// The order of authority is: a byte-order mark, then UTF-16 recognized from
121/// its null bytes, then the `1 CHAR` declaration, then a fallback that prefers
122/// UTF-8 when the bytes are valid UTF-8 and ANSEL when they are not.
123///
124/// # Errors
125///
126/// Returns [`GedcomError`] when the input exceeds the limit. A file that
127/// cannot be decoded cleanly is *not* an error: undecodable bytes become
128/// U+FFFD and are counted, so the user can see the damage and decide, rather
129/// than losing the whole file to a few bytes.
130pub fn decode_gedcom(bytes: &[u8]) -> Result<(String, EncodingReport), GedcomError> {
131    decode_gedcom_with(bytes, Limits::DEFAULT)
132}
133
134/// Reads GEDCOM bytes as text with explicit limits.
135///
136/// # Errors
137///
138/// As [`decode_gedcom`].
139pub fn decode_gedcom_with(
140    bytes: &[u8],
141    limits: Limits,
142) -> Result<(String, EncodingReport), GedcomError> {
143    if bytes.len() > limits.input_bytes {
144        return Err(fault(
145            0,
146            GedcomErrorKind::Limit,
147            format!(
148                "the file is {} MB and the limit is {} MB",
149                bytes.len() / (1024 * 1024),
150                limits.input_bytes / (1024 * 1024)
151            ),
152        ));
153    }
154
155    let mut report = EncodingReport::default();
156
157    let (body, marked) = strip_byte_order_mark(bytes);
158    report.byte_order_mark = marked.is_some();
159
160    // A mark settles it: the file says what it is in its own first bytes.
161    if let Some(encoding) = marked {
162        report.declared = declared_charset(body, encoding);
163        let text = decode_as(body, encoding, &mut report);
164        finish(encoding, &mut report);
165        return Ok((text, report));
166    }
167
168    // UTF-16 without a mark still announces itself: every GEDCOM begins with
169    // the ASCII digit `0`, so one half of that code unit is a null byte.
170    if let Some(encoding) = unmarked_utf16(body) {
171        report.declared = declared_charset(body, encoding);
172        report.warnings.push(format!(
173            "File is {encoding} but carries no byte-order mark."
174        ));
175        let text = decode_as(body, encoding, &mut report);
176        finish(encoding, &mut report);
177        return Ok((text, report));
178    }
179
180    // Everything below is byte-oriented, so the header can be read as ASCII.
181    report.declared = declared_charset(body, GedcomEncoding::Utf8);
182    let declared = report
183        .declared
184        .as_ref()
185        .map(|value| value.trim().to_owned());
186    let encoding = match declared.as_deref() {
187        Some(declared) => match_declared(declared, body, &mut report),
188        None => {
189            // No declaration. UTF-8 is a superset of ASCII, so trying it first
190            // never damages a plain file, and failing it is good evidence the
191            // file is one of the older single-byte sets.
192            if std::str::from_utf8(body).is_ok() {
193                GedcomEncoding::Utf8
194            } else {
195                report.warnings.push(
196                    "File declares no CHAR and is not valid UTF-8; read as ANSEL, the GEDCOM 5.5 default."
197                        .to_owned(),
198                );
199                GedcomEncoding::Ansel
200            }
201        }
202    };
203
204    let text = decode_as(body, encoding, &mut report);
205    finish(encoding, &mut report);
206    Ok((text, report))
207}
208
209fn finish(encoding: GedcomEncoding, report: &mut EncodingReport) {
210    report.used = Some(encoding);
211    if report.undecodable_bytes > 0 {
212        report.warnings.push(format!(
213            "{} byte(s) had no mapping in {encoding} and were replaced with U+FFFD.",
214            report.undecodable_bytes
215        ));
216    }
217}
218
219/// Resolves a `CHAR` payload to an encoding, warning when the file's contents
220/// contradict it.
221fn match_declared(declared: &str, body: &[u8], report: &mut EncodingReport) -> GedcomEncoding {
222    let normalized = declared
223        .chars()
224        .filter(char::is_ascii_alphanumeric)
225        .collect::<String>()
226        .to_ascii_uppercase();
227
228    match normalized.as_str() {
229        "UTF8" => {
230            if std::str::from_utf8(body).is_ok() {
231                GedcomEncoding::Utf8
232            } else {
233                // Windows-1252 maps every byte, so it recovers the text rather
234                // than losing the file. The mismatch is the story to tell.
235                report.warnings.push(
236                    "File declares UTF-8 but contains invalid UTF-8; read as Windows-1252 instead."
237                        .to_owned(),
238                );
239                GedcomEncoding::Ansi
240            }
241        }
242        "ANSEL" => GedcomEncoding::Ansel,
243        "ANSI" | "WINDOWS1252" | "CP1252" | "IBMWINDOWS" => GedcomEncoding::Ansi,
244        "ASCII" | "USASCII" | "ANSIZ3947" => GedcomEncoding::Ascii,
245        "UNICODE" | "UTF16" => {
246            // 5.5.1 requires a byte-order mark for UNICODE and this file has
247            // none, or an earlier branch would have taken it.
248            report.warnings.push(
249                "File declares UNICODE but has no byte-order mark and no UTF-16 structure; read as UTF-8."
250                    .to_owned(),
251            );
252            GedcomEncoding::Utf8
253        }
254        _ => {
255            report.warnings.push(format!(
256                "Unrecognized CHAR value {declared:?}; read as UTF-8."
257            ));
258            GedcomEncoding::Utf8
259        }
260    }
261}
262
263fn strip_byte_order_mark(bytes: &[u8]) -> (&[u8], Option<GedcomEncoding>) {
264    if let Some(rest) = bytes.strip_prefix(&[0xEF, 0xBB, 0xBF]) {
265        return (rest, Some(GedcomEncoding::Utf8));
266    }
267    if let Some(rest) = bytes.strip_prefix(&[0xFE, 0xFF]) {
268        return (rest, Some(GedcomEncoding::Utf16Be));
269    }
270    if let Some(rest) = bytes.strip_prefix(&[0xFF, 0xFE]) {
271        return (rest, Some(GedcomEncoding::Utf16Le));
272    }
273    (bytes, None)
274}
275
276const fn unmarked_utf16(bytes: &[u8]) -> Option<GedcomEncoding> {
277    match bytes {
278        [0x00, second, ..] if *second != 0x00 => Some(GedcomEncoding::Utf16Be),
279        [first, 0x00, ..] if *first != 0x00 => Some(GedcomEncoding::Utf16Le),
280        _ => None,
281    }
282}
283
284/// Finds the `1 CHAR` payload by reading the header as ASCII.
285///
286/// Every encoding here agrees with ASCII on the bytes a header is written in,
287/// so this works before the encoding is known — which is the point.
288fn declared_charset(bytes: &[u8], encoding: GedcomEncoding) -> Option<String> {
289    let head = &bytes[..bytes.len().min(HEADER_SCAN_BYTES)];
290    let text = match encoding {
291        GedcomEncoding::Utf16Be | GedcomEncoding::Utf16Le => {
292            let mut scratch = EncodingReport::default();
293            decode_utf16(head, encoding == GedcomEncoding::Utf16Be, &mut scratch)
294        }
295        _ => head.iter().map(|byte| char::from(*byte)).collect(),
296    };
297
298    text.lines()
299        .map(str::trim_end)
300        .find_map(|line| line.strip_prefix("1 CHAR "))
301        .map(str::trim)
302        .filter(|value| !value.is_empty())
303        .map(str::to_owned)
304}
305
306fn decode_as(bytes: &[u8], encoding: GedcomEncoding, report: &mut EncodingReport) -> String {
307    match encoding {
308        GedcomEncoding::Utf16Be => decode_utf16(bytes, true, report),
309        GedcomEncoding::Utf16Le => decode_utf16(bytes, false, report),
310        GedcomEncoding::Ansel => decode_ansel(bytes, report),
311        GedcomEncoding::Ansi => decode_windows_1252(bytes),
312        GedcomEncoding::Ascii => decode_ascii(bytes, report),
313        GedcomEncoding::Utf8 => std::str::from_utf8(bytes).map_or_else(
314            |_| {
315                let text = String::from_utf8_lossy(bytes).into_owned();
316                report.undecodable_bytes += text.matches('\u{FFFD}').count();
317                text
318            },
319            str::to_owned,
320        ),
321    }
322}
323
324fn decode_ascii(bytes: &[u8], report: &mut EncodingReport) -> String {
325    bytes
326        .iter()
327        .map(|byte| {
328            if byte.is_ascii() {
329                char::from(*byte)
330            } else {
331                report.undecodable_bytes += 1;
332                '\u{FFFD}'
333            }
334        })
335        .collect()
336}
337
338fn decode_utf16(bytes: &[u8], big_endian: bool, report: &mut EncodingReport) -> String {
339    if !bytes.len().is_multiple_of(2) {
340        report
341            .warnings
342            .push("UTF-16 file has an odd number of bytes; the final byte was ignored.".to_owned());
343    }
344    let units = bytes
345        .as_chunks::<2>()
346        .0
347        .iter()
348        .map(|pair| {
349            if big_endian {
350                u16::from_be_bytes([pair[0], pair[1]])
351            } else {
352                u16::from_le_bytes([pair[0], pair[1]])
353            }
354        })
355        .collect::<Vec<_>>();
356
357    let mut text = String::with_capacity(units.len());
358    for unit in char::decode_utf16(units) {
359        text.push(unit.unwrap_or_else(|_| {
360            report.undecodable_bytes += 1;
361            '\u{FFFD}'
362        }));
363    }
364    text
365}
366
367/// Windows-1252, which differs from Latin-1 only in `0x80`–`0x9F`.
368fn decode_windows_1252(bytes: &[u8]) -> String {
369    bytes
370        .iter()
371        .map(|byte| match byte {
372            0x80..=0x9F => WINDOWS_1252_HIGH[usize::from(byte - 0x80)],
373            other => char::from(*other),
374        })
375        .collect()
376}
377
378/// The five positions Windows-1252 leaves undefined map to the C1 control they
379/// sit on, which keeps the text the same length and loses nothing meaningful.
380pub const WINDOWS_1252_HIGH: [char; 32] = [
381    '\u{20AC}', '\u{0081}', '\u{201A}', '\u{0192}', '\u{201E}', '\u{2026}', '\u{2020}', '\u{2021}',
382    '\u{02C6}', '\u{2030}', '\u{0160}', '\u{2039}', '\u{0152}', '\u{008D}', '\u{017D}', '\u{008F}',
383    '\u{0090}', '\u{2018}', '\u{2019}', '\u{201C}', '\u{201D}', '\u{2022}', '\u{2013}', '\u{2014}',
384    '\u{02DC}', '\u{2122}', '\u{0161}', '\u{203A}', '\u{0153}', '\u{009D}', '\u{017E}', '\u{0178}',
385];
386
387/// Decodes ANSEL as GEDCOM 5.5 Appendix D defines it.
388///
389/// The one structural difference from every other encoding here: a diacritic
390/// comes *before* the letter it modifies, where Unicode puts it after. So a
391/// run of marks is buffered, the base character is emitted, and the marks
392/// follow it in the order they were written.
393fn decode_ansel(bytes: &[u8], report: &mut EncodingReport) -> String {
394    let mut text = String::with_capacity(bytes.len());
395    let mut marks = Vec::new();
396    let mut index = 0;
397
398    while index < bytes.len() {
399        let byte = bytes[index];
400        index += 1;
401
402        if let Some(mark) = ansel_combining(byte) {
403            marks.push(mark);
404            continue;
405        }
406
407        let base = if byte.is_ascii() {
408            char::from(byte)
409        } else if let Some(character) = ansel_graphic(byte) {
410            character
411        } else {
412            report.undecodable_bytes += 1;
413            '\u{FFFD}'
414        };
415
416        text.push(base);
417        // A mark with nothing after it is written anyway rather than dropped;
418        // losing evidence silently is worse than an odd-looking character.
419        text.extend(marks.iter().copied());
420        marks.clear();
421    }
422
423    text.extend(marks);
424    text
425}
426
427/// The combining diacritics, `0xE0`–`0xFB` and `0xFE`.
428const fn ansel_combining(byte: u8) -> Option<char> {
429    Some(match byte {
430        0xE0 => '\u{0309}', // hook above
431        0xE1 => '\u{0300}', // grave
432        0xE2 => '\u{0301}', // acute
433        0xE3 => '\u{0302}', // circumflex
434        0xE4 => '\u{0303}', // tilde
435        0xE5 => '\u{0304}', // macron
436        0xE6 => '\u{0306}', // breve
437        0xE7 => '\u{0307}', // dot above
438        0xE8 => '\u{0308}', // diaeresis
439        0xE9 => '\u{030C}', // caron
440        0xEA => '\u{030A}', // ring above
441        0xEB => '\u{FE20}', // ligature, left half
442        0xEC => '\u{FE21}', // ligature, right half
443        0xED => '\u{0315}', // comma above right
444        0xEE => '\u{030B}', // double acute
445        0xEF => '\u{0310}', // candrabindu
446        0xF0 => '\u{0327}', // cedilla
447        0xF1 => '\u{0328}', // ogonek
448        0xF2 => '\u{0323}', // dot below
449        0xF3 => '\u{0324}', // diaeresis below
450        0xF4 => '\u{0325}', // ring below
451        0xF5 => '\u{0333}', // double low line
452        0xF6 => '\u{0332}', // line below
453        0xF7 => '\u{0326}', // comma below
454        0xF8 => '\u{031C}', // left half ring below
455        0xF9 => '\u{032E}', // breve below
456        0xFA => '\u{FE22}', // double tilde, left half
457        0xFB => '\u{FE23}', // double tilde, right half
458        0xFE => '\u{0313}', // comma above
459        _ => return None,
460    })
461}
462
463/// The non-combining graphic characters, `0xA1`–`0xCF`, including the two LDS
464/// box extensions and the two midline letters GEDCOM 5.5 adds.
465const fn ansel_graphic(byte: u8) -> Option<char> {
466    Some(match byte {
467        0xA1 => 'Ł',
468        0xA2 => 'Ø',
469        0xA3 => 'Đ',
470        0xA4 => 'Þ',
471        0xA5 => 'Æ',
472        0xA6 => 'Œ',
473        0xA7 => '\u{02B9}', // single prime
474        0xA8 => '·',
475        0xA9 => '\u{266D}', // musical flat
476        0xAA => '®',
477        0xAB => '±',
478        0xAC => 'Ơ',
479        0xAD => 'Ư',
480        0xAE => '\u{02BB}', // left half ring
481        0xB0 => '\u{02BC}', // right half ring
482        0xB1 => 'ł',
483        0xB2 => 'ø',
484        0xB3 => 'đ',
485        0xB4 => 'þ',
486        0xB5 => 'æ',
487        0xB6 => 'œ',
488        0xB7 => '\u{02BA}', // double prime
489        0xB8 => 'ı',
490        0xB9 => '£',
491        0xBA => 'ð',
492        0xBC => 'ơ',
493        0xBD => 'ư',
494        0xBE => '\u{25A1}', // empty box, an LDS extension
495        0xBF => '\u{25A0}', // black box, an LDS extension
496        0xC0 => '°',
497        0xC1 => '\u{2113}', // script l
498        0xC2 => '\u{2117}', // phonograph copyright
499        0xC3 => '©',
500        0xC4 => '\u{266F}', // musical sharp
501        0xC5 => '¿',
502        0xC6 => '¡',
503        0xCD => 'e', // midline e, an LDS extension with no Unicode of its own
504        0xCE => 'o', // midline o, likewise
505        0xCF => 'ß',
506        _ => return None,
507    })
508}
509
510#[cfg(test)]
511mod tests {
512    use super::*;
513
514    #[test]
515    fn a_utf8_byte_order_mark_is_stripped_and_reported() {
516        let bytes = b"\xEF\xBB\xBF0 HEAD\n0 TRLR\n";
517
518        let (text, report) = decode_gedcom(bytes).expect("decode");
519
520        assert!(
521            text.starts_with("0 HEAD"),
522            "mark must not survive: {text:?}"
523        );
524        assert!(report.byte_order_mark);
525        assert_eq!(report.used, Some(GedcomEncoding::Utf8));
526    }
527
528    #[test]
529    fn utf16_is_read_with_or_without_a_byte_order_mark() {
530        let marked = b"\xFF\xFE0\x00 \x00H\x00E\x00A\x00D\x00";
531        let (text, report) = decode_gedcom(marked).expect("decode marked");
532        assert_eq!(text, "0 HEAD");
533        assert_eq!(report.used, Some(GedcomEncoding::Utf16Le));
534        assert!(report.byte_order_mark);
535
536        let bare = b"\x000\x00 \x00H\x00E\x00A\x00D";
537        let (text, report) = decode_gedcom(bare).expect("decode bare");
538        assert_eq!(text, "0 HEAD");
539        assert_eq!(report.used, Some(GedcomEncoding::Utf16Be));
540        assert!(!report.byte_order_mark);
541        assert!(
542            !report.warnings.is_empty(),
543            "a missing mark is worth saying"
544        );
545    }
546
547    #[test]
548    fn ansel_puts_a_diacritic_after_the_letter_it_modifies() {
549        // "1 NAME Jos<acute>e" — ANSEL writes the accent first.
550        let bytes = b"0 HEAD\n1 CHAR ANSEL\n0 @I1@ INDI\n1 NAME Jos\xE2e\n";
551
552        let (text, report) = decode_gedcom(bytes).expect("decode");
553
554        assert_eq!(report.used, Some(GedcomEncoding::Ansel));
555        assert_eq!(report.declared.as_deref(), Some("ANSEL"));
556        assert!(text.contains("Jose\u{0301}"), "got {text:?}");
557        assert_eq!(report.undecodable_bytes, 0);
558    }
559
560    #[test]
561    fn ansel_graphic_characters_decode() {
562        let bytes = b"0 HEAD\n1 CHAR ANSEL\n1 NOTE \xA2 \xB2 \xCF\n";
563
564        let (text, _) = decode_gedcom(bytes).expect("decode");
565
566        assert!(text.contains("Ø ø ß"), "got {text:?}");
567    }
568
569    #[test]
570    fn a_declared_charset_that_the_bytes_contradict_is_reported_not_obeyed() {
571        // Declares UTF-8, contains a lone 0xE9 — Windows-1252 recovers it.
572        let bytes = b"0 HEAD\n1 CHAR UTF-8\n1 NOTE caf\xE9\n";
573
574        let (text, report) = decode_gedcom(bytes).expect("decode");
575
576        assert_eq!(report.used, Some(GedcomEncoding::Ansi));
577        assert_eq!(report.declared.as_deref(), Some("UTF-8"));
578        assert!(text.contains("café"), "got {text:?}");
579        assert!(
580            report
581                .warnings
582                .iter()
583                .any(|warning| warning.contains("declares UTF-8")),
584            "{:?}",
585            report.warnings
586        );
587    }
588
589    #[test]
590    fn an_undeclared_file_prefers_utf8_and_falls_back_to_ansel() {
591        let utf8 = "0 HEAD\n1 NOTE café\n".as_bytes();
592        let (text, report) = decode_gedcom(utf8).expect("decode utf8");
593        assert_eq!(report.used, Some(GedcomEncoding::Utf8));
594        assert!(text.contains("café"));
595        assert!(report.declared.is_none());
596
597        let ansel = b"0 HEAD\n1 NOTE caf\xE2e\n";
598        let (text, report) = decode_gedcom(ansel).expect("decode ansel");
599        assert_eq!(report.used, Some(GedcomEncoding::Ansel));
600        assert!(text.contains("cafe\u{0301}"), "got {text:?}");
601    }
602
603    #[test]
604    fn utf16_surrogate_pairs_decode_to_the_characters_they_spell() {
605        // "0 HEAD\n1 NOTE 𝄞" — the clef is U+1D11E, a surrogate pair in
606        // UTF-16, which the per-unit reader must join rather than replace.
607        let mut bytes = vec![0xFFu8, 0xFE];
608        for unit in "0 HEAD\n1 NOTE \u{1D11E}\n".encode_utf16() {
609            bytes.extend_from_slice(&unit.to_le_bytes());
610        }
611
612        let (text, report) = decode_gedcom(&bytes).expect("decode");
613
614        assert_eq!(report.undecodable_bytes, 0);
615        assert!(text.contains('\u{1D11E}'), "got {text:?}");
616    }
617
618    #[test]
619    fn a_char_line_beyond_the_header_scan_window_falls_back_honestly() {
620        // The declaration hunt reads a bounded prefix; a CHAR pushed past it
621        // by an absurd header is treated as undeclared — valid UTF-8 reads as
622        // UTF-8 — rather than making the scan unbounded.
623        use std::fmt::Write as _;
624        let mut text = String::from("0 HEAD\n");
625        for index in 0..600 {
626            let _ = writeln!(
627                text,
628                "1 NOTE padding line number {index} to push the declaration far down"
629            );
630        }
631        text.push_str("1 CHAR ANSEL\n0 TRLR\n");
632
633        let (_, report) = decode_gedcom(text.as_bytes()).expect("decode");
634
635        assert_eq!(report.declared, None, "the declaration is out of reach");
636        assert_eq!(report.used, Some(GedcomEncoding::Utf8));
637    }
638
639    #[test]
640    fn an_oversized_file_is_refused_with_the_limit_named() {
641        // The limit is passed in rather than allocating the default one, which
642        // is 128 MB and would make this test cost more than it proves.
643        let limits = Limits {
644            input_bytes: 2 * 1024 * 1024,
645            ..Limits::DEFAULT
646        };
647        let bytes = vec![b'0'; limits.input_bytes + 1];
648
649        let error = decode_gedcom_with(&bytes, limits).expect_err("must refuse");
650
651        let message = error.to_string();
652        assert!(message.contains("2 MB"), "limit must be named: {message}");
653    }
654
655    #[test]
656    fn undecodable_bytes_are_counted_rather_than_failing_the_import() {
657        let bytes = b"0 HEAD\n1 CHAR ANSEL\n1 NOTE \xD0\n";
658
659        let (text, report) = decode_gedcom(bytes).expect("decode");
660
661        assert_eq!(report.undecodable_bytes, 1);
662        assert!(text.contains('\u{FFFD}'));
663        assert!(!report.warnings.is_empty());
664    }
665}