gedcomkit 0.1.6

A byte-preserving GEDCOM document model: decoding, parsing, readings, version conversion, plausibility checks, and the GEDZIP container, for GEDCOM 5.5 through 7.x.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
//! Turning the bytes of a GEDCOM file into text.
//!
//! This is a layer of its own because getting it wrong silently rejects most
//! real files. A byte-order mark is the common case — every GEDCOM 7 file
//! gedcom.io publishes carries one, as does every export from Family Tree
//! Maker 24, Legacy 10, `RootsMagic` 8, and Ancestris 11 — and an unstripped mark
//! makes the first line unparseable for a reason that names the wrong thing.
//!
//! Older files are not UTF-8 at all. GEDCOM 5.5 defaults to ANSEL, a 1980s
//! library-cataloguing set whose diacritics *precede* the letter they modify,
//! and files declaring `ANSI` are Windows-1252 in practice.
//!
//! Nothing here guesses in silence. Every decision, and every disagreement
//! between what a file declares and what it contains, is reported so the import
//! preview can show it before anything is written.

use crate::{GedcomError, GedcomErrorKind, Limits, fault};
use std::fmt;

/// How many leading bytes are searched for the `CHAR` declaration. The header
/// is the first record, and no real one approaches this.
const HEADER_SCAN_BYTES: usize = 8 * 1024;

/// The character encoding a GEDCOM file was read as.
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
#[non_exhaustive]
#[cfg_attr(feature = "serde", derive(serde::Serialize))]
pub enum GedcomEncoding {
    /// Seven-bit ASCII, which every other encoding here agrees with.
    Ascii,
    /// UTF-8, with or without a byte-order mark.
    Utf8,
    /// UTF-16, big-endian.
    Utf16Be,
    /// UTF-16, little-endian.
    Utf16Le,
    /// ANSEL, the GEDCOM 5.5 default.
    Ansel,
    /// Windows-1252, which GEDCOM files call `ANSI`.
    Ansi,
}

impl GedcomEncoding {
    /// The name to show a user.
    #[must_use]
    pub const fn label(self) -> &'static str {
        match self {
            Self::Ascii => "ASCII",
            Self::Utf8 => "UTF-8",
            Self::Utf16Be => "UTF-16 (big-endian)",
            Self::Utf16Le => "UTF-16 (little-endian)",
            Self::Ansel => "ANSEL",
            Self::Ansi => "ANSI (Windows-1252)",
        }
    }
}

impl fmt::Display for GedcomEncoding {
    fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
        formatter.write_str(self.label())
    }
}

/// What decoding did, and what it had to decide.
#[derive(Clone, Debug, Default, Eq, PartialEq)]
#[non_exhaustive]
#[cfg_attr(feature = "serde", derive(serde::Serialize))]
#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))]
pub struct EncodingReport {
    /// The encoding the bytes were read as.
    pub used: Option<GedcomEncoding>,
    /// The `1 CHAR` payload exactly as the file wrote it, when it had one.
    pub declared: Option<String>,
    /// Whether a byte-order mark was found and removed.
    pub byte_order_mark: bool,
    /// Bytes that no mapping covered and that became U+FFFD.
    pub undecodable_bytes: usize,
    /// Everything a user should read before accepting the import.
    pub warnings: Vec<String>,
}

impl EncodingReport {
    /// A report that says only which encoding applies — what a caller that
    /// serialized the document itself (always UTF-8) reports, since nothing
    /// was decoded and there is nothing to warn about.
    #[must_use]
    pub fn for_encoding(encoding: GedcomEncoding) -> Self {
        Self {
            used: Some(encoding),
            ..Self::default()
        }
    }

    /// A one-line summary for the import preview.
    #[must_use]
    pub fn summary(&self) -> String {
        let used = self.used.map_or("unknown", GedcomEncoding::label);
        let mut text = format!("Read as {used}");
        if self.byte_order_mark {
            text.push_str(" (byte-order mark)");
        }
        match &self.declared {
            Some(declared) => {
                let _ = fmt::Write::write_fmt(&mut text, format_args!(", declared {declared}"));
            }
            None => text.push_str(", no CHAR declared"),
        }
        if self.undecodable_bytes > 0 {
            let _ = fmt::Write::write_fmt(
                &mut text,
                format_args!(", {} byte(s) undecodable", self.undecodable_bytes),
            );
        }
        text
    }
}

/// Reads GEDCOM bytes as text, choosing an encoding and reporting the choice.
///
/// The order of authority is: a byte-order mark, then UTF-16 recognized from
/// its null bytes, then the `1 CHAR` declaration, then a fallback that prefers
/// UTF-8 when the bytes are valid UTF-8 and ANSEL when they are not.
///
/// # Errors
///
/// Returns [`GedcomError`] when the input exceeds the limit. A file that
/// cannot be decoded cleanly is *not* an error: undecodable bytes become
/// U+FFFD and are counted, so the user can see the damage and decide, rather
/// than losing the whole file to a few bytes.
pub fn decode_gedcom(bytes: &[u8]) -> Result<(String, EncodingReport), GedcomError> {
    decode_gedcom_with(bytes, Limits::DEFAULT)
}

/// Reads GEDCOM bytes as text with explicit limits.
///
/// # Errors
///
/// As [`decode_gedcom`].
pub fn decode_gedcom_with(
    bytes: &[u8],
    limits: Limits,
) -> Result<(String, EncodingReport), GedcomError> {
    if bytes.len() > limits.input_bytes {
        return Err(fault(
            0,
            GedcomErrorKind::Limit,
            format!(
                "the file is {} MB and the limit is {} MB",
                bytes.len() / (1024 * 1024),
                limits.input_bytes / (1024 * 1024)
            ),
        ));
    }

    let mut report = EncodingReport::default();

    let (body, marked) = strip_byte_order_mark(bytes);
    report.byte_order_mark = marked.is_some();

    // A mark settles it: the file says what it is in its own first bytes.
    if let Some(encoding) = marked {
        report.declared = declared_charset(body, encoding);
        let text = decode_as(body, encoding, &mut report);
        finish(encoding, &mut report);
        return Ok((text, report));
    }

    // UTF-16 without a mark still announces itself: every GEDCOM begins with
    // the ASCII digit `0`, so one half of that code unit is a null byte.
    if let Some(encoding) = unmarked_utf16(body) {
        report.declared = declared_charset(body, encoding);
        report.warnings.push(format!(
            "File is {encoding} but carries no byte-order mark."
        ));
        let text = decode_as(body, encoding, &mut report);
        finish(encoding, &mut report);
        return Ok((text, report));
    }

    // Everything below is byte-oriented, so the header can be read as ASCII.
    report.declared = declared_charset(body, GedcomEncoding::Utf8);
    let declared = report
        .declared
        .as_ref()
        .map(|value| value.trim().to_owned());
    let encoding = match declared.as_deref() {
        Some(declared) => match_declared(declared, body, &mut report),
        None => {
            // No declaration. UTF-8 is a superset of ASCII, so trying it first
            // never damages a plain file, and failing it is good evidence the
            // file is one of the older single-byte sets.
            if std::str::from_utf8(body).is_ok() {
                GedcomEncoding::Utf8
            } else {
                report.warnings.push(
                    "File declares no CHAR and is not valid UTF-8; read as ANSEL, the GEDCOM 5.5 default."
                        .to_owned(),
                );
                GedcomEncoding::Ansel
            }
        }
    };

    let text = decode_as(body, encoding, &mut report);
    finish(encoding, &mut report);
    Ok((text, report))
}

fn finish(encoding: GedcomEncoding, report: &mut EncodingReport) {
    report.used = Some(encoding);
    if report.undecodable_bytes > 0 {
        report.warnings.push(format!(
            "{} byte(s) had no mapping in {encoding} and were replaced with U+FFFD.",
            report.undecodable_bytes
        ));
    }
}

/// Resolves a `CHAR` payload to an encoding, warning when the file's contents
/// contradict it.
fn match_declared(declared: &str, body: &[u8], report: &mut EncodingReport) -> GedcomEncoding {
    let normalized = declared
        .chars()
        .filter(char::is_ascii_alphanumeric)
        .collect::<String>()
        .to_ascii_uppercase();

    match normalized.as_str() {
        "UTF8" => {
            if std::str::from_utf8(body).is_ok() {
                GedcomEncoding::Utf8
            } else {
                // Windows-1252 maps every byte, so it recovers the text rather
                // than losing the file. The mismatch is the story to tell.
                report.warnings.push(
                    "File declares UTF-8 but contains invalid UTF-8; read as Windows-1252 instead."
                        .to_owned(),
                );
                GedcomEncoding::Ansi
            }
        }
        "ANSEL" => GedcomEncoding::Ansel,
        "ANSI" | "WINDOWS1252" | "CP1252" | "IBMWINDOWS" => GedcomEncoding::Ansi,
        "ASCII" | "USASCII" | "ANSIZ3947" => GedcomEncoding::Ascii,
        "UNICODE" | "UTF16" => {
            // 5.5.1 requires a byte-order mark for UNICODE and this file has
            // none, or an earlier branch would have taken it.
            report.warnings.push(
                "File declares UNICODE but has no byte-order mark and no UTF-16 structure; read as UTF-8."
                    .to_owned(),
            );
            GedcomEncoding::Utf8
        }
        _ => {
            report.warnings.push(format!(
                "Unrecognized CHAR value {declared:?}; read as UTF-8."
            ));
            GedcomEncoding::Utf8
        }
    }
}

fn strip_byte_order_mark(bytes: &[u8]) -> (&[u8], Option<GedcomEncoding>) {
    if let Some(rest) = bytes.strip_prefix(&[0xEF, 0xBB, 0xBF]) {
        return (rest, Some(GedcomEncoding::Utf8));
    }
    if let Some(rest) = bytes.strip_prefix(&[0xFE, 0xFF]) {
        return (rest, Some(GedcomEncoding::Utf16Be));
    }
    if let Some(rest) = bytes.strip_prefix(&[0xFF, 0xFE]) {
        return (rest, Some(GedcomEncoding::Utf16Le));
    }
    (bytes, None)
}

const fn unmarked_utf16(bytes: &[u8]) -> Option<GedcomEncoding> {
    match bytes {
        [0x00, second, ..] if *second != 0x00 => Some(GedcomEncoding::Utf16Be),
        [first, 0x00, ..] if *first != 0x00 => Some(GedcomEncoding::Utf16Le),
        _ => None,
    }
}

/// Finds the `1 CHAR` payload by reading the header as ASCII.
///
/// Every encoding here agrees with ASCII on the bytes a header is written in,
/// so this works before the encoding is known — which is the point.
fn declared_charset(bytes: &[u8], encoding: GedcomEncoding) -> Option<String> {
    let head = &bytes[..bytes.len().min(HEADER_SCAN_BYTES)];
    let text = match encoding {
        GedcomEncoding::Utf16Be | GedcomEncoding::Utf16Le => {
            let mut scratch = EncodingReport::default();
            decode_utf16(head, encoding == GedcomEncoding::Utf16Be, &mut scratch)
        }
        _ => head.iter().map(|byte| char::from(*byte)).collect(),
    };

    text.lines()
        .map(str::trim_end)
        .find_map(|line| line.strip_prefix("1 CHAR "))
        .map(str::trim)
        .filter(|value| !value.is_empty())
        .map(str::to_owned)
}

fn decode_as(bytes: &[u8], encoding: GedcomEncoding, report: &mut EncodingReport) -> String {
    match encoding {
        GedcomEncoding::Utf16Be => decode_utf16(bytes, true, report),
        GedcomEncoding::Utf16Le => decode_utf16(bytes, false, report),
        GedcomEncoding::Ansel => decode_ansel(bytes, report),
        GedcomEncoding::Ansi => decode_windows_1252(bytes),
        GedcomEncoding::Ascii => decode_ascii(bytes, report),
        GedcomEncoding::Utf8 => std::str::from_utf8(bytes).map_or_else(
            |_| {
                let text = String::from_utf8_lossy(bytes).into_owned();
                report.undecodable_bytes += text.matches('\u{FFFD}').count();
                text
            },
            str::to_owned,
        ),
    }
}

fn decode_ascii(bytes: &[u8], report: &mut EncodingReport) -> String {
    bytes
        .iter()
        .map(|byte| {
            if byte.is_ascii() {
                char::from(*byte)
            } else {
                report.undecodable_bytes += 1;
                '\u{FFFD}'
            }
        })
        .collect()
}

fn decode_utf16(bytes: &[u8], big_endian: bool, report: &mut EncodingReport) -> String {
    if !bytes.len().is_multiple_of(2) {
        report
            .warnings
            .push("UTF-16 file has an odd number of bytes; the final byte was ignored.".to_owned());
    }
    let units = bytes
        .as_chunks::<2>()
        .0
        .iter()
        .map(|pair| {
            if big_endian {
                u16::from_be_bytes([pair[0], pair[1]])
            } else {
                u16::from_le_bytes([pair[0], pair[1]])
            }
        })
        .collect::<Vec<_>>();

    let mut text = String::with_capacity(units.len());
    for unit in char::decode_utf16(units) {
        text.push(unit.unwrap_or_else(|_| {
            report.undecodable_bytes += 1;
            '\u{FFFD}'
        }));
    }
    text
}

/// Windows-1252, which differs from Latin-1 only in `0x80`–`0x9F`.
fn decode_windows_1252(bytes: &[u8]) -> String {
    bytes
        .iter()
        .map(|byte| match byte {
            0x80..=0x9F => WINDOWS_1252_HIGH[usize::from(byte - 0x80)],
            other => char::from(*other),
        })
        .collect()
}

/// The five positions Windows-1252 leaves undefined map to the C1 control they
/// sit on, which keeps the text the same length and loses nothing meaningful.
pub const WINDOWS_1252_HIGH: [char; 32] = [
    '\u{20AC}', '\u{0081}', '\u{201A}', '\u{0192}', '\u{201E}', '\u{2026}', '\u{2020}', '\u{2021}',
    '\u{02C6}', '\u{2030}', '\u{0160}', '\u{2039}', '\u{0152}', '\u{008D}', '\u{017D}', '\u{008F}',
    '\u{0090}', '\u{2018}', '\u{2019}', '\u{201C}', '\u{201D}', '\u{2022}', '\u{2013}', '\u{2014}',
    '\u{02DC}', '\u{2122}', '\u{0161}', '\u{203A}', '\u{0153}', '\u{009D}', '\u{017E}', '\u{0178}',
];

/// Decodes ANSEL as GEDCOM 5.5 Appendix D defines it.
///
/// The one structural difference from every other encoding here: a diacritic
/// comes *before* the letter it modifies, where Unicode puts it after. So a
/// run of marks is buffered, the base character is emitted, and the marks
/// follow it in the order they were written.
fn decode_ansel(bytes: &[u8], report: &mut EncodingReport) -> String {
    let mut text = String::with_capacity(bytes.len());
    let mut marks = Vec::new();
    let mut index = 0;

    while index < bytes.len() {
        let byte = bytes[index];
        index += 1;

        if let Some(mark) = ansel_combining(byte) {
            marks.push(mark);
            continue;
        }

        let base = if byte.is_ascii() {
            char::from(byte)
        } else if let Some(character) = ansel_graphic(byte) {
            character
        } else {
            report.undecodable_bytes += 1;
            '\u{FFFD}'
        };

        text.push(base);
        // A mark with nothing after it is written anyway rather than dropped;
        // losing evidence silently is worse than an odd-looking character.
        text.extend(marks.iter().copied());
        marks.clear();
    }

    text.extend(marks);
    text
}

/// The combining diacritics, `0xE0`–`0xFB` and `0xFE`.
const fn ansel_combining(byte: u8) -> Option<char> {
    Some(match byte {
        0xE0 => '\u{0309}', // hook above
        0xE1 => '\u{0300}', // grave
        0xE2 => '\u{0301}', // acute
        0xE3 => '\u{0302}', // circumflex
        0xE4 => '\u{0303}', // tilde
        0xE5 => '\u{0304}', // macron
        0xE6 => '\u{0306}', // breve
        0xE7 => '\u{0307}', // dot above
        0xE8 => '\u{0308}', // diaeresis
        0xE9 => '\u{030C}', // caron
        0xEA => '\u{030A}', // ring above
        0xEB => '\u{FE20}', // ligature, left half
        0xEC => '\u{FE21}', // ligature, right half
        0xED => '\u{0315}', // comma above right
        0xEE => '\u{030B}', // double acute
        0xEF => '\u{0310}', // candrabindu
        0xF0 => '\u{0327}', // cedilla
        0xF1 => '\u{0328}', // ogonek
        0xF2 => '\u{0323}', // dot below
        0xF3 => '\u{0324}', // diaeresis below
        0xF4 => '\u{0325}', // ring below
        0xF5 => '\u{0333}', // double low line
        0xF6 => '\u{0332}', // line below
        0xF7 => '\u{0326}', // comma below
        0xF8 => '\u{031C}', // left half ring below
        0xF9 => '\u{032E}', // breve below
        0xFA => '\u{FE22}', // double tilde, left half
        0xFB => '\u{FE23}', // double tilde, right half
        0xFE => '\u{0313}', // comma above
        _ => return None,
    })
}

/// The non-combining graphic characters, `0xA1`–`0xCF`, including the two LDS
/// box extensions and the two midline letters GEDCOM 5.5 adds.
const fn ansel_graphic(byte: u8) -> Option<char> {
    Some(match byte {
        0xA1 => 'Ł',
        0xA2 => 'Ø',
        0xA3 => 'Đ',
        0xA4 => 'Þ',
        0xA5 => 'Æ',
        0xA6 => 'Œ',
        0xA7 => '\u{02B9}', // single prime
        0xA8 => '·',
        0xA9 => '\u{266D}', // musical flat
        0xAA => '®',
        0xAB => '±',
        0xAC => 'Ơ',
        0xAD => 'Ư',
        0xAE => '\u{02BB}', // left half ring
        0xB0 => '\u{02BC}', // right half ring
        0xB1 => 'ł',
        0xB2 => 'ø',
        0xB3 => 'đ',
        0xB4 => 'þ',
        0xB5 => 'æ',
        0xB6 => 'œ',
        0xB7 => '\u{02BA}', // double prime
        0xB8 => 'ı',
        0xB9 => '£',
        0xBA => 'ð',
        0xBC => 'ơ',
        0xBD => 'ư',
        0xBE => '\u{25A1}', // empty box, an LDS extension
        0xBF => '\u{25A0}', // black box, an LDS extension
        0xC0 => '°',
        0xC1 => '\u{2113}', // script l
        0xC2 => '\u{2117}', // phonograph copyright
        0xC3 => '©',
        0xC4 => '\u{266F}', // musical sharp
        0xC5 => '¿',
        0xC6 => '¡',
        0xCD => 'e', // midline e, an LDS extension with no Unicode of its own
        0xCE => 'o', // midline o, likewise
        0xCF => 'ß',
        _ => return None,
    })
}

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn a_utf8_byte_order_mark_is_stripped_and_reported() {
        let bytes = b"\xEF\xBB\xBF0 HEAD\n0 TRLR\n";

        let (text, report) = decode_gedcom(bytes).expect("decode");

        assert!(
            text.starts_with("0 HEAD"),
            "mark must not survive: {text:?}"
        );
        assert!(report.byte_order_mark);
        assert_eq!(report.used, Some(GedcomEncoding::Utf8));
    }

    #[test]
    fn utf16_is_read_with_or_without_a_byte_order_mark() {
        let marked = b"\xFF\xFE0\x00 \x00H\x00E\x00A\x00D\x00";
        let (text, report) = decode_gedcom(marked).expect("decode marked");
        assert_eq!(text, "0 HEAD");
        assert_eq!(report.used, Some(GedcomEncoding::Utf16Le));
        assert!(report.byte_order_mark);

        let bare = b"\x000\x00 \x00H\x00E\x00A\x00D";
        let (text, report) = decode_gedcom(bare).expect("decode bare");
        assert_eq!(text, "0 HEAD");
        assert_eq!(report.used, Some(GedcomEncoding::Utf16Be));
        assert!(!report.byte_order_mark);
        assert!(
            !report.warnings.is_empty(),
            "a missing mark is worth saying"
        );
    }

    #[test]
    fn ansel_puts_a_diacritic_after_the_letter_it_modifies() {
        // "1 NAME Jos<acute>e" — ANSEL writes the accent first.
        let bytes = b"0 HEAD\n1 CHAR ANSEL\n0 @I1@ INDI\n1 NAME Jos\xE2e\n";

        let (text, report) = decode_gedcom(bytes).expect("decode");

        assert_eq!(report.used, Some(GedcomEncoding::Ansel));
        assert_eq!(report.declared.as_deref(), Some("ANSEL"));
        assert!(text.contains("Jose\u{0301}"), "got {text:?}");
        assert_eq!(report.undecodable_bytes, 0);
    }

    #[test]
    fn ansel_graphic_characters_decode() {
        let bytes = b"0 HEAD\n1 CHAR ANSEL\n1 NOTE \xA2 \xB2 \xCF\n";

        let (text, _) = decode_gedcom(bytes).expect("decode");

        assert!(text.contains("Ø ø ß"), "got {text:?}");
    }

    #[test]
    fn a_declared_charset_that_the_bytes_contradict_is_reported_not_obeyed() {
        // Declares UTF-8, contains a lone 0xE9 — Windows-1252 recovers it.
        let bytes = b"0 HEAD\n1 CHAR UTF-8\n1 NOTE caf\xE9\n";

        let (text, report) = decode_gedcom(bytes).expect("decode");

        assert_eq!(report.used, Some(GedcomEncoding::Ansi));
        assert_eq!(report.declared.as_deref(), Some("UTF-8"));
        assert!(text.contains("café"), "got {text:?}");
        assert!(
            report
                .warnings
                .iter()
                .any(|warning| warning.contains("declares UTF-8")),
            "{:?}",
            report.warnings
        );
    }

    #[test]
    fn an_undeclared_file_prefers_utf8_and_falls_back_to_ansel() {
        let utf8 = "0 HEAD\n1 NOTE café\n".as_bytes();
        let (text, report) = decode_gedcom(utf8).expect("decode utf8");
        assert_eq!(report.used, Some(GedcomEncoding::Utf8));
        assert!(text.contains("café"));
        assert!(report.declared.is_none());

        let ansel = b"0 HEAD\n1 NOTE caf\xE2e\n";
        let (text, report) = decode_gedcom(ansel).expect("decode ansel");
        assert_eq!(report.used, Some(GedcomEncoding::Ansel));
        assert!(text.contains("cafe\u{0301}"), "got {text:?}");
    }

    #[test]
    fn utf16_surrogate_pairs_decode_to_the_characters_they_spell() {
        // "0 HEAD\n1 NOTE 𝄞" — the clef is U+1D11E, a surrogate pair in
        // UTF-16, which the per-unit reader must join rather than replace.
        let mut bytes = vec![0xFFu8, 0xFE];
        for unit in "0 HEAD\n1 NOTE \u{1D11E}\n".encode_utf16() {
            bytes.extend_from_slice(&unit.to_le_bytes());
        }

        let (text, report) = decode_gedcom(&bytes).expect("decode");

        assert_eq!(report.undecodable_bytes, 0);
        assert!(text.contains('\u{1D11E}'), "got {text:?}");
    }

    #[test]
    fn a_char_line_beyond_the_header_scan_window_falls_back_honestly() {
        // The declaration hunt reads a bounded prefix; a CHAR pushed past it
        // by an absurd header is treated as undeclared — valid UTF-8 reads as
        // UTF-8 — rather than making the scan unbounded.
        use std::fmt::Write as _;
        let mut text = String::from("0 HEAD\n");
        for index in 0..600 {
            let _ = writeln!(
                text,
                "1 NOTE padding line number {index} to push the declaration far down"
            );
        }
        text.push_str("1 CHAR ANSEL\n0 TRLR\n");

        let (_, report) = decode_gedcom(text.as_bytes()).expect("decode");

        assert_eq!(report.declared, None, "the declaration is out of reach");
        assert_eq!(report.used, Some(GedcomEncoding::Utf8));
    }

    #[test]
    fn an_oversized_file_is_refused_with_the_limit_named() {
        // The limit is passed in rather than allocating the default one, which
        // is 128 MB and would make this test cost more than it proves.
        let limits = Limits {
            input_bytes: 2 * 1024 * 1024,
            ..Limits::DEFAULT
        };
        let bytes = vec![b'0'; limits.input_bytes + 1];

        let error = decode_gedcom_with(&bytes, limits).expect_err("must refuse");

        let message = error.to_string();
        assert!(message.contains("2 MB"), "limit must be named: {message}");
    }

    #[test]
    fn undecodable_bytes_are_counted_rather_than_failing_the_import() {
        let bytes = b"0 HEAD\n1 CHAR ANSEL\n1 NOTE \xD0\n";

        let (text, report) = decode_gedcom(bytes).expect("decode");

        assert_eq!(report.undecodable_bytes, 1);
        assert!(text.contains('\u{FFFD}'));
        assert!(!report.warnings.is_empty());
    }
}