gedcomkit 0.1.11

A byte-preserving GEDCOM document model: decoding, parsing, readings, version conversion, plausibility checks, and the GEDZIP container, for GEDCOM 5.5 through 7.x.
Documentation
//! The vendored gedcom7code test corpus, run unconditionally.
//!
//! `fixtures/vendored/gedcom7code/` is 51 public-domain files from
//! [gedcom7code/test-files], nearly every scenario twice — once as 5.5.1 and
//! once as GEDCOM 7.0 — covering encodings, `@` escaping, the whole date
//! grammar, obsolete tags, and cross-reference case. Unlike the fetched
//! external corpus and the private requirement corpus, these are committed,
//! so this test is the spec-adherence gate that can never be skipped.
//!
//! Two invariants, applied to every file:
//!
//! 1. **Nothing is mangled in silence.** A file either decodes and parses, or
//!    is refused with a bounded error that names a line. (The `-invalid`
//!    files are invalid at the payload grammar, which this reader
//!    deliberately preserves rather than rejects, so most of them parse.)
//! 2. **What parses writes back byte for byte** — against the decoded text,
//!    because re-encoding to UTF-16 or writing a byte-order mark is the
//!    caller's business, not the document model's.
//!
//! [gedcom7code/test-files]: https://github.com/gedcom7code/test-files

use gedcomkit::{Document, decode_gedcom};
use std::path::{Path, PathBuf};

fn corpus_root() -> PathBuf {
    Path::new(env!("CARGO_MANIFEST_DIR")).join("fixtures/vendored/gedcom7code")
}

fn corpus_files(version: &str) -> Vec<PathBuf> {
    let directory = corpus_root().join(version);
    let mut files: Vec<PathBuf> = std::fs::read_dir(&directory)
        .unwrap_or_else(|error| panic!("{}: {error}", directory.display()))
        .map(|entry| entry.expect("read the directory").path())
        .filter(|path| path.extension().is_some_and(|extension| extension == "ged"))
        .collect();
    files.sort();
    files
}

#[test]
fn every_vendored_file_decodes_parses_and_writes_back_byte_for_byte() {
    let mut round_tripped = 0usize;
    let mut refused = Vec::new();

    for version in ["5", "7"] {
        for path in corpus_files(version) {
            let name = format!("{version}/{}", path.file_name().unwrap().to_string_lossy());
            let bytes = std::fs::read(&path).expect("read the fixture");
            let text = match decode_gedcom(&bytes) {
                Ok((text, _)) => text,
                Err(error) => {
                    refused.push(format!("{name}: decode: {error}"));
                    continue;
                }
            };
            match Document::parse(&text) {
                Ok(document) => {
                    assert_eq!(
                        document.to_text(),
                        text,
                        "{name} did not write back byte for byte"
                    );
                    round_tripped += 1;
                }
                Err(error) => {
                    assert!(
                        error.to_string().len() < 500,
                        "{name}: the refusal is not bounded: {error}"
                    );
                    refused.push(format!("{name}: parse: {error}"));
                }
            }
        }
    }

    eprintln!("{round_tripped} files round-tripped; refused: {refused:#?}");
    // The corpus holds 51 files. Payload-level invalidity is preserved rather
    // than rejected, so nearly everything must round-trip; a lower number
    // means either the fixtures moved or the reader broke.
    assert!(
        round_tripped >= 49,
        "only {round_tripped} of the vendored corpus round-tripped; refused: {refused:#?}"
    );
}

/// The encoding cases: the same document as ASCII, UTF-8, and UTF-16 in both
/// endiannesses, with and without a byte-order mark. Whatever the bytes, the
/// decoded document must say the same thing.
#[test]
fn the_character_encoding_variants_all_read_to_the_same_records() {
    for version in ["5", "7"] {
        let mut shapes: Vec<(String, usize)> = Vec::new();
        for path in corpus_files(version) {
            let name = path.file_name().unwrap().to_string_lossy().to_string();
            if !name.starts_with("char_") {
                continue;
            }
            let bytes = std::fs::read(&path).expect("read the fixture");
            let (text, report) =
                decode_gedcom(&bytes).unwrap_or_else(|error| panic!("{version}/{name}: {error}"));
            let document =
                Document::parse(&text).unwrap_or_else(|error| panic!("{version}/{name}: {error}"));
            eprintln!("{version}/{name}: {}", report.summary());
            shapes.push((name, document.records.len()));
        }
        assert!(
            shapes.len() >= 7,
            "expected the encoding variants, found {shapes:?}"
        );
        // Every variant of the same numbered document has the same records.
        let ascii_1 = shapes
            .iter()
            .find(|(name, _)| name == "char_ascii_1.ged")
            .map(|(_, records)| *records)
            .expect("the ASCII baseline");
        for (name, records) in &shapes {
            if name.ends_with("-1.ged") || name.ends_with("_1.ged") {
                assert_eq!(
                    *records, ascii_1,
                    "{version}/{name} decoded to a different number of records"
                );
            }
        }
    }
}

/// Every 5.5.1 file with a 7.0 twin converts up to something that reads back,
/// keeps its record count, and declares no character set — the properties the
/// upstream corpus was built to exercise. Upstream's own README warns that
/// equivalent files can differ textually, so this asserts invariants rather
/// than comparing bytes with the twin.
#[test]
fn every_paired_five_five_one_file_converts_up_to_version_seven() {
    let mut converted = 0usize;
    for path in corpus_files("5") {
        let name = path.file_name().unwrap().to_string_lossy().to_string();
        if !corpus_root().join("7").join(&name).is_file() {
            continue;
        }
        let bytes = std::fs::read(&path).expect("read the fixture");
        let Ok((text, _)) = decode_gedcom(&bytes) else {
            continue;
        };
        let Ok(document) = Document::parse(&text) else {
            continue;
        };

        let outcome = gedcomkit::convert::to_version_7(&document);
        let converted_text = outcome.document.to_text();
        assert!(
            converted_text.contains("2 VERS 7.0"),
            "5/{name} does not declare 7.0"
        );
        assert!(
            !converted_text.contains("\n1 CHAR "),
            "5/{name} kept a character set declaration"
        );
        let reread = Document::parse(&converted_text)
            .unwrap_or_else(|error| panic!("5/{name} did not read back after conversion: {error}"));
        assert_eq!(
            reread.records.len(),
            document.records.len(),
            "5/{name} lost a record in conversion"
        );
        converted += 1;
    }
    eprintln!("{converted} paired files converted up");
    assert!(
        converted >= 20,
        "only {converted} paired files converted; the corpus should pair nearly everything"
    );
}

/// Archives written by an independent ZIP implementation (Python `zipfile`;
/// see `fixtures/vendored/zip-interop/README.md`). The suite above proves the
/// reader against this crate's own writer, which is a closed loop: real
/// producers write directory entries, timestamps, and — forced on here —
/// Zip64 structures this crate's writer never emits.
#[cfg(feature = "gedzip")]
#[test]
fn archives_from_an_independent_writer_read_back_whole() {
    let root = Path::new(env!("CARGO_MANIFEST_DIR")).join("fixtures/vendored/zip-interop");
    let expected_document = b"0 HEAD
1 GEDC
2 VERS 7.0
0 @I1@ INDI
1 NAME Interop /Fixture/
0 TRLR
";
    let expected_media: Vec<u8> = (0..=255u8).collect::<Vec<u8>>().repeat(8);

    for name in ["python-zipfile.gdz", "python-zip64.gdz"] {
        let bytes =
            std::fs::read(root.join(name)).unwrap_or_else(|error| panic!("{name}: {error}"));
        let archive = gedcomkit::gedzip::Archive::open(&bytes)
            .unwrap_or_else(|error| panic!("{name} did not open: {error}"));
        assert_eq!(
            archive
                .document_name()
                .unwrap_or_else(|error| panic!("{name}: {error}")),
            "gedcom.ged"
        );
        let document = archive
            .read("gedcom.ged")
            .unwrap_or_else(|error| panic!("{name} document: {error}"));
        assert_eq!(document, expected_document, "{name}");
        let media = archive
            .read("media/photo.jpg")
            .unwrap_or_else(|error| panic!("{name} media: {error}"));
        assert_eq!(media, expected_media, "{name}");

        let (text, _) = gedcomkit::decode_gedcom(&document).expect("decode");
        let parsed = gedcomkit::Document::parse(&text).expect("parse");
        assert_eq!(parsed.to_text().as_bytes(), expected_document, "{name}");
    }
}