#![warn(missing_docs)]
pub mod convert;
pub mod dates;
pub mod decode;
pub mod family;
#[cfg(feature = "gedzip")]
pub mod gedzip;
pub mod names;
pub mod schema;
#[cfg(feature = "fixture")]
pub mod synthetic;
pub mod tags;
pub mod validate;
pub mod view;
pub use decode::{EncodingReport, GedcomEncoding, decode_gedcom};
use std::collections::{HashMap, HashSet};
use std::fmt;
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
#[non_exhaustive]
pub struct Limits {
pub input_bytes: usize,
pub line_bytes: usize,
pub depth: u16,
pub records: usize,
pub structures: usize,
}
impl Limits {
pub const DEFAULT: Self = Self {
input_bytes: 128 * 1024 * 1024,
line_bytes: 64 * 1024,
depth: 64,
records: 2_000_000,
structures: 20_000_000,
};
#[must_use]
pub const fn with_input_bytes(mut self, input_bytes: usize) -> Self {
self.input_bytes = input_bytes;
self
}
#[must_use]
pub const fn with_line_bytes(mut self, line_bytes: usize) -> Self {
self.line_bytes = line_bytes;
self
}
#[must_use]
pub const fn with_depth(mut self, depth: u16) -> Self {
self.depth = depth;
self
}
#[must_use]
pub const fn with_records(mut self, records: usize) -> Self {
self.records = records;
self
}
#[must_use]
pub const fn with_structures(mut self, structures: usize) -> Self {
self.structures = structures;
self
}
}
impl Default for Limits {
fn default() -> Self {
Self::DEFAULT
}
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
#[non_exhaustive]
pub enum GedcomErrorKind {
Limit,
Syntax,
Structure,
Archive,
}
#[derive(Clone, Debug, Eq, PartialEq)]
pub struct GedcomError {
line: usize,
kind: GedcomErrorKind,
message: String,
}
impl GedcomError {
#[must_use]
pub const fn line(&self) -> usize {
self.line
}
#[must_use]
pub const fn kind(&self) -> GedcomErrorKind {
self.kind
}
#[must_use]
pub fn message(&self) -> &str {
&self.message
}
}
impl fmt::Display for GedcomError {
fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
if self.line == 0 {
formatter.write_str(&self.message)
} else {
write!(formatter, "GEDCOM line {}: {}", self.line, self.message)
}
}
}
impl std::error::Error for GedcomError {}
fn error(line: usize, message: impl Into<String>) -> GedcomError {
fault(line, GedcomErrorKind::Syntax, message)
}
fn fault(line: usize, kind: GedcomErrorKind, message: impl Into<String>) -> GedcomError {
GedcomError {
line,
kind,
message: message.into(),
}
}
#[derive(Clone, Debug, Eq, PartialEq)]
pub struct Node {
xref: Option<String>,
tag: String,
value: Option<String>,
pub children: Vec<Self>,
verbatim: Option<Box<str>>,
}
impl Node {
#[must_use]
pub fn new(tag: impl Into<String>) -> Self {
let tag = tag.into();
assert!(
is_tag(&tag),
"{tag:?} is not a GEDCOM tag: ASCII letters, digits, or underscore"
);
Self {
xref: None,
tag,
value: None,
children: Vec::new(),
verbatim: None,
}
}
#[must_use]
pub fn with_value(tag: impl Into<String>, value: impl Into<String>) -> Self {
let value = value.into();
if value.contains('\r') || value.contains('\n') {
let mut node = Self::new(tag);
node.set_logical_value(&value);
return node;
}
Self {
value: Some(value),
..Self::new(tag)
}
}
#[must_use]
pub fn record(xref: impl Into<String>, tag: impl Into<String>) -> Self {
let xref = xref.into();
assert!(
is_xref(&xref),
"{xref:?} is not a cross-reference identifier: the @I1@ shape"
);
Self {
xref: Some(xref),
..Self::new(tag)
}
}
#[must_use]
pub fn xref(&self) -> Option<&str> {
self.xref.as_deref()
}
#[must_use]
pub fn tag(&self) -> &str {
&self.tag
}
#[must_use]
pub fn value(&self) -> Option<&str> {
self.value.as_deref()
}
#[must_use]
pub fn child(mut self, child: Self) -> Self {
self.push(child);
self
}
pub fn push(&mut self, child: Self) {
self.children.push(child);
}
pub fn forget_verbatim(&mut self) {
self.verbatim = None;
}
#[must_use]
pub fn verbatim(&self) -> Option<&str> {
self.verbatim.as_deref()
}
pub fn set_value(&mut self, value: Option<String>) {
assert!(
value
.as_deref()
.is_none_or(|value| !value.contains('\r') && !value.contains('\n')),
"a single-line node payload cannot contain a line ending"
);
self.verbatim = None;
self.value = value;
}
pub fn set_tag(&mut self, tag: impl Into<String>) {
let tag = tag.into();
assert!(is_tag(&tag), "{tag:?} is not a GEDCOM tag");
self.verbatim = None;
self.tag = tag;
}
pub fn set_xref(&mut self, xref: Option<String>) {
assert!(
xref.as_deref().is_none_or(is_xref),
"the cross-reference identifier is not in @XREF@ form"
);
self.verbatim = None;
self.xref = xref;
}
#[must_use]
pub fn logical_value(&self) -> String {
let mut text = self.value.clone().unwrap_or_default();
for child in &self.children {
match child.tag.as_str() {
"CONT" => {
text.push('\n');
text.push_str(child.value.as_deref().unwrap_or_default());
}
"CONC" => text.push_str(child.value.as_deref().unwrap_or_default()),
_ => (),
}
}
text
}
pub fn set_logical_value(&mut self, value: &str) {
self.verbatim = None;
self.children
.retain(|child| !matches!(child.tag.as_str(), "CONT" | "CONC"));
let normalized;
let value = if value.contains('\r') {
normalized = value.replace("\r\n", "\n").replace('\r', "\n");
normalized.as_str()
} else {
value
};
let mut lines = value.split('\n');
self.set_value(
lines
.next()
.map(str::to_owned)
.filter(|line| !line.is_empty()),
);
for (index, line) in lines.map(|line| Self::with_value("CONT", line)).enumerate() {
self.children.insert(index, line);
}
}
#[must_use]
pub fn pointer(&self) -> Option<&str> {
self.value.as_deref().filter(|value| is_xref(value))
}
#[must_use]
pub fn is_void_pointer(&self) -> bool {
self.pointer() == Some("@VOID@")
}
#[must_use]
pub fn first(&self, tag: &str) -> Option<&Self> {
self.children.iter().find(|child| child.tag == tag)
}
pub fn first_mut(&mut self, tag: &str) -> Option<&mut Self> {
self.children.iter_mut().find(|child| child.tag == tag)
}
pub fn all<'a>(&'a self, tag: &'a str) -> impl Iterator<Item = &'a Self> {
self.children.iter().filter(move |child| child.tag == tag)
}
#[must_use]
pub fn value_of(&self, tag: &str) -> Option<String> {
self.first(tag).map(Self::logical_value)
}
pub fn substructures(&self) -> impl Iterator<Item = &Self> {
self.children
.iter()
.filter(|child| !matches!(child.tag.as_str(), "CONT" | "CONC"))
}
#[must_use]
pub fn at(&self, path: &[usize]) -> Option<&Self> {
let mut node = self;
for index in path {
node = node.children.get(*index)?;
}
Some(node)
}
pub fn at_mut(&mut self, path: &[usize]) -> Option<&mut Self> {
let mut node = self;
for index in path {
node = node.children.get_mut(*index)?;
}
Some(node)
}
pub fn remove_at(&mut self, path: &[usize]) -> Option<Self> {
let (last, parents) = path.split_last()?;
let parent = self.at_mut(parents)?;
(*last < parent.children.len()).then(|| parent.children.remove(*last))
}
pub fn walk(&self) -> impl Iterator<Item = &Self> {
let mut stack = vec![self];
std::iter::from_fn(move || {
let node = stack.pop()?;
stack.extend(node.children.iter().rev());
Some(node)
})
}
#[must_use]
pub fn line_count(&self) -> usize {
1 + self.children.iter().map(Self::line_count).sum::<usize>()
}
fn render_into(&self, level: u16, output: &mut String) {
if let Some(verbatim) = &self.verbatim {
output.push_str(verbatim);
} else {
render_line(
level,
self.xref.as_deref(),
&self.tag,
self.value.as_deref(),
output,
);
}
output.push('\n');
for child in &self.children {
child.render_into(level + 1, output);
}
}
}
fn render_line(
level: u16,
xref: Option<&str>,
tag: &str,
value: Option<&str>,
output: &mut String,
) {
let mut buffer = itoa(level);
output.push_str(buffer.as_str());
buffer.clear();
output.push(' ');
if let Some(xref) = xref {
output.push_str(xref);
output.push(' ');
}
output.push_str(tag);
if let Some(value) = value {
output.push(' ');
output.push_str(value);
}
}
fn itoa(level: u16) -> String {
level.to_string()
}
#[derive(Clone, Debug, PartialEq, Eq)]
#[non_exhaustive]
#[cfg_attr(feature = "serde", derive(serde::Serialize))]
#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))]
pub struct UnresolvedPointer {
pub record: String,
pub path: Vec<usize>,
pub tag: String,
pub pointer: String,
}
#[derive(Clone, Debug, Default, Eq, PartialEq)]
pub struct Document {
pub records: Vec<Node>,
}
impl Document {
pub fn from_bytes(bytes: &[u8]) -> Result<(Self, EncodingReport), GedcomError> {
Self::from_bytes_with(bytes, Limits::DEFAULT)
}
pub fn from_bytes_with(
bytes: &[u8],
limits: Limits,
) -> Result<(Self, EncodingReport), GedcomError> {
let (text, report) = decode::decode_gedcom_with(bytes, limits)?;
let document = Self::parse_with(&text, limits)?;
Ok((document, report))
}
pub fn parse(input: &str) -> Result<Self, GedcomError> {
Self::parse_with(input, Limits::DEFAULT)
}
pub fn parse_with(input: &str, limits: Limits) -> Result<Self, GedcomError> {
if input.len() > limits.input_bytes {
return Err(fault(
0,
GedcomErrorKind::Limit,
format!(
"the file is {} MB and the limit is {} MB",
input.len() / (1024 * 1024),
limits.input_bytes / (1024 * 1024)
),
));
}
let normalized;
let input = if input.contains('\r') {
normalized = input.replace("\r\n", "\n").replace('\r', "\n");
normalized.as_str()
} else {
input
};
let mut parsed = Vec::new();
for (index, line) in input.lines().enumerate() {
if line.is_empty() {
continue;
}
if parsed.len() >= limits.structures {
return Err(fault(
index + 1,
GedcomErrorKind::Limit,
"the file has too many lines",
));
}
parsed.push(parse_line(index + 1, line, limits)?);
}
let mut position = 0;
let mut records = Vec::new();
while position < parsed.len() {
if parsed[position].level != 0 {
return Err(fault(
parsed[position].line,
GedcomErrorKind::Structure,
"a record must begin at level 0",
));
}
if records.len() >= limits.records {
return Err(fault(
parsed[position].line,
GedcomErrorKind::Limit,
"the file has too many records",
));
}
records.push(build_node(&parsed, &mut position, 0)?);
}
Ok(Self { records })
}
#[must_use]
pub fn to_text(&self) -> String {
let mut output = String::new();
for record in &self.records {
record.render_into(0, &mut output);
}
output
}
#[must_use]
pub fn to_bytes(&self) -> Vec<u8> {
let needs_fix = self.header().is_some_and(|header| {
header
.first("CHAR")
.is_some_and(|declared| !declares_utf8(declared))
});
if !needs_fix {
return self.to_text().into_bytes();
}
let mut corrected = self.clone();
corrected.declare_utf8_encoding();
corrected.to_text().into_bytes()
}
pub fn declare_utf8_encoding(&mut self) -> bool {
let Some(header) = self.records.iter_mut().find(|record| record.tag == "HEAD") else {
return false;
};
let Some(declared) = header.first_mut("CHAR") else {
return false;
};
if declares_utf8(declared) {
return false;
}
declared.set_value(Some("UTF-8".to_owned()));
true
}
#[must_use]
pub fn xref_index(&self) -> HashMap<String, usize> {
let mut index = HashMap::with_capacity(self.records.len());
for (position, record) in self.records.iter().enumerate() {
if let Some(xref) = &record.xref {
index.entry(xref.to_ascii_uppercase()).or_insert(position);
}
}
index
}
#[must_use]
pub fn record(&self, xref: &str) -> Option<&Node> {
let wanted = xref.to_ascii_uppercase();
self.records.iter().find(|record| {
record
.xref
.as_ref()
.is_some_and(|own| own.to_ascii_uppercase() == wanted)
})
}
pub fn record_mut(&mut self, xref: &str) -> Option<&mut Node> {
let wanted = xref.to_ascii_uppercase();
self.records.iter_mut().find(|record| {
record
.xref
.as_ref()
.is_some_and(|own| own.to_ascii_uppercase() == wanted)
})
}
pub fn records_of<'a>(&'a self, tag: &'a str) -> impl Iterator<Item = &'a Node> {
self.records.iter().filter(move |record| record.tag == tag)
}
#[must_use]
pub fn header(&self) -> Option<&Node> {
self.records.first().filter(|record| record.tag == "HEAD")
}
#[must_use]
pub fn version(&self) -> Option<String> {
self.header()?.first("GEDC")?.value_of("VERS")
}
#[must_use]
pub fn is_version_7_or_later(&self) -> bool {
self.version()
.and_then(|version| version.split('.').next()?.trim().parse::<u32>().ok())
.is_some_and(|major| major >= 7)
}
#[must_use]
pub fn next_xref(&self, prefix: &str) -> String {
let mut highest = 0u64;
for record in &self.records {
let Some(xref) = &record.xref else { continue };
let inner = xref.trim_matches('@');
let Some(digits) = inner.strip_prefix(prefix) else {
continue;
};
if let Ok(number) = digits.parse::<u64>() {
highest = highest.max(number);
}
}
format!("@{prefix}{}@", highest + 1)
}
#[deprecated(
since = "0.1.0",
note = "use `unresolved_pointers`, which addresses each fault precisely enough to repair"
)]
#[must_use]
pub fn dangling_pointers(&self) -> Vec<(String, String)> {
self.unresolved_pointers()
.into_iter()
.map(|found| (found.record, found.pointer))
.collect()
}
#[must_use]
pub fn unresolved_pointers(&self) -> Vec<UnresolvedPointer> {
let known: HashSet<String> = self
.records
.iter()
.filter_map(|record| record.xref.as_ref().map(|xref| xref.to_ascii_uppercase()))
.collect();
let mut problems = Vec::new();
for record in &self.records {
let owner = record.xref.clone().unwrap_or_else(|| record.tag.clone());
let mut stack: Vec<(Vec<usize>, &Node)> = vec![(Vec::new(), record)];
while let Some((path, node)) = stack.pop() {
if let Some(pointer) = node.pointer()
&& pointer != "@VOID@"
&& !known.contains(&pointer.to_ascii_uppercase())
{
problems.push(UnresolvedPointer {
record: owner.clone(),
path: path.clone(),
tag: node.tag.clone(),
pointer: pointer.to_owned(),
});
}
for (index, child) in node.children.iter().enumerate().rev() {
let mut child_path = path.clone();
child_path.push(index);
stack.push((child_path, child));
}
}
}
problems
}
#[must_use]
pub fn line_count(&self) -> usize {
self.records.iter().map(Node::line_count).sum()
}
#[must_use]
pub fn content_records(&self) -> usize {
self.records
.iter()
.filter(|record| !matches!(record.tag.as_str(), "HEAD" | "TRLR"))
.count()
}
#[must_use]
pub fn new_v7() -> Self {
Self {
records: vec![
Node::new("HEAD").child(Node::new("GEDC").child(Node::with_value("VERS", "7.0"))),
Node::new("TRLR"),
],
}
}
#[must_use]
pub fn new_v551(source_system: &str) -> Self {
Self {
records: vec![
Node::new("HEAD")
.child(Node::with_value("SOUR", source_system))
.child(
Node::new("GEDC")
.child(Node::with_value("VERS", "5.5.1"))
.child(Node::with_value("FORM", "LINEAGE-LINKED")),
)
.child(Node::with_value("CHAR", "UTF-8")),
Node::new("TRLR"),
],
}
}
}
fn declares_utf8(declared: &Node) -> bool {
declared
.value
.as_deref()
.is_some_and(|value| value.trim().eq_ignore_ascii_case("UTF-8"))
}
#[derive(Clone, Debug)]
struct ParsedLine {
line: usize,
level: u16,
xref: Option<String>,
tag: String,
value: Option<String>,
verbatim: Option<Box<str>>,
}
fn parse_line(line_number: usize, line: &str, limits: Limits) -> Result<ParsedLine, GedcomError> {
if line.len() > limits.line_bytes {
return Err(fault(
line_number,
GedcomErrorKind::Limit,
"the line is longer than the limit",
));
}
let (level_text, remainder) = line
.split_once(' ')
.ok_or_else(|| error(line_number, "the line has no tag"))?;
let level = level_text
.parse::<u16>()
.ok()
.ok_or_else(|| error(line_number, "the line does not begin with a level"))?;
if level > limits.depth {
return Err(fault(
line_number,
GedcomErrorKind::Limit,
"the line is nested deeper than the limit",
));
}
let remainder = remainder.trim_start_matches(' ');
let (first, remainder) = remainder
.split_once(' ')
.map_or((remainder, None), |(first, rest)| (first, Some(rest)));
if first.is_empty() {
return Err(error(line_number, "the line has no tag"));
}
let (xref, tag, value) = if is_xref(first) {
let rest = remainder
.filter(|value| !value.is_empty())
.ok_or_else(|| error(line_number, "the record has an identifier but no tag"))?;
let (tag, value) = rest
.split_once(' ')
.map_or((rest, None), |(tag, value)| (tag, Some(value)));
(Some(first.to_owned()), tag, value)
} else if first.starts_with('@') {
return Err(error(
line_number,
"the cross-reference identifier is not closed, or contains a space",
));
} else {
(None, first, remainder)
};
if !is_tag(tag) {
return Err(error(
line_number,
"a tag must be ASCII letters, digits, or underscore",
));
}
let mut canonical = String::with_capacity(line.len());
render_line(level, xref.as_deref(), tag, value, &mut canonical);
let verbatim = (canonical != line).then(|| Box::from(line));
Ok(ParsedLine {
line: line_number,
level,
xref,
tag: tag.to_owned(),
value: value.map(str::to_owned),
verbatim,
})
}
fn build_node(
lines: &[ParsedLine],
position: &mut usize,
expected_level: u16,
) -> Result<Node, GedcomError> {
let current = &lines[*position];
if current.level != expected_level {
return Err(fault(
current.line,
GedcomErrorKind::Structure,
"the line is nested at the wrong level",
));
}
let mut node = Node {
xref: current.xref.clone(),
tag: current.tag.clone(),
value: current.value.clone(),
children: Vec::new(),
verbatim: current.verbatim.clone(),
};
*position += 1;
while let Some(next) = lines.get(*position) {
if next.level <= expected_level {
break;
}
if next.level != expected_level + 1 {
return Err(fault(
next.line,
GedcomErrorKind::Structure,
"the line skips a nesting level",
));
}
node.children
.push(build_node(lines, position, expected_level + 1)?);
}
Ok(node)
}
fn is_xref(value: &str) -> bool {
value.len() >= 3
&& value.starts_with('@')
&& value.ends_with('@')
&& !value[1..value.len() - 1].contains('@')
}
fn is_tag(value: &str) -> bool {
!value.is_empty()
&& value
.bytes()
.all(|byte| byte.is_ascii_alphanumeric() || byte == b'_')
}
#[cfg(test)]
mod tests {
use super::*;
const SAMPLE: &str = "0 HEAD\n1 GEDC\n2 VERS 7.0\n1 CHAR UTF-8\n0 @I1@ INDI\n1 NAME Ada /Example/\n1 _VENDOR retained verbatim\n2 DATA extension child\n0 TRLR\n";
#[test]
fn unknown_extension_tags_and_values_round_trip() {
let document = Document::parse(SAMPLE).expect("parse sample");
assert_eq!(document.records.len(), 3);
assert_eq!(document.records[1].xref.as_deref(), Some("@I1@"));
assert_eq!(document.records[1].children[1].tag, "_VENDOR");
assert_eq!(document.to_text(), SAMPLE);
}
#[test]
fn an_unusual_line_is_written_back_exactly_as_it_arrived() {
let input = "0 HEAD\n0 _PUBLISH\n0 TRLR\n";
let document = Document::parse(input).expect("parse doubled delimiter");
assert_eq!(document.records[1].tag, "_PUBLISH");
assert_eq!(document.to_text(), input);
}
#[test]
fn editing_a_node_gives_up_its_verbatim_line() {
let mut document = Document::parse("0 HEAD\n0 _PUBLISH\n0 TRLR\n").expect("parse");
document.records[1].set_value(Some("now".to_owned()));
assert_eq!(document.to_text(), "0 HEAD\n0 _PUBLISH now\n0 TRLR\n");
}
#[test]
fn continuation_lines_fold_for_reading_and_stay_put_for_writing() {
let input = "0 HEAD\n0 @N1@ NOTE first\n1 CONT second\n1 CONC and more\n0 TRLR\n";
let document = Document::parse(input).expect("parse continuations");
let note = document.record("@N1@").expect("note record");
assert_eq!(note.logical_value(), "first\nsecond and more");
assert_eq!(note.substructures().count(), 0);
assert_eq!(document.to_text(), input);
}
#[test]
fn parser_rejects_skipped_nesting_and_invalid_tags() {
assert!(Document::parse("0 HEAD\n2 CHAR UTF-8\n").is_err());
assert!(Document::parse("0 bad-tag\n").is_err());
}
#[test]
fn a_space_inside_a_cross_reference_identifier_is_named_as_the_fault() {
let failure = Document::parse("0 @NoTe ref@ NOTE text\n").expect_err("must fail");
assert!(
failure.message().contains("cross-reference"),
"message blamed the wrong thing: {failure}"
);
}
#[test]
fn parser_accepts_every_line_ending_gedcom_allows() {
for input in [
"0 HEAD\n0 TRLR\n",
"0 HEAD\r\n0 TRLR\r\n",
"0 HEAD\r0 TRLR\r",
] {
let document = Document::parse(input).expect("parse line endings");
assert_eq!(document.records.len(), 2);
}
}
#[test]
fn pointers_resolve_case_insensitively_and_void_is_not_dangling() {
let document = Document::parse(
"0 HEAD\n0 @i1@ INDI\n1 FAMS @F1@\n1 ASSO @VOID@\n1 FAMC @F9@\n0 @f1@ FAM\n0 TRLR\n",
)
.expect("parse pointers");
assert!(document.record("@I1@").is_some());
let unresolved = document.unresolved_pointers();
assert_eq!(unresolved.len(), 1);
assert_eq!(unresolved[0].record, "@i1@");
assert_eq!(unresolved[0].pointer, "@F9@");
#[allow(
deprecated,
reason = "the deprecated form stays equivalent until removed"
)]
{
assert_eq!(
document.dangling_pointers(),
vec![("@i1@".to_owned(), "@F9@".to_owned())]
);
}
}
#[test]
fn next_xref_continues_the_files_own_numbering() {
let document = Document::parse("0 HEAD\n0 @I1@ INDI\n0 @I7@ INDI\n0 @F1@ FAM\n0 TRLR\n")
.expect("parse");
assert_eq!(document.next_xref("I"), "@I8@");
assert_eq!(document.next_xref("F"), "@F2@");
assert_eq!(document.next_xref("S"), "@S1@");
}
#[test]
fn version_is_read_from_the_header() {
let five = Document::parse("0 HEAD\n1 GEDC\n2 VERS 5.5.1\n0 TRLR\n").expect("parse 5.5.1");
let seven = Document::parse("0 HEAD\n1 GEDC\n2 VERS 7.0\n0 TRLR\n").expect("parse 7.0");
assert_eq!(five.version().as_deref(), Some("5.5.1"));
assert!(!five.is_version_7_or_later());
assert!(seven.is_version_7_or_later());
}
#[test]
fn a_structure_is_addressed_by_the_path_the_record_view_shows() {
let mut document = Document::parse(
"0 HEAD\n0 @I1@ INDI\n1 NAME Ada /Example/\n1 BIRT\n2 DATE 1815\n2 PLAC London\n0 TRLR\n",
)
.expect("parse");
let record = document.record_mut("@I1@").expect("individual");
assert_eq!(record.at(&[1, 0]).expect("date").logical_value(), "1815");
record
.at_mut(&[1, 0])
.expect("date")
.set_value(Some("10 December 1815".to_owned()));
assert_eq!(record.remove_at(&[1, 1]).expect("place").tag, "PLAC");
assert_eq!(
document.to_text(),
"0 HEAD\n0 @I1@ INDI\n1 NAME Ada /Example/\n1 BIRT\n2 DATE 10 December 1815\n0 TRLR\n"
);
}
#[test]
fn set_logical_value_is_the_inverse_of_logical_value() {
let mut node = Node::with_value("NOTE", "first");
node.push(Node::with_value("CONC", " and more"));
node.push(Node::with_value("SOUR", "@S1@"));
node.set_logical_value("one\ntwo\nthree");
assert_eq!(node.logical_value(), "one\ntwo\nthree");
let tags: Vec<&str> = node
.children
.iter()
.map(|child| child.tag.as_str())
.collect();
assert_eq!(tags, ["CONT", "CONT", "SOUR"]);
node.set_logical_value("");
assert_eq!(node.value, None);
assert_eq!(node.logical_value(), "");
}
#[test]
fn content_records_exclude_the_header_and_trailer() {
let document = Document::parse(SAMPLE).expect("parse sample");
assert_eq!(document.records.len(), 3);
assert_eq!(document.content_records(), 1);
}
#[test]
fn limits_refuse_a_line_longer_than_the_bound() {
let limits = Limits::DEFAULT.with_line_bytes(32);
let long = format!("0 HEAD\n1 NOTE {}\n0 TRLR\n", "x".repeat(64));
let failure = Document::parse_with(&long, limits).expect_err("must refuse");
assert_eq!(failure.kind(), GedcomErrorKind::Limit);
assert_eq!(failure.line(), 2, "the long line is named");
}
#[test]
fn limits_refuse_nesting_deeper_than_the_bound() {
let limits = Limits::DEFAULT.with_depth(3);
let deep = "0 HEAD\n1 A\n2 B\n3 C\n4 D\n0 TRLR\n";
let failure = Document::parse_with(deep, limits).expect_err("must refuse");
assert_eq!(failure.kind(), GedcomErrorKind::Limit);
}
#[test]
fn limits_refuse_a_flat_file_with_too_many_structures() {
let limits = Limits::DEFAULT.with_structures(4);
let flat = "0 HEAD\n1 NOTE a\n1 NOTE b\n1 NOTE c\n1 NOTE d\n0 TRLR\n";
let failure = Document::parse_with(flat, limits).expect_err("must refuse");
assert_eq!(failure.kind(), GedcomErrorKind::Limit);
}
#[test]
fn error_kinds_tell_a_bound_from_a_malformed_line() {
assert_eq!(
Document::parse("0 bad-tag\n").expect_err("syntax").kind(),
GedcomErrorKind::Syntax
);
assert_eq!(
Document::parse("0 HEAD\n2 CHAR UTF-8\n")
.expect_err("structure")
.kind(),
GedcomErrorKind::Structure
);
}
#[test]
fn a_builder_payload_with_newlines_becomes_continuation_lines() {
let node = Node::with_value("NOTE", "first\nsecond");
let text = Document {
records: vec![node],
}
.to_text();
assert_eq!(text, "0 NOTE first\n1 CONT second\n");
Document::parse(&text).expect("the built document re-parses");
}
#[test]
fn a_builder_normalizes_every_line_ending_into_continuations() {
let node = Node::with_value("NOTE", "first\r\nsecond\rthird\nfourth");
assert_eq!(node.logical_value(), "first\nsecond\nthird\nfourth");
assert!(
!Document {
records: vec![node.clone()]
}
.to_text()
.contains('\r')
);
assert_eq!(node.all("CONT").count(), 3);
}
#[test]
fn programmatic_fields_that_could_inject_a_line_are_refused() {
assert!(std::panic::catch_unwind(|| Node::new("NAME\n0 TRLR")).is_err());
assert!(std::panic::catch_unwind(|| Node::record("@I1@\n0 TRLR", "INDI")).is_err());
assert!(
std::panic::catch_unwind(|| {
let mut node = Node::new("NOTE");
node.set_value(Some("text\n0 TRLR".to_owned()));
})
.is_err()
);
assert!(
std::panic::catch_unwind(|| {
let mut node = Node::new("NOTE");
node.set_tag("NOTE\n0 TRLR");
})
.is_err()
);
assert!(
std::panic::catch_unwind(|| {
let mut node = Node::new("NOTE");
node.set_xref(Some("@I1@\n0 TRLR".to_owned()));
})
.is_err()
);
}
#[test]
fn an_edit_beside_an_odd_line_leaves_the_odd_line_alone() {
let input = "0 HEAD\n0 _PUBLISH\n1 DATE 1900\n0 TRLR\n";
let mut document = Document::parse(input).expect("parse");
document.records[1]
.first_mut("DATE")
.expect("date")
.set_value(Some("1901".to_owned()));
assert_eq!(
document.to_text(),
"0 HEAD\n0 _PUBLISH\n1 DATE 1901\n0 TRLR\n"
);
}
#[test]
fn to_bytes_corrects_a_char_declaration_the_bytes_would_contradict() {
let mut document =
Document::parse("0 HEAD\n1 CHAR ANSEL\n0 @N1@ NOTE K\u{f6}nig\n0 TRLR\n")
.expect("parse");
let bytes = document.to_bytes();
let text = String::from_utf8(bytes).expect("UTF-8 out");
assert!(text.contains("1 CHAR UTF-8"), "{text}");
assert!(text.contains("K\u{f6}nig"), "{text}");
assert!(document.to_text().contains("1 CHAR ANSEL"));
assert!(document.declare_utf8_encoding());
assert!(document.to_text().contains("1 CHAR UTF-8"));
assert!(!document.declare_utf8_encoding(), "already true");
}
#[test]
fn limits_refuse_a_file_with_too_many_records() {
let limits = Limits {
records: 2,
..Limits::DEFAULT
};
let failure =
Document::parse_with("0 HEAD\n0 @I1@ INDI\n0 TRLR\n", limits).expect_err("must refuse");
assert!(failure.to_string().contains("too many records"));
}
}