use crate::Node;
use std::fmt::{self, Write as _};
#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
#[non_exhaustive]
#[cfg_attr(feature = "serde", derive(serde::Serialize))]
pub enum Calendar {
#[default]
Gregorian,
Julian,
Hebrew,
FrenchRepublican,
Roman,
Unknown,
}
impl Calendar {
#[must_use]
pub const fn label(self) -> &'static str {
match self {
Self::Gregorian => "Gregorian",
Self::Julian => "Julian",
Self::Hebrew => "Hebrew",
Self::FrenchRepublican => "French Republican",
Self::Roman => "Roman",
Self::Unknown => "unrecognized calendar",
}
}
}
#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
#[non_exhaustive]
#[cfg_attr(feature = "serde", derive(serde::Serialize))]
pub enum DateKind {
#[default]
Exact,
About,
Calculated,
Estimated,
Before,
After,
Between,
Period,
Interpreted,
Unparsed,
}
impl DateKind {
#[must_use]
pub const fn label(self) -> &'static str {
match self {
Self::Exact => "",
Self::About => "about",
Self::Calculated => "calculated",
Self::Estimated => "estimated",
Self::Before => "before",
Self::After => "after",
Self::Between => "between",
Self::Period => "during",
Self::Interpreted => "interpreted as",
Self::Unparsed => "as written",
}
}
}
#[derive(Clone, Copy, Debug, Default, Eq, PartialEq, Ord, PartialOrd)]
#[non_exhaustive]
#[cfg_attr(feature = "serde", derive(serde::Serialize))]
pub enum Precision {
#[default]
None,
Year,
Month,
Day,
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
#[non_exhaustive]
#[cfg_attr(feature = "serde", derive(serde::Serialize))]
#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))]
pub struct DatePoint {
pub year: i32,
pub month: Option<u8>,
pub day: Option<u8>,
}
impl DatePoint {
const fn year(year: i32) -> Self {
Self {
year,
month: None,
day: None,
}
}
#[must_use]
pub fn from_ymd(year: i32, month: Option<u8>, day: Option<u8>) -> Option<Self> {
if let Some(month) = month
&& !(1..=12).contains(&month)
{
return None;
}
match day {
Some(_) if month.is_none() => return None,
Some(day) if !(1..=31).contains(&day) => return None,
_ => (),
}
Some(Self { year, month, day })
}
#[must_use]
pub fn to_gedcom(&self) -> String {
let (era_year, era) = if self.year < 0 {
(-self.year, " BCE")
} else {
(self.year, "")
};
match (self.month, self.day) {
(Some(month), Some(day)) => {
format!("{day} {} {era_year}{era}", month_abbreviation(month))
}
(Some(month), None) => format!("{} {era_year}{era}", month_abbreviation(month)),
_ => format!("{era_year}{era}"),
}
}
#[must_use]
pub const fn precision(&self) -> Precision {
match (self.month, self.day) {
(Some(_), Some(_)) => Precision::Day,
(Some(_), None) => Precision::Month,
_ => Precision::Year,
}
}
#[must_use]
pub fn sort_key(&self) -> i64 {
let month = self.month.map_or(0, i64::from);
let day = self.day.map_or(0, i64::from);
i64::from(self.year) * 10_000 + month * 100 + day
}
}
impl fmt::Display for DatePoint {
fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
match (self.month, self.day) {
(Some(month), Some(day)) => {
write!(formatter, "{day} {} {}", month_name(month), self.year)
}
(Some(month), None) => write!(formatter, "{} {}", month_name(month), self.year),
_ => write!(formatter, "{}", self.year),
}
}
}
#[derive(Clone, Debug, Eq, PartialEq)]
#[non_exhaustive]
#[cfg_attr(feature = "serde", derive(serde::Serialize))]
#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))]
pub struct GedcomDate {
pub original: String,
pub calendar: Calendar,
pub kind: DateKind,
pub earliest: Option<DatePoint>,
pub latest: Option<DatePoint>,
pub phrase: Option<String>,
}
impl GedcomDate {
#[must_use]
pub fn parse(payload: &str) -> Self {
let original = payload.to_owned();
let (calendar, rest) = split_calendar(payload.trim());
let (rest, phrase) = split_phrase(&rest);
let mut date = parse_expression(&rest).unwrap_or_else(|| Self {
original: original.clone(),
calendar,
kind: DateKind::Unparsed,
earliest: None,
latest: None,
phrase: Some(payload.trim().to_owned()),
});
date.original = original;
date.calendar = calendar;
if let Some(phrase) = phrase {
date.phrase = Some(phrase);
}
date
}
#[must_use]
pub const fn is_parsed(&self) -> bool {
!matches!(self.kind, DateKind::Unparsed)
}
#[must_use]
pub fn precision(&self) -> Precision {
self.earliest
.map_or(Precision::None, |point| point.precision())
}
#[must_use]
pub fn year_hint(&self) -> Option<i32> {
if let Some(point) = self.earliest {
return Some(point.year);
}
four_digit_year(&self.original)
}
#[must_use]
pub fn sort_key(&self) -> Option<i64> {
self.earliest
.map(|point| point.sort_key())
.or_else(|| self.year_hint().map(|year| i64::from(year) * 10_000))
}
#[must_use]
pub fn display(&self) -> String {
match (self.kind, self.earliest, self.latest) {
(DateKind::Unparsed, ..) => self.original.trim().to_owned(),
(DateKind::Exact, Some(point), _) => point.to_string(),
(DateKind::Between, Some(from), Some(to)) => format!("between {from} and {to}"),
(DateKind::Period, Some(from), Some(to)) => format!("from {from} to {to}"),
(DateKind::Period, Some(from), None) => format!("from {from}"),
(DateKind::Period, None, Some(to)) => format!("to {to}"),
(kind, Some(point), _) | (kind, None, Some(point)) => {
format!("{} {point}", kind.label())
}
_ => self.original.trim().to_owned(),
}
}
#[must_use]
pub fn to_gedcom_5_5_1(&self) -> Option<String> {
let escape = match self.calendar {
Calendar::Gregorian => "",
Calendar::Julian => "@#DJULIAN@ ",
_ => return None,
};
let gregorian = self.calendar == Calendar::Gregorian;
let before_era = [self.earliest, self.latest]
.iter()
.flatten()
.any(|point| point.year < 0);
if before_era && !gregorian {
return None;
}
let dual = dual_year(&self.original);
let spelled_dual = match dual {
Some((first, second)) if second == first + 1 && gregorian && !before_era => {
Some(second)
}
Some(_) if self.kind != DateKind::Exact => return None,
_ => None,
};
let point = |point: DatePoint| -> String {
let mut text = String::from(escape);
text.push_str(&point_5_5_1(point));
if let Some(second) = spelled_dual {
let _ = write!(text, "/{:02}", second.rem_euclid(100));
}
text
};
let reading = match (self.kind, self.earliest, self.latest) {
(DateKind::Exact | DateKind::Interpreted, Some(at), _) => point(at),
(DateKind::About, Some(at), _) => format!("ABT {}", point(at)),
(DateKind::Calculated, Some(at), _) => format!("CAL {}", point(at)),
(DateKind::Estimated, Some(at), _) => format!("EST {}", point(at)),
(DateKind::Before, _, Some(at)) => format!("BEF {}", point(at)),
(DateKind::After, Some(at), _) => format!("AFT {}", point(at)),
(DateKind::Between, Some(from), Some(to)) => {
format!("BET {} AND {}", point(from), point(to))
}
(DateKind::Period, Some(from), Some(to)) => {
format!("FROM {} TO {}", point(from), point(to))
}
(DateKind::Period, Some(from), None) => format!("FROM {}", point(from)),
(DateKind::Period, None, Some(to)) => format!("TO {}", point(to)),
_ => return None,
};
let phrase = match (&self.phrase, dual) {
(Some(phrase), _) => Some(phrase.clone()),
(None, Some(_)) if spelled_dual.is_none() => Some(self.original.trim().to_owned()),
_ => None,
};
match phrase {
None if self.kind == DateKind::Interpreted => None,
None => Some(reading),
Some(phrase) if matches!(self.kind, DateKind::Exact | DateKind::Interpreted) => {
let phrase = phrase.replace('(', "[").replace(')', "]");
Some(format!("INT {reading} ({phrase})"))
}
Some(_) => None,
}
}
#[must_use]
pub fn to_gedcom_7(&self) -> Option<String> {
let calendar_word = match self.calendar {
Calendar::Gregorian => "",
Calendar::Julian => "JULIAN ",
Calendar::Hebrew => "HEBREW ",
Calendar::FrenchRepublican => "FRENCH_R ",
Calendar::Roman | Calendar::Unknown => return None,
};
if !matches!(self.calendar, Calendar::Gregorian | Calendar::Julian)
&& [self.earliest, self.latest]
.iter()
.flatten()
.any(|point| point.month.is_some())
{
return None;
}
let gregorian = self.calendar == Calendar::Gregorian;
let before_era = [self.earliest, self.latest]
.iter()
.flatten()
.any(|point| point.year < 0);
if before_era && !gregorian {
return None;
}
if dual_year(&self.original).is_some() {
return None;
}
let point = |point: DatePoint| -> String {
format!("{calendar_word}{}", point.to_gedcom())
};
match (self.kind, self.earliest, self.latest) {
(DateKind::Exact | DateKind::Interpreted, Some(at), _) => Some(point(at)),
(DateKind::About, Some(at), _) => Some(format!("ABT {}", point(at))),
(DateKind::Calculated, Some(at), _) => Some(format!("CAL {}", point(at))),
(DateKind::Estimated, Some(at), _) => Some(format!("EST {}", point(at))),
(DateKind::Before, _, Some(at)) => Some(format!("BEF {}", point(at))),
(DateKind::After, Some(at), _) => Some(format!("AFT {}", point(at))),
(DateKind::Between, Some(from), Some(to)) => {
Some(format!("BET {} AND {}", point(from), point(to)))
}
(DateKind::Period, Some(from), Some(to)) => {
Some(format!("FROM {} TO {}", point(from), point(to)))
}
(DateKind::Period, Some(from), None) => Some(format!("FROM {}", point(from))),
(DateKind::Period, None, Some(to)) => Some(format!("TO {}", point(to))),
_ => None,
}
}
}
fn point_5_5_1(point: DatePoint) -> String {
if point.year < 0 {
let positive = DatePoint {
year: -point.year,
..point
};
return format!("{} B.C.", positive.to_gedcom());
}
point.to_gedcom()
}
fn dual_year(text: &str) -> Option<(i32, i32)> {
let bytes = text.as_bytes();
let mut index = 0;
while index < bytes.len() {
if !bytes[index].is_ascii_digit() {
index += 1;
continue;
}
let start = index;
while index < bytes.len() && bytes[index].is_ascii_digit() {
index += 1;
}
let first_text = &text[start..index];
if index >= bytes.len() || bytes[index] != b'/' || !(3..=4).contains(&first_text.len()) {
continue;
}
let second_start = index + 1;
let mut end = second_start;
while end < bytes.len() && bytes[end].is_ascii_digit() {
end += 1;
}
let second_text = &text[second_start..end];
if second_text.is_empty() || bytes.get(end) == Some(&b'/') {
index = end;
continue;
}
let first = first_text.parse::<i32>().ok()?;
let second = if second_text.len() >= first_text.len() {
second_text.parse::<i32>().ok()?
} else {
let modulus = 10_i32.pow(u32::try_from(second_text.len()).ok()?);
let tail = second_text.parse::<i32>().ok()?;
let mut second = first - first.rem_euclid(modulus) + tail;
if second <= first {
second += modulus;
}
second
};
return Some((first, second));
}
None
}
fn parse_expression(input: &str) -> Option<GedcomDate> {
let trimmed = input.trim();
if trimmed.is_empty() {
return None;
}
let build = |kind: DateKind, earliest: Option<DatePoint>, latest: Option<DatePoint>| {
Some(GedcomDate {
original: String::new(),
calendar: Calendar::Gregorian,
kind,
earliest,
latest,
phrase: None,
})
};
if let Some((left, right)) = split_keyword(trimmed, &["AND", "ET", "UND", "OG"])
&& let Some(prefix) =
strip_keyword(left, &["BET", "BETWEEN", "ENTRE", "ZWISCHEN", "MELLOM"])
{
let from = parse_point(prefix);
let to = parse_point(right);
if from.is_some() || to.is_some() {
return build(DateKind::Between, from, to);
}
}
if let Some((left, right)) = split_keyword(trimmed, &["TO", "AU", "BIS", "TIL"])
&& let Some(prefix) = strip_keyword(left, &["FROM", "DE", "DU", "VON", "FRA"])
{
let from = parse_point(prefix);
let to = parse_point(right);
if from.is_some() || to.is_some() {
return build(DateKind::Period, from, to);
}
}
if let Some(rest) = strip_keyword(trimmed, &["FROM", "SINCE"]) {
return build(DateKind::Period, parse_point(rest), None);
}
if let Some(rest) = strip_keyword(trimmed, &["TO", "UNTIL"]) {
return build(DateKind::Period, None, parse_point(rest));
}
for (kind, words) in PREFIXES {
if let Some(rest) = strip_keyword(trimmed, words) {
let point = parse_point(rest)?;
return match kind {
DateKind::Before => build(DateKind::Before, None, Some(point)),
DateKind::After => build(DateKind::After, Some(point), None),
other => build(*other, Some(point), Some(point)),
};
}
}
if let Some(inner) = trimmed
.strip_prefix('<')
.and_then(|rest| rest.strip_suffix('>'))
{
let point = parse_point(inner)?;
return build(DateKind::About, Some(point), Some(point));
}
let point = parse_point(trimmed)?;
build(DateKind::Exact, Some(point), Some(point))
}
const PREFIXES: &[(DateKind, &[&str])] = &[
(
DateKind::About,
&[
"ABT",
"ABOUT",
"CIRCA",
"CA",
"APPROX",
"ENVIRON",
"UM",
"RUNDT",
"OMKRING",
"CERCA",
"VERSO",
"VERS",
"AROUND",
"APPROXIMATELY",
"APROXIMADAMENTE",
"OMTRENT",
"CIRKA",
"CIR",
"C",
"PROBABLY",
"PROBABLEMENT",
],
),
(DateKind::Calculated, &["CAL", "CALCULATED"]),
(DateKind::Estimated, &["EST", "ESTIMATED"]),
(
DateKind::Before,
&["BEF", "BEFORE", "AVANT", "VOR", "PRIMA", "ANTES", "FØR"],
),
(
DateKind::After,
&[
"AFT", "AFTER", "APRES", "APRÈS", "NACH", "DOPO", "DESPUES", "ETTER", "EFTER",
],
),
(DateKind::Interpreted, &["INT", "INTERPRETED"]),
];
fn split_calendar(input: &str) -> (Calendar, String) {
if let Some(rest) = input.strip_prefix("@#D")
&& let Some((name, remainder)) = rest.split_once('@')
{
return (calendar_from(name), remainder.trim().to_owned());
}
for (name, calendar) in [
("GREGORIAN", Calendar::Gregorian),
("JULIAN", Calendar::Julian),
("HEBREW", Calendar::Hebrew),
("FRENCH_R", Calendar::FrenchRepublican),
("ROMAN", Calendar::Roman),
] {
if let Some(rest) = input.strip_prefix(name)
&& rest.starts_with(' ')
{
return (calendar, rest.trim().to_owned());
}
}
(Calendar::Gregorian, input.to_owned())
}
fn calendar_from(name: &str) -> Calendar {
match name.trim().to_ascii_uppercase().as_str() {
"GREGORIAN" => Calendar::Gregorian,
"JULIAN" => Calendar::Julian,
"HEBREW" => Calendar::Hebrew,
"FRENCH R" | "FRENCH_R" => Calendar::FrenchRepublican,
"ROMAN" => Calendar::Roman,
_ => Calendar::Unknown,
}
}
fn split_phrase(input: &str) -> (String, Option<String>) {
let trimmed = input.trim();
let Some(open) = trimmed.find('(') else {
return (trimmed.to_owned(), None);
};
let Some(close) = trimmed.rfind(')') else {
return (trimmed.to_owned(), None);
};
if close < open {
return (trimmed.to_owned(), None);
}
let phrase = trimmed[open + 1..close].trim().to_owned();
let rest = format!("{} {}", &trimmed[..open], &trimmed[close + 1..]);
(
rest.trim().to_owned(),
(!phrase.is_empty()).then_some(phrase),
)
}
fn split_keyword<'a>(input: &'a str, words: &[&str]) -> Option<(&'a str, &'a str)> {
let upper = input.to_ascii_uppercase();
for word in words {
let needle = format!(" {word} ");
if let Some(position) = upper.find(&needle) {
return Some((&input[..position], &input[position + needle.len()..]));
}
}
None
}
fn strip_keyword<'a>(input: &'a str, words: &[&str]) -> Option<&'a str> {
let trimmed = input.trim();
for word in words {
let Some(after) = strip_prefix_ignoring_case(trimmed, word) else {
continue;
};
let rest = match after.chars().next() {
Some(' ' | '.') => after[1..].trim(),
Some(digit) if digit.is_ascii_digit() => after,
_ => continue,
};
if !rest.is_empty() {
return Some(rest);
}
}
None
}
fn strip_prefix_ignoring_case<'a>(text: &'a str, word: &str) -> Option<&'a str> {
let mut characters = text.char_indices();
let mut end = 0;
for expected in word.chars() {
let (index, actual) = characters.next()?;
if !actual.to_lowercase().eq(expected.to_lowercase()) {
return None;
}
end = index + actual.len_utf8();
}
Some(&text[end..])
}
fn parse_point(input: &str) -> Option<DatePoint> {
let cleaned = input.trim().trim_end_matches([',', '.', ';']).trim();
if cleaned.is_empty() {
return None;
}
if let Some(rest) = ["BCE", "bce", "B.C", "b.c"]
.iter()
.find_map(|suffix| cleaned.strip_suffix(suffix))
.map(str::trim_end)
.filter(|rest| !rest.is_empty())
{
let mut point = parse_point(rest)?;
point.year = -point.year;
return Some(point);
}
if let Some(point) = parse_numeric(cleaned) {
return Some(point);
}
if let Some(point) = parse_dual_year(cleaned) {
return Some(point);
}
if let Some((head, tail)) = cleaned.rsplit_once(' ')
&& let Some(year) = parse_dual_year(tail)
{
return parse_point(&format!("{head} {}", year.year));
}
let tokens: Vec<String> = cleaned
.split([' ', '\u{a0}', ','])
.flat_map(split_runs)
.filter(|token| {
!matches!(
token.to_ascii_lowercase().as_str(),
"de" | "of" | "den" | "le"
)
})
.collect();
let tokens: Vec<&str> = tokens.iter().map(String::as_str).collect();
match tokens.as_slice() {
[year] => parse_year(year).map(DatePoint::year),
[month, year] => {
let year = parse_year(year)?;
Some(DatePoint {
year,
month: Some(month_number(month)?),
day: None,
})
}
[first, second, year] => {
let year = parse_year(year)?;
let named = |token: &str| !token.bytes().all(|byte| byte.is_ascii_digit());
let (month, day) = if named(second) {
(month_number(second)?, first)
} else if named(first) {
(month_number(first)?, second)
} else {
month_number(second).map_or_else(
|| month_number(first).map(|month| (month, second)),
|month| Some((month, first)),
)?
};
Some(DatePoint {
year,
month: Some(month),
day: Some(parse_day(day)?),
})
}
[first, second, third, fourth] => {
let year = parse_year(fourth)?;
let month = month_number(first).or_else(|| month_number(second))?;
let day = parse_day(second).or_else(|| parse_day(third))?;
Some(DatePoint {
year,
month: Some(month),
day: Some(day),
})
}
_ => None,
}
}
fn parse_numeric(input: &str) -> Option<DatePoint> {
for (separator, month_first) in [('.', false), ('/', true), ('-', false)] {
let parts: Vec<&str> = input.split(separator).collect();
if parts.len() != 3 {
continue;
}
if !parts
.iter()
.all(|part| !part.is_empty() && part.bytes().all(|byte| byte.is_ascii_digit()))
{
continue;
}
if parts[0].len() == 4 {
let year = parts[0].parse::<i32>().ok()?;
return Some(DatePoint {
year,
month: parse_month_number(parts[1]),
day: parse_day(parts[2]),
});
}
let year = parse_year(parts[2])?;
let (day, month) = if month_first {
(parts[1], parts[0])
} else {
(parts[0], parts[1])
};
return Some(DatePoint {
year,
month: parse_month_number(month),
day: parse_day(day),
});
}
None
}
fn parse_dual_year(input: &str) -> Option<DatePoint> {
let (first, second) = input.split_once('/')?;
let digits = |value: &str| !value.is_empty() && value.bytes().all(|byte| byte.is_ascii_digit());
if !digits(first) || !digits(second) || !(3..=4).contains(&first.len()) {
return None;
}
Some(DatePoint::year(first.parse::<i32>().ok()?))
}
fn split_runs(token: &str) -> Vec<String> {
let mut runs = Vec::new();
let mut current = String::new();
let mut in_digits = false;
for character in token.chars() {
let digit = character.is_ascii_digit();
if !current.is_empty() && digit != in_digits {
runs.push(std::mem::take(&mut current));
}
in_digits = digit;
current.push(character);
}
if !current.is_empty() {
runs.push(current);
}
runs.retain(|run| run.chars().any(char::is_alphanumeric));
runs
}
fn parse_year(token: &str) -> Option<i32> {
let cleaned = token.trim_matches(|character: char| !character.is_ascii_digit());
if cleaned.is_empty() {
return None;
}
let first = cleaned.split('/').next()?;
let year = first.parse::<i32>().ok()?;
(0..=9999).contains(&year).then_some(year)
}
fn parse_day(token: &str) -> Option<u8> {
let cleaned = token.trim_matches(|character: char| !character.is_ascii_digit());
let day = cleaned.parse::<u8>().ok()?;
(1..=31).contains(&day).then_some(day)
}
fn parse_month_number(token: &str) -> Option<u8> {
let month = token.trim().parse::<u8>().ok()?;
(1..=12).contains(&month).then_some(month)
}
fn month_number(token: &str) -> Option<u8> {
let key = token
.trim_matches(|character: char| !character.is_alphanumeric())
.to_lowercase();
if key.is_empty() {
return None;
}
if let Some(month) = parse_month_number(&key) {
return Some(month);
}
for (index, names) in MONTH_NAMES.iter().enumerate() {
if names.iter().any(|name| *name == key) {
return u8::try_from(index + 1).ok();
}
}
for (index, names) in MONTH_NAMES.iter().enumerate() {
if key.len() >= 3 && names[1].starts_with(&key) {
return u8::try_from(index + 1).ok();
}
}
None
}
const MONTH_NAMES: [&[&str]; 12] = [
&[
"jan", "january", "janvier", "janv", "januar", "januari", "gennaio", "enero", "jänner",
],
&[
"feb", "february", "février", "fevrier", "févr", "fév", "februar", "februari", "febbraio",
"febrero",
],
&[
"mar", "march", "mars", "märz", "marz", "maerz", "marzo", "marts",
],
&["apr", "april", "avril", "avr", "aprile", "abril"],
&["may", "mai", "maggio", "mayo", "mei"],
&["jun", "june", "juin", "juni", "giugno", "junio"],
&["jul", "july", "juillet", "juil", "juli", "luglio", "julio"],
&["aug", "august", "août", "aout", "agosto", "augusti"],
&[
"sep",
"september",
"septembre",
"settembre",
"septiembre",
"sept",
],
&[
"oct", "october", "octobre", "oktober", "ottobre", "octubre", "okt",
],
&["nov", "november", "novembre", "noviembre"],
&[
"dec",
"december",
"décembre",
"decembre",
"dezember",
"desember",
"déc",
"dicembre",
"diciembre",
"des",
],
];
fn month_abbreviation(month: u8) -> &'static str {
const NAMES: [&str; 12] = [
"JAN", "FEB", "MAR", "APR", "MAY", "JUN", "JUL", "AUG", "SEP", "OCT", "NOV", "DEC",
];
NAMES
.get(usize::from(month.saturating_sub(1)))
.copied()
.unwrap_or("")
}
fn month_name(month: u8) -> &'static str {
const NAMES: [&str; 12] = [
"January",
"February",
"March",
"April",
"May",
"June",
"July",
"August",
"September",
"October",
"November",
"December",
];
NAMES
.get(usize::from(month.saturating_sub(1)))
.copied()
.unwrap_or("")
}
#[derive(Clone, Debug, Eq, PartialEq)]
#[non_exhaustive]
#[cfg_attr(feature = "serde", derive(serde::Serialize))]
#[cfg_attr(feature = "serde", serde(rename_all = "camelCase"))]
pub struct Timestamp {
pub date: GedcomDate,
pub time: Option<String>,
}
impl Timestamp {
#[must_use]
pub fn display(&self) -> String {
self.time.as_ref().map_or_else(
|| self.date.display(),
|time| format!("{} {time}", self.date.display()),
)
}
}
#[must_use]
pub fn changed(record: &Node) -> Option<Timestamp> {
["CHAN", "CREA"]
.iter()
.find_map(|tag| record.first(tag))
.and_then(timestamp_of)
}
#[must_use]
pub fn timestamp_of(node: &Node) -> Option<Timestamp> {
let date = node.first("DATE")?;
Some(Timestamp {
time: date.value_of("TIME").or_else(|| node.value_of("TIME")),
date: GedcomDate::parse(&date.logical_value()),
})
}
#[must_use]
pub fn timestamp_node(tag: &str, point: DatePoint, time: Option<&str>) -> Node {
let mut date = Node::with_value("DATE", point.to_gedcom());
if let Some(time) = time {
date = date.child(Node::with_value("TIME", time));
}
Node::new(tag).child(date)
}
fn four_digit_year(input: &str) -> Option<i32> {
let bytes = input.as_bytes();
let mut index = 0;
while index < bytes.len() {
if !bytes[index].is_ascii_digit() {
index += 1;
continue;
}
let start = index;
while index < bytes.len() && bytes[index].is_ascii_digit() {
index += 1;
}
if index - start == 4
&& let Ok(year) = input[start..index].parse::<i32>()
&& (1000..=2999).contains(&year)
{
return Some(year);
}
}
None
}
#[cfg(test)]
mod tests {
use super::*;
fn parsed(payload: &str) -> GedcomDate {
GedcomDate::parse(payload)
}
#[test]
fn the_payload_is_never_modified() {
let date = parsed(" ABT 1900 ");
assert_eq!(date.original, " ABT 1900 ");
}
#[test]
fn a_month_is_never_spelled_in_a_calendar_whose_months_are_unknown() {
assert_eq!(parsed("@#DFRENCH R@ 22 3 5").to_gedcom_7(), None);
assert_eq!(
parsed("@#DFRENCH R@ 5").to_gedcom_7().as_deref(),
Some("FRENCH_R 5")
);
assert_eq!(
parsed("@#DJULIAN@ 10 MAR 1700").to_gedcom_7().as_deref(),
Some("JULIAN 10 MAR 1700")
);
}
#[test]
fn a_reading_is_written_in_the_5_5_1_grammar() {
let written = |text: &str| parsed(text).to_gedcom_5_5_1();
let cases: &[(&str, Option<&str>)] = &[
("5 JAN 1882", Some("5 JAN 1882")),
("29 May 1823", Some("29 MAY 1823")),
("Apr 1850", Some("APR 1850")),
("about 1575", Some("ABT 1575")),
("Abt. 1656", Some("ABT 1656")),
("before 29 September 1793", Some("BEF 29 SEP 1793")),
("after 1850", Some("AFT 1850")),
("8. Juni 1886", Some("8 JUN 1886")),
("14.10.1635", Some("14 OCT 1635")),
("<1873>", Some("ABT 1873")),
("May 22, 1942", Some("22 MAY 1942")),
("23Jun1807", Some("23 JUN 1807")),
("environ janvier 1620", Some("ABT JAN 1620")),
("avant 6 septembre 1892", Some("BEF 6 SEP 1892")),
("BET 1898 AND 1902", Some("BET 1898 AND 1902")),
("FROM 1701", Some("FROM 1701")),
("CAL 1824", Some("CAL 1824")),
("EST 1630", Some("EST 1630")),
("1699/00", Some("1699/00")),
("1679/1683", Some("INT 1679 (1679/1683)")),
("ABT 1679/1683", None),
(
"INT 5 JAN 1882 (the fifth, by the ink)",
Some("INT 5 JAN 1882 (the fifth, by the ink)"),
),
("@#DJULIAN@ 10 MAR 1700", Some("@#DJULIAN@ 10 MAR 1700")),
("@#DHEBREW@ 1 TSH 5600", None),
("INT 1900", None),
("1740/1741 BCE", Some("INT 1740 B.C. (1740/1741 BCE)")),
("@#DJULIAN@ 44 BCE", None),
(
"@#DJULIAN@ 1740/41",
Some("INT @#DJULIAN@ 1740 (@#DJULIAN@ 1740/41)"),
),
("ABT @#DJULIAN@ 1740/41", None),
("44 BCE", Some("44 B.C.")),
("DEAD", None),
("", None),
];
for (text, expected) in cases {
assert_eq!(written(text).as_deref(), *expected, "{text:?}");
}
}
#[test]
fn what_is_written_reads_back_as_the_same_reading() {
for text in [
"about 5 January 1882",
"before July 1843",
"from 1701 to 1742",
"1740/41",
"44 BCE",
] {
let reading = parsed(text);
let written = reading.to_gedcom_5_5_1().expect("writable");
let reread = parsed(&written);
assert_eq!(
(reread.kind, reread.earliest, reread.latest),
(reading.kind, reading.earliest, reading.latest),
"{text:?} -> {written:?}"
);
}
}
#[test]
fn a_reading_is_written_in_the_7_0_grammar() {
let written = |text: &str| parsed(text).to_gedcom_7();
let cases: &[(&str, Option<&str>)] = &[
("5 JAN 1882", Some("5 JAN 1882")),
("29 May 1823", Some("29 MAY 1823")),
("Apr 1850", Some("APR 1850")),
("about 1575", Some("ABT 1575")),
("Abt. 1656", Some("ABT 1656")),
("before 29 September 1793", Some("BEF 29 SEP 1793")),
("after 1850", Some("AFT 1850")),
("8. Juni 1886", Some("8 JUN 1886")),
("14.10.1635", Some("14 OCT 1635")),
("<1873>", Some("ABT 1873")),
("May 22, 1942", Some("22 MAY 1942")),
("23Jun1807", Some("23 JUN 1807")),
("environ janvier 1620", Some("ABT JAN 1620")),
("avant 6 septembre 1892", Some("BEF 6 SEP 1892")),
("BET 1898 AND 1902", Some("BET 1898 AND 1902")),
("FROM 1701", Some("FROM 1701")),
("CAL 1824", Some("CAL 1824")),
("EST 1630", Some("EST 1630")),
("13. september 1900", Some("13 SEP 1900")),
("ABT Oct 1803", Some("ABT OCT 1803")),
("44 BCE", Some("44 BCE")),
("44 B.C.", Some("44 BCE")),
("ABT 100 BCE", Some("ABT 100 BCE")),
("@#DJULIAN@ 10 MAR 1700", Some("JULIAN 10 MAR 1700")),
("JULIAN 10 MAR 1700", Some("JULIAN 10 MAR 1700")),
("@#DHEBREW@ 1 TSH 5600", None),
("@#DJULIAN@ 44 BCE", None),
("INT 1900 (a reading)", Some("1900")),
("1740/41", None),
("1699/00", None),
("1679/1683", None),
("@#DJULIAN@ 1740/41", None),
("DEAD", None),
("", None),
];
for (text, expected) in cases {
assert_eq!(written(text).as_deref(), *expected, "{text:?}");
}
}
#[test]
fn what_version_7_writes_reads_back_as_the_same_reading() {
for text in [
"about 5 January 1882",
"before July 1843",
"from 1701 to 1742",
"44 BCE",
"@#DJULIAN@ 10 MAR 1700",
] {
let reading = parsed(text);
let written = reading.to_gedcom_7().expect("writable");
let reread = parsed(&written);
assert_eq!(
(reread.kind, reread.earliest, reread.latest),
(reading.kind, reading.earliest, reading.latest),
"{text:?} -> {written:?}"
);
}
}
#[test]
fn the_spellings_a_producer_writing_its_own_data_met() {
let written = |text: &str| parsed(text).to_gedcom_5_5_1();
let cases: &[(&str, &str)] = &[
("après 12 juillet 1670", "AFT 12 JUL 1670"),
("før 1679", "BEF 1679"),
("etter 1649", "AFT 1649"),
("vers 1646", "ABT 1646"),
("Cir 1700", "ABT 1700"),
("c. 1625", "ABT 1625"),
("c1700", "ABT 1700"),
("abt1810", "ABT 1810"),
("around 1580", "ABT 1580"),
("before29 Sept 1793", "BEF 29 SEP 1793"),
("April 9, 1950", "9 APR 1950"),
("12 DÉC 1790", "12 DEC 1790"),
("desember 1702", "DEC 1702"),
("10 January 1566/67", "10 JAN 1566/67"),
("entre 1700 et 1705", "BET 1700 AND 1705"),
("du 1656 au 1663", "FROM 1656 TO 1663"),
];
for (text, expected) in cases {
assert_eq!(written(text).as_deref(), Some(*expected), "{text:?}");
}
assert_eq!(parsed("CAL 1824").kind, DateKind::Calculated);
assert_eq!(parsed("Catherine").kind, DateKind::Unparsed);
}
#[test]
fn a_numeric_date_is_not_a_dual_year() {
assert_eq!(dual_year("4/18/1616"), None);
assert_eq!(dual_year("8.2.1664"), None);
assert_eq!(dual_year("1740/41"), Some((1740, 1741)));
assert_eq!(dual_year("5 JAN 1699/00"), Some((1699, 1700)));
}
#[test]
fn specification_forms_read_as_written() {
assert_eq!(parsed("1821").earliest, Some(DatePoint::year(1821)));
let exact = parsed("5 JAN 1882");
assert_eq!(exact.kind, DateKind::Exact);
assert_eq!(
exact.earliest,
Some(DatePoint {
year: 1882,
month: Some(1),
day: Some(5)
})
);
assert_eq!(parsed("ABT 1900").kind, DateKind::About);
assert_eq!(parsed("EST 1630").kind, DateKind::Estimated);
assert_eq!(parsed("CAL 1824").kind, DateKind::Calculated);
assert_eq!(parsed("BEF 1815").kind, DateKind::Before);
assert_eq!(parsed("AFT 1649").kind, DateKind::After);
}
#[test]
fn ranges_and_periods_keep_both_ends() {
let between = parsed("BET 1898 AND 1902");
assert_eq!(between.kind, DateKind::Between);
assert_eq!(between.earliest, Some(DatePoint::year(1898)));
assert_eq!(between.latest, Some(DatePoint::year(1902)));
let period = parsed("from 1701 to 1742");
assert_eq!(period.kind, DateKind::Period);
assert_eq!(period.earliest, Some(DatePoint::year(1701)));
assert_eq!(period.latest, Some(DatePoint::year(1742)));
}
#[test]
fn the_forms_the_requirement_corpus_actually_contains() {
type Case = (&'static str, i32, Option<u8>, Option<u8>, DateKind);
let cases: &[Case] = &[
("5 January 1882", 1882, Some(1), Some(5), DateKind::Exact),
("22 March 1635", 1635, Some(3), Some(22), DateKind::Exact),
("about 1575", 1575, None, None, DateKind::About),
("Abt 1872", 1872, None, None, DateKind::About),
("abt. 1625", 1625, None, None, DateKind::About),
("rundt 1645", 1645, None, None, DateKind::About),
("um 1660", 1660, None, None, DateKind::About),
("environ 1830", 1830, None, None, DateKind::About),
("4 juillet 1764", 1764, Some(7), Some(4), DateKind::Exact),
("21. Januar 1809", 1809, Some(1), Some(21), DateKind::Exact),
("22. März 1873", 1873, Some(3), Some(22), DateKind::Exact),
("16 gennaio 1847", 1847, Some(1), Some(16), DateKind::Exact),
(
"13 de enero de 1855",
1855,
Some(1),
Some(13),
DateKind::Exact,
),
("January 1652", 1652, Some(1), None, DateKind::Exact),
("8.2.1664", 1664, Some(2), Some(8), DateKind::Exact),
("4/18/1616", 1616, Some(4), Some(18), DateKind::Exact),
("<1802>", 1802, None, None, DateKind::About),
(
"before 29 March 1795",
1795,
Some(3),
Some(29),
DateKind::Before,
),
("Vor Februar 1843", 1843, Some(2), None, DateKind::Before),
("19JUL1818", 1818, Some(7), Some(19), DateKind::Exact),
("21 Nov1896", 1896, Some(11), Some(21), DateKind::Exact),
("April 15, 1873", 1873, Some(4), Some(15), DateKind::Exact),
("07 October 1853,", 1853, Some(10), Some(7), DateKind::Exact),
("1679/1683", 1679, None, None, DateKind::Exact),
("ABT Oct 1803", 1803, Some(10), None, DateKind::About),
(
"About 21 Jun 1969",
1969,
Some(6),
Some(21),
DateKind::About,
),
];
for (payload, year, month, day, kind) in cases {
let date = parsed(payload);
assert_eq!(date.kind, *kind, "kind of {payload:?}");
let point = date
.earliest
.or(date.latest)
.unwrap_or_else(|| panic!("{payload:?} produced no point"));
assert_eq!(point.year, *year, "year of {payload:?}");
assert_eq!(point.month, *month, "month of {payload:?}");
assert_eq!(point.day, *day, "day of {payload:?}");
}
}
#[test]
fn what_cannot_be_read_keeps_its_text_and_says_so() {
for payload in ["Infant", "DEAD", "Unknown", "1799 or 12 April 1811"] {
let date = parsed(payload);
assert_eq!(date.kind, DateKind::Unparsed, "{payload:?}");
assert_eq!(date.display(), payload);
assert_eq!(date.phrase.as_deref(), Some(payload));
}
}
#[test]
fn an_unreadable_expression_can_still_offer_a_year_to_sort_by() {
let date = parsed("19 Jun 1820 age 26");
assert_eq!(date.kind, DateKind::Unparsed);
assert_eq!(date.year_hint(), Some(1820));
assert!(date.sort_key().is_some());
}
#[test]
fn a_calendar_escape_is_recognized_in_both_versions() {
let five = parsed("@#DJULIAN@ 12 FEB 1740");
assert_eq!(five.calendar, Calendar::Julian);
assert_eq!(five.earliest.expect("point").day, Some(12));
let seven = parsed("JULIAN 12 FEB 1740");
assert_eq!(seven.calendar, Calendar::Julian);
}
#[test]
fn an_interpretation_keeps_its_phrase() {
let date = parsed("INT 1900 (the year the church burned)");
assert_eq!(date.kind, DateKind::Interpreted);
assert_eq!(date.earliest, Some(DatePoint::year(1900)));
assert_eq!(date.phrase.as_deref(), Some("the year the church burned"));
}
#[test]
fn a_version_seven_bce_year_reads_negative_and_writes_back() {
let date = parsed("44 BCE");
assert_eq!(date.kind, DateKind::Exact);
assert_eq!(date.earliest, Some(DatePoint::year(-44)));
assert_eq!(date.earliest.expect("point").to_gedcom(), "44 BCE");
let about = parsed("ABT 100 BCE");
assert_eq!(about.kind, DateKind::About);
assert_eq!(about.earliest, Some(DatePoint::year(-100)));
assert!(
parsed("100 BCE").sort_key() < parsed("44 BCE").sort_key(),
"100 BCE came after 44 BCE"
);
}
#[test]
fn a_before_only_date_sorts_at_its_boundary_year() {
let date = parsed("BEF 1815");
assert_eq!(date.earliest, None);
assert_eq!(date.year_hint(), Some(1815));
assert_eq!(date.sort_key(), Some(1815 * 10_000));
}
#[test]
fn dates_sort_by_the_earliest_instant_they_admit() {
let mut dates = ["1900", "5 JAN 1882", "ABT 1875", "BET 1890 AND 1895"]
.map(parsed)
.to_vec();
dates.sort_by_key(GedcomDate::sort_key);
let order: Vec<&str> = dates.iter().map(|date| date.original.as_str()).collect();
assert_eq!(
order,
["ABT 1875", "5 JAN 1882", "BET 1890 AND 1895", "1900"]
);
}
}