use serde::{Deserialize, Deserializer, Serialize};
use std::collections::{HashMap, HashSet};
#[derive(Debug, Clone, Default, Deserialize, Serialize)]
#[serde(rename_all = "kebab-case")]
pub struct Reference {
pub id: String,
#[serde(rename = "type")]
pub ref_type: String,
#[serde(skip_serializing_if = "Option::is_none")]
pub author: Option<Vec<Name>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub editor: Option<Vec<Name>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub translator: Option<Vec<Name>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub recipient: Option<Vec<Name>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub director: Option<Vec<Name>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub interviewer: Option<Vec<Name>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub title: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub container_title: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub collection_title: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub collection_number: Option<StringOrNumber>,
#[serde(skip_serializing_if = "Option::is_none")]
pub issued: Option<DateVariable>,
#[serde(skip_serializing_if = "Option::is_none")]
pub accessed: Option<DateVariable>,
#[serde(skip_serializing_if = "Option::is_none")]
pub volume: Option<StringOrNumber>,
#[serde(skip_serializing_if = "Option::is_none")]
pub issue: Option<StringOrNumber>,
#[serde(skip_serializing_if = "Option::is_none")]
pub page: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub edition: Option<StringOrNumber>,
#[serde(skip_serializing_if = "Option::is_none")]
#[serde(rename = "DOI")]
pub doi: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
#[serde(rename = "URL")]
pub url: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
#[serde(rename = "ISBN")]
pub isbn: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
#[serde(rename = "ISSN")]
pub issn: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub publisher: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub publisher_place: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub authority: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub section: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub event: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub medium: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub archive: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
#[serde(alias = "archive_location")]
pub archive_location: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub number: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub chapter_number: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub printing_number: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub genre: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub language: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub original_title: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
#[serde(rename = "abstract")]
pub abstract_text: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub note: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub number_of_pages: Option<StringOrNumber>,
#[serde(skip_serializing_if = "Option::is_none")]
pub number_of_volumes: Option<StringOrNumber>,
#[serde(flatten)]
pub extra: HashMap<String, serde_json::Value>,
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Default)]
pub struct Name {
pub family: Option<String>,
pub given: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub literal: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub suffix: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub dropping_particle: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub non_dropping_particle: Option<String>,
}
impl Name {
#[must_use]
pub fn new(family: &str, given: &str) -> Self {
Self {
family: Some(family.to_string()),
given: Some(given.to_string()),
..Default::default()
}
}
#[must_use]
pub fn literal(name: &str) -> Self {
Self {
literal: Some(name.to_string()),
..Default::default()
}
}
#[must_use]
pub fn family_or_literal(&self) -> &str {
self.family
.as_deref()
.or(self.literal.as_deref())
.unwrap_or("")
}
}
#[derive(Debug, Clone, Default, Deserialize, Serialize)]
#[serde(rename_all = "kebab-case")]
pub struct DateVariable {
#[serde(
default,
skip_serializing_if = "Option::is_none",
deserialize_with = "deserialize_date_parts_opt"
)]
pub date_parts: Option<Vec<Vec<i32>>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub literal: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub raw: Option<String>,
#[serde(
default,
skip_serializing_if = "Option::is_none",
deserialize_with = "deserialize_season_opt"
)]
pub season: Option<i32>,
#[serde(skip_serializing_if = "Option::is_none")]
pub circa: Option<bool>,
}
#[derive(Debug, Deserialize)]
#[serde(untagged)]
enum IntOrString {
Int(i32),
String(String),
}
fn deserialize_date_parts_opt<'de, D>(deserializer: D) -> Result<Option<Vec<Vec<i32>>>, D::Error>
where
D: Deserializer<'de>,
{
let raw = Option::<Vec<Vec<IntOrString>>>::deserialize(deserializer)?;
raw.map(|rows| {
rows.into_iter()
.map(|row| {
row.into_iter()
.map(|value| match value {
IntOrString::Int(n) => Ok(n),
IntOrString::String(text) => text.parse::<i32>().map_err(|_| {
serde::de::Error::custom(format!(
"invalid date-parts component {:?}: expected integer or integer-like string",
text
))
}),
})
.collect::<Result<Vec<_>, D::Error>>()
})
.collect::<Result<Vec<_>, D::Error>>()
})
.transpose()
}
fn deserialize_season_opt<'de, D>(deserializer: D) -> Result<Option<i32>, D::Error>
where
D: Deserializer<'de>,
{
let raw = Option::<IntOrString>::deserialize(deserializer)?;
raw.map(|value| match value {
IntOrString::Int(n) => Ok(n),
IntOrString::String(text) => {
let normalized = text.trim().to_ascii_lowercase();
match normalized.as_str() {
"spring" => Ok(1),
"summer" => Ok(2),
"fall" | "autumn" => Ok(3),
"winter" => Ok(4),
_ => text.parse::<i32>().map_err(|_| {
serde::de::Error::custom(format!(
"invalid season {:?}: expected integer, integer-like string, or named season",
text
))
}),
}
}
})
.transpose()
}
impl DateVariable {
#[must_use]
pub fn year(year: i32) -> Self {
Self {
date_parts: Some(vec![vec![year]]),
..Default::default()
}
}
#[must_use]
pub fn year_month(year: i32, month: i32) -> Self {
Self {
date_parts: Some(vec![vec![year, month]]),
..Default::default()
}
}
#[must_use]
pub fn full(year: i32, month: i32, day: i32) -> Self {
Self {
date_parts: Some(vec![vec![year, month, day]]),
..Default::default()
}
}
#[must_use]
pub fn year_value(&self) -> Option<i32> {
self.date_parts
.as_ref()
.and_then(|parts| parts.first())
.and_then(|date| date.first())
.copied()
}
#[must_use]
pub fn month_value(&self) -> Option<i32> {
self.date_parts
.as_ref()
.and_then(|parts| parts.first())
.and_then(|date| date.get(1))
.copied()
}
#[must_use]
pub fn day_value(&self) -> Option<i32> {
self.date_parts
.as_ref()
.and_then(|parts| parts.first())
.and_then(|date| date.get(2))
.copied()
}
}
#[derive(Debug, Clone, Deserialize, Serialize)]
#[serde(untagged)]
pub enum StringOrNumber {
String(String),
Number(i64),
}
impl std::fmt::Display for StringOrNumber {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
match self {
Self::String(s) => write!(f, "{s}"),
Self::Number(n) => write!(f, "{n}"),
}
}
}
impl Reference {
pub fn parse_note_field_hacks(&mut self) {
let note = match &self.note {
Some(n) if !n.is_empty() => n.clone(),
_ => return,
};
let lines: Vec<&str> = note.lines().collect();
if lines.is_empty() {
return;
}
let mut parsed_indices = HashSet::new();
for (idx, line) in lines.iter().enumerate() {
let trimmed = line.trim();
if trimmed.is_empty() {
continue;
}
let Some((key, value)) = parse_key_value(trimmed) else {
continue;
};
if key.eq_ignore_ascii_case("type") {
self.ref_type = value.to_string();
parsed_indices.insert(idx);
} else if is_date_variable(key) {
handle_date_variable(self, key, value);
parsed_indices.insert(idx);
} else if is_name_variable(key) {
handle_name_variable(self, key, value);
parsed_indices.insert(idx);
} else if is_string_variable(key) {
handle_string_variable(self, key, value);
parsed_indices.insert(idx);
}
}
let remaining_lines: Vec<&str> = lines
.iter()
.enumerate()
.filter(|(idx, _)| !parsed_indices.contains(idx))
.map(|(_, line)| *line)
.collect();
if remaining_lines.is_empty() {
self.note = None;
} else {
self.note = Some(remaining_lines.join("\n"));
}
}
}
fn parse_key_value(line: &str) -> Option<(&str, &str)> {
if line.is_empty() {
return None;
}
let colon_pos = line.find(':')?;
#[allow(
clippy::string_slice,
reason = "colon_pos is found via find(':'), which is a valid 1-byte ASCII boundary"
)]
let key = &line[..colon_pos];
let first_char = key.chars().next()?;
if !first_char.is_ascii_alphabetic() {
return None;
}
if !key
.chars()
.skip(1)
.all(|c| c.is_ascii_alphanumeric() || c == '_' || c == '-')
{
return None;
}
#[allow(
clippy::string_slice,
reason = "colon_pos + 1 is a valid boundary after ':' (1-byte ASCII)"
)]
let value = line[colon_pos + 1..].trim();
Some((key, value))
}
fn is_date_variable(key: &str) -> bool {
matches!(
key,
"issued" | "event-date" | "original-date" | "available-date" | "accessed" | "submitted"
)
}
fn handle_date_variable(ref_obj: &mut Reference, key: &str, value: &str) {
let date = parse_date_variable(value);
match key {
"issued" if ref_obj.issued.is_none() => {
ref_obj.issued = Some(date);
}
"accessed" if ref_obj.accessed.is_none() => {
ref_obj.accessed = Some(date);
}
"event-date" | "original-date" | "available-date" | "submitted" => {
ref_obj.extra.insert(
key.to_string(),
serde_json::Value::String(value.to_string()),
);
}
_ => {}
}
}
fn parse_date_variable(value: &str) -> DateVariable {
let trimmed = value.trim();
let (sep_pos, sep_len) = if let Some(pos) = trimmed.find('/') {
(Some(pos), 1)
} else if let Some(pos) = trimmed.find('–') {
(Some(pos), '–'.len_utf8())
} else {
(None, 0)
};
if let Some(pos) = sep_pos {
#[allow(
clippy::string_slice,
reason = "pos is a valid byte boundary found via .find()"
)]
let start_str = trimmed[..pos].trim();
#[allow(
clippy::string_slice,
reason = "pos + sep_len is a valid byte boundary for '/' or '–'"
)]
let end_str = trimmed[pos + sep_len..].trim();
if let (Some(start_parts), Some(end_parts)) =
(parse_date_parts(start_str), parse_date_parts(end_str))
{
return DateVariable {
date_parts: Some(vec![start_parts, end_parts]),
..Default::default()
};
}
}
if let Some(parts) = parse_date_parts(trimmed) {
DateVariable {
date_parts: Some(vec![parts]),
..Default::default()
}
} else {
DateVariable {
raw: Some(trimmed.to_string()),
..Default::default()
}
}
}
fn parse_date_parts(s: &str) -> Option<Vec<i32>> {
let parts: Vec<&str> = s.split('-').collect();
if parts.is_empty() || parts.len() > 3 {
return None;
}
let mut result = Vec::new();
for part in parts {
result.push(part.trim().parse::<i32>().ok()?);
}
Some(result)
}
fn is_name_variable(key: &str) -> bool {
matches!(
key,
"editor"
| "translator"
| "interviewer"
| "director"
| "recipient"
| "author"
| "collection-editor"
| "container-author"
| "editorial-director"
| "illustrator"
| "original-author"
| "reviewed-author"
| "composer"
| "narrator"
| "performer"
| "producer"
| "script-writer"
| "compiler"
)
}
fn handle_name_variable(ref_obj: &mut Reference, key: &str, value: &str) {
let name = parse_name_variable(value);
match key {
"author" if ref_obj.author.is_none() => {
ref_obj.author = Some(vec![name]);
}
"editor" if ref_obj.editor.is_none() => {
ref_obj.editor = Some(vec![name]);
}
"translator" if ref_obj.translator.is_none() => {
ref_obj.translator = Some(vec![name]);
}
"interviewer" if ref_obj.interviewer.is_none() => {
ref_obj.interviewer = Some(vec![name]);
}
"director" if ref_obj.director.is_none() => {
ref_obj.director = Some(vec![name]);
}
"recipient" if ref_obj.recipient.is_none() => {
ref_obj.recipient = Some(vec![name]);
}
"collection-editor" | "container-author" | "editorial-director" | "illustrator"
| "original-author" | "reviewed-author" | "composer" | "narrator" | "performer"
| "producer" | "script-writer" | "compiler" => {
ref_obj.extra.insert(
key.to_string(),
serde_json::to_value(vec![name]).unwrap_or(serde_json::Value::Null),
);
}
_ => {}
}
}
fn parse_name_variable(value: &str) -> Name {
let trimmed = value.trim();
if let Some(sep_pos) = trimmed.find("||") {
#[allow(
clippy::string_slice,
reason = "sep_pos is a valid byte boundary for '||'"
)]
let family = trimmed[..sep_pos].trim().to_string();
#[allow(
clippy::string_slice,
reason = "sep_pos + 2 is a valid boundary after '||' (2-byte ASCII)"
)]
let given = trimmed[sep_pos + 2..].trim().to_string();
Name {
family: Some(family),
given: Some(given),
..Default::default()
}
} else {
Name {
literal: Some(trimmed.to_string()),
..Default::default()
}
}
}
fn is_string_variable(key: &str) -> bool {
matches!(
key,
"container-title"
| "collection-title"
| "volume-title"
| "event-title"
| "event-place"
| "event-location"
| "publisher"
| "publisher-place"
| "archive"
| "archive-place"
| "archive-location"
| "archive-collection"
| "archive_collection"
| "genre"
| "medium"
| "dimensions"
| "section"
| "volume"
| "issue"
| "page"
| "edition"
| "number"
| "number-of-volumes"
| "number-of-pages"
| "chapter-number"
| "part-number"
| "part-title"
| "supplement-number"
| "references"
| "source"
| "status"
| "DOI"
| "ISBN"
| "ISSN"
| "URL"
| "original-title"
| "reviewed-title"
| "reviewed-genre"
| "call-number"
| "abstract"
| "language"
| "jurisdiction"
| "authority"
| "citation-number"
| "citation-label"
| "annote"
| "keyword"
| "title-short"
| "collection-number"
)
}
#[allow(
clippy::too_many_lines,
clippy::cognitive_complexity,
reason = "match statement for all CSL string variable mappings"
)]
fn handle_string_variable(ref_obj: &mut Reference, key: &str, value: &str) {
let trimmed = value.trim();
match key {
"container-title" if ref_obj.container_title.is_none() => {
ref_obj.container_title = Some(trimmed.to_string());
}
"collection-title" if ref_obj.collection_title.is_none() => {
ref_obj.collection_title = Some(trimmed.to_string());
}
"collection-number" if ref_obj.collection_number.is_none() => {
ref_obj.collection_number = Some(StringOrNumber::String(trimmed.to_string()));
}
"part-title" => {
ref_obj.extra.insert(
key.to_string(),
serde_json::Value::String(trimmed.to_string()),
);
}
"publisher-place" if ref_obj.publisher_place.is_none() => {
ref_obj.publisher_place = Some(trimmed.to_string());
}
"publisher" if ref_obj.publisher.is_none() => {
ref_obj.publisher = Some(trimmed.to_string());
}
"archive-place" | "archive-location" if ref_obj.archive_location.is_none() => {
ref_obj.archive_location = Some(trimmed.to_string());
}
"archive" if ref_obj.archive.is_none() => {
ref_obj.archive = Some(trimmed.to_string());
}
"volume" if ref_obj.volume.is_none() => {
ref_obj.volume = Some(StringOrNumber::String(trimmed.to_string()));
}
"issue" if ref_obj.issue.is_none() => {
ref_obj.issue = Some(StringOrNumber::String(trimmed.to_string()));
}
"page" if ref_obj.page.is_none() => {
ref_obj.page = Some(trimmed.to_string());
}
"edition" if ref_obj.edition.is_none() => {
ref_obj.edition = Some(StringOrNumber::String(trimmed.to_string()));
}
"number-of-pages" if ref_obj.number_of_pages.is_none() => {
ref_obj.number_of_pages = Some(StringOrNumber::String(trimmed.to_string()));
}
"number-of-volumes" if ref_obj.number_of_volumes.is_none() => {
ref_obj.number_of_volumes = Some(StringOrNumber::String(trimmed.to_string()));
}
"chapter-number" if ref_obj.chapter_number.is_none() => {
ref_obj.chapter_number = Some(trimmed.to_string());
}
"genre" if ref_obj.genre.is_none() => {
ref_obj.genre = Some(trimmed.to_string());
}
"medium" if ref_obj.medium.is_none() => {
ref_obj.medium = Some(trimmed.to_string());
}
"language" if ref_obj.language.is_none() => {
ref_obj.language = Some(trimmed.to_string());
}
"original-title" if ref_obj.original_title.is_none() => {
ref_obj.original_title = Some(trimmed.to_string());
}
"abstract" if ref_obj.abstract_text.is_none() => {
ref_obj.abstract_text = Some(trimmed.to_string());
}
"DOI" if ref_obj.doi.is_none() => {
ref_obj.doi = Some(trimmed.to_string());
}
"ISBN" if ref_obj.isbn.is_none() => {
ref_obj.isbn = Some(trimmed.to_string());
}
"ISSN" if ref_obj.issn.is_none() => {
ref_obj.issn = Some(trimmed.to_string());
}
"URL" if ref_obj.url.is_none() => {
ref_obj.url = Some(trimmed.to_string());
}
"authority" if ref_obj.authority.is_none() => {
ref_obj.authority = Some(trimmed.to_string());
}
"section" if ref_obj.section.is_none() => {
ref_obj.section = Some(trimmed.to_string());
}
"number" if ref_obj.number.is_none() => {
ref_obj.number = Some(trimmed.to_string());
}
"volume-title" | "event-title" | "event-place" | "event-location"
| "archive-collection" | "archive_collection" | "dimensions" | "part-number"
| "supplement-number" | "references" | "source" | "status" | "reviewed-title"
| "reviewed-genre" | "call-number" | "jurisdiction" | "citation-number"
| "citation-label" | "annote" | "keyword" | "title-short" => {
let key_to_store = match key {
"archive_collection" => "archive-collection",
"event-location" => "event-place",
_ => key,
};
ref_obj.extra.insert(
key_to_store.to_string(),
serde_json::Value::String(trimmed.to_string()),
);
}
_ => {}
}
}
pub type Bibliography = indexmap::IndexMap<String, Reference>;
#[cfg(test)]
#[allow(
clippy::unwrap_used,
clippy::expect_used,
clippy::panic,
clippy::indexing_slicing,
clippy::todo,
clippy::unimplemented,
clippy::unreachable,
clippy::get_unwrap,
reason = "Panicking is acceptable and often desired in tests."
)]
mod tests {
use super::*;
#[test]
fn test_parse_csl_json() {
let json = r#"{
"id": "kuhn1962",
"type": "book",
"author": [{"family": "Kuhn", "given": "Thomas S."}],
"title": "The Structure of Scientific Revolutions",
"issued": {"date-parts": [[1962]]},
"publisher": "University of Chicago Press",
"publisher-place": "Chicago"
}"#;
let reference: Reference = serde_json::from_str(json).unwrap();
assert_eq!(reference.id, "kuhn1962");
assert_eq!(reference.ref_type, "book");
assert_eq!(
reference.author.as_ref().unwrap()[0].family,
Some("Kuhn".to_string())
);
assert_eq!(reference.issued.as_ref().unwrap().year_value(), Some(1962));
}
#[test]
fn test_date_variable() {
let date = DateVariable::year(2023);
assert_eq!(date.year_value(), Some(2023));
assert_eq!(date.month_value(), None);
let date = DateVariable::year_month(2023, 6);
assert_eq!(date.year_value(), Some(2023));
assert_eq!(date.month_value(), Some(6));
}
#[test]
fn test_parse_note_field_type_override() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("type: article-journal".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.ref_type, "article-journal");
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_string_variable() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("genre: H.R.\nstatus: enacted".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.genre, Some("H.R.".to_string()));
assert!(ref_obj.extra.contains_key("status"));
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_date() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("issued: 2020-03-15".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert!(ref_obj.issued.is_some());
let issued = ref_obj.issued.unwrap();
assert_eq!(issued.year_value(), Some(2020));
assert_eq!(issued.month_value(), Some(3));
assert_eq!(issued.day_value(), Some(15));
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_name() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("editor: Smith || John".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert!(ref_obj.editor.is_some());
let editors = ref_obj.editor.unwrap();
assert_eq!(editors.len(), 1);
assert_eq!(editors[0].family, Some("Smith".to_string()));
assert_eq!(editors[0].given, Some("John".to_string()));
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_preserves_existing() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
publisher: Some("OldPub".to_string()),
note: Some("publisher: NewPub".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.publisher, Some("OldPub".to_string()));
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_strips_parsed() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("genre: H.R.\nThis is an extra note line".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.genre, Some("H.R.".to_string()));
assert_eq!(ref_obj.note, Some("This is an extra note line".to_string()));
}
#[test]
fn test_parse_note_field_empty() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: None,
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_first_line_free_text() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("This is free text on first line\ngenre: H.R.\nstatus: enacted".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.genre, Some("H.R.".to_string()));
assert!(ref_obj.extra.contains_key("status"));
assert_eq!(
ref_obj.note,
Some("This is free text on first line".to_string())
);
}
#[test]
fn test_parse_note_field_date_range() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("issued: 2020/2021".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert!(ref_obj.issued.is_some());
let issued = ref_obj.issued.unwrap();
assert!(issued.date_parts.is_some());
let parts = issued.date_parts.unwrap();
assert_eq!(parts.len(), 2);
assert_eq!(parts[0][0], 2020);
assert_eq!(parts[1][0], 2021);
}
#[test]
fn test_parse_note_field_name_literal() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("author: United Nations".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert!(ref_obj.author.is_some());
let authors = ref_obj.author.unwrap();
assert_eq!(authors.len(), 1);
assert_eq!(authors[0].literal, Some("United Nations".to_string()));
}
#[test]
fn test_parse_note_field_uppercase_keys() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "article-journal".to_string(),
note: Some(
"DOI: 10.1000/xyz123\nISBN: 978-3-16-148410-0\nURL: https://example.org"
.to_string(),
),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.doi, Some("10.1000/xyz123".to_string()));
assert_eq!(ref_obj.isbn, Some("978-3-16-148410-0".to_string()));
assert_eq!(ref_obj.url, Some("https://example.org".to_string()));
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_unknown_key_preserved() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("genre: memoir\nfoo: bar\npublisher: MIT Press".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.genre, Some("memoir".to_string()));
assert_eq!(ref_obj.publisher, Some("MIT Press".to_string()));
assert_eq!(ref_obj.note, Some("foo: bar".to_string()));
}
#[test]
fn test_parse_note_field_archive_location() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("archive-location: Box 5, Folder 3".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(
ref_obj.archive_location,
Some("Box 5, Folder 3".to_string())
);
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_part_title_stored_in_extra() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "webpage".to_string(),
note: Some("part-title: Part title".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(
ref_obj.extra.get("part-title"),
Some(&serde_json::Value::String("Part title".to_string()))
);
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_recognized_keys_after_free_text() {
let mut ref_obj = Reference {
id: "broadcast-test".to_string(),
ref_type: "broadcast".to_string(),
note: Some(
"Some free-form description with no colon\ngenre: Documentary\nevent-place: United States".to_string(),
),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.genre, Some("Documentary".to_string()));
assert_eq!(
ref_obj
.extra
.get("event-place")
.map(|v| v.as_str().unwrap_or("")),
Some("United States"),
);
assert!(
ref_obj
.note
.as_deref()
.unwrap_or("")
.contains("Some free-form description")
);
}
#[test]
fn test_parse_note_field_recognized_keys_after_midnote_free_text() {
let mut ref_obj = Reference {
id: "bill-test".to_string(),
ref_type: "bill".to_string(),
note: Some(
"genre: H.R.\nsome unrecognized prose line here\nstatus: enacted".to_string(),
),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.genre, Some("H.R.".to_string()));
assert!(ref_obj.extra.contains_key("status"));
assert_eq!(
ref_obj.extra.get("status").and_then(|v| v.as_str()),
Some("enacted"),
);
assert!(
ref_obj
.note
.as_deref()
.unwrap_or("")
.contains("some unrecognized prose line")
);
}
}