use serde::{Deserialize, Deserializer, Serialize};
use std::collections::{HashMap, HashSet};
pub const CSL_TYPES: &[&str] = &[
"article",
"article-journal",
"article-magazine",
"article-newspaper",
"bill",
"book",
"broadcast",
"chapter",
"classic",
"collection",
"dataset",
"document",
"entry",
"entry-dictionary",
"entry-encyclopedia",
"event",
"figure",
"graphic",
"hearing",
"interview",
"legal_case",
"legislation",
"manuscript",
"map",
"motion_picture",
"musical_score",
"pamphlet",
"paper-conference",
"patent",
"performance",
"periodical",
"personal_communication",
"post",
"post-weblog",
"regulation",
"report",
"review",
"review-book",
"software",
"song",
"speech",
"standard",
"thesis",
"treaty",
"webpage",
];
pub const CSL_TYPE_EXTENSIONS: &[&str] = &[
"manual",
"presentation",
"personal-communication",
"legal-case",
"statute",
];
#[derive(Debug, Clone, Default, Deserialize, Serialize)]
#[serde(rename_all = "kebab-case")]
pub struct Reference {
pub id: String,
#[serde(rename = "type")]
pub ref_type: String,
#[serde(skip_serializing_if = "Option::is_none")]
pub author: Option<Vec<Name>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub editor: Option<Vec<Name>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub translator: Option<Vec<Name>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub recipient: Option<Vec<Name>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub director: Option<Vec<Name>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub contributor: Option<Vec<Name>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub interviewer: Option<Vec<Name>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub title: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub container_title: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub collection_title: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub collection_number: Option<StringOrNumber>,
#[serde(skip_serializing_if = "Option::is_none")]
pub issued: Option<DateVariable>,
#[serde(skip_serializing_if = "Option::is_none")]
pub accessed: Option<DateVariable>,
#[serde(skip_serializing_if = "Option::is_none")]
pub volume: Option<StringOrNumber>,
#[serde(skip_serializing_if = "Option::is_none")]
pub issue: Option<StringOrNumber>,
#[serde(skip_serializing_if = "Option::is_none")]
pub page: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub edition: Option<StringOrNumber>,
#[serde(skip_serializing_if = "Option::is_none")]
#[serde(rename = "DOI")]
pub doi: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
#[serde(rename = "URL")]
pub url: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
#[serde(rename = "ISBN")]
pub isbn: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
#[serde(rename = "ISSN")]
pub issn: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub publisher: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub publisher_place: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub authority: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub section: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub event: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub medium: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub dimensions: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub archive: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
#[serde(alias = "archive_location")]
pub archive_location: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub number: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub chapter_number: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub printing_number: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub genre: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub language: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub original_title: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
#[serde(rename = "abstract")]
pub abstract_text: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub note: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub number_of_pages: Option<StringOrNumber>,
#[serde(skip_serializing_if = "Option::is_none")]
pub number_of_volumes: Option<StringOrNumber>,
#[serde(flatten)]
pub extra: HashMap<String, serde_json::Value>,
}
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
#[serde(tag = "code", rename_all = "kebab-case")]
pub enum NoteFieldDiagnostic {
ConflictingSupplementaryIdentifier {
item_id: String,
identifier: String,
kept: String,
ignored: Vec<String>,
},
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Default)]
pub struct Name {
pub family: Option<String>,
pub given: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub literal: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub suffix: Option<String>,
#[serde(
rename = "dropping-particle",
alias = "dropping_particle",
skip_serializing_if = "Option::is_none"
)]
pub dropping_particle: Option<String>,
#[serde(
rename = "non-dropping-particle",
alias = "non_dropping_particle",
skip_serializing_if = "Option::is_none"
)]
pub non_dropping_particle: Option<String>,
}
impl Name {
#[must_use]
pub fn new(family: &str, given: &str) -> Self {
Self {
family: Some(family.to_string()),
given: Some(given.to_string()),
..Default::default()
}
}
#[must_use]
pub fn literal(name: &str) -> Self {
Self {
literal: Some(name.to_string()),
..Default::default()
}
}
#[must_use]
pub fn family_or_literal(&self) -> &str {
self.family
.as_deref()
.or(self.literal.as_deref())
.unwrap_or("")
}
}
#[derive(Debug, Clone, Default, Deserialize, Serialize)]
#[serde(rename_all = "kebab-case")]
pub struct DateVariable {
#[serde(
default,
skip_serializing_if = "Option::is_none",
deserialize_with = "deserialize_date_parts_opt"
)]
pub date_parts: Option<Vec<Vec<i32>>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub literal: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub raw: Option<String>,
#[serde(
default,
skip_serializing_if = "Option::is_none",
deserialize_with = "deserialize_season_opt"
)]
pub season: Option<i32>,
#[serde(skip_serializing_if = "Option::is_none")]
pub circa: Option<bool>,
}
#[derive(Debug, Deserialize)]
#[serde(untagged)]
enum IntOrString {
Int(i32),
String(String),
}
fn deserialize_date_parts_opt<'de, D>(deserializer: D) -> Result<Option<Vec<Vec<i32>>>, D::Error>
where
D: Deserializer<'de>,
{
let raw = Option::<Vec<Vec<IntOrString>>>::deserialize(deserializer)?;
raw.map(|rows| {
rows.into_iter()
.map(|row| {
row.into_iter()
.map(|value| match value {
IntOrString::Int(n) => Ok(n),
IntOrString::String(text) => text.parse::<i32>().map_err(|_| {
serde::de::Error::custom(format!(
"invalid date-parts component {:?}: expected integer or integer-like string",
text
))
}),
})
.collect::<Result<Vec<_>, D::Error>>()
})
.collect::<Result<Vec<_>, D::Error>>()
})
.transpose()
}
fn deserialize_season_opt<'de, D>(deserializer: D) -> Result<Option<i32>, D::Error>
where
D: Deserializer<'de>,
{
let raw = Option::<IntOrString>::deserialize(deserializer)?;
raw.map(|value| match value {
IntOrString::Int(n) => Ok(n),
IntOrString::String(text) => {
let normalized = text.trim().to_ascii_lowercase();
match normalized.as_str() {
"spring" => Ok(1),
"summer" => Ok(2),
"fall" | "autumn" => Ok(3),
"winter" => Ok(4),
_ => text.parse::<i32>().map_err(|_| {
serde::de::Error::custom(format!(
"invalid season {:?}: expected integer, integer-like string, or named season",
text
))
}),
}
}
})
.transpose()
}
impl DateVariable {
#[must_use]
pub fn year(year: i32) -> Self {
Self {
date_parts: Some(vec![vec![year]]),
..Default::default()
}
}
#[must_use]
pub fn year_month(year: i32, month: i32) -> Self {
Self {
date_parts: Some(vec![vec![year, month]]),
..Default::default()
}
}
#[must_use]
pub fn full(year: i32, month: i32, day: i32) -> Self {
Self {
date_parts: Some(vec![vec![year, month, day]]),
..Default::default()
}
}
#[must_use]
pub fn year_value(&self) -> Option<i32> {
self.date_parts
.as_ref()
.and_then(|parts| parts.first())
.and_then(|date| date.first())
.copied()
}
#[must_use]
pub fn month_value(&self) -> Option<i32> {
self.date_parts
.as_ref()
.and_then(|parts| parts.first())
.and_then(|date| date.get(1))
.copied()
}
#[must_use]
pub fn day_value(&self) -> Option<i32> {
self.date_parts
.as_ref()
.and_then(|parts| parts.first())
.and_then(|date| date.get(2))
.copied()
}
}
#[derive(Debug, Clone, Deserialize, Serialize)]
#[serde(untagged)]
pub enum StringOrNumber {
String(String),
Number(i64),
}
impl std::fmt::Display for StringOrNumber {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
match self {
Self::String(s) => write!(f, "{s}"),
Self::Number(n) => write!(f, "{n}"),
}
}
}
impl Reference {
pub fn parse_note_field_hacks(&mut self) {
let _ = self.parse_note_field_hacks_with_diagnostics();
}
#[must_use]
pub fn parse_note_field_hacks_with_diagnostics(&mut self) -> Vec<NoteFieldDiagnostic> {
let (mut direct_cstr, mut tex_cstr) = take_cstr_extra_values(&mut self.extra);
let note = match &self.note {
Some(n) if !n.is_empty() => n.clone(),
_ => {
return finish_cstr_extraction(self, direct_cstr, tex_cstr);
}
};
let lines: Vec<&str> = note.lines().collect();
if lines.is_empty() {
return finish_cstr_extraction(self, direct_cstr, tex_cstr);
}
let mut parsed_indices = HashSet::new();
for (idx, line) in lines.iter().enumerate() {
let trimmed = line.trim();
if trimmed.is_empty() {
continue;
}
if let Some((source, value)) = parse_cstr_key_value(trimmed) {
if !value.is_empty() {
match source {
CstrSource::Direct => direct_cstr.push(value.to_string()),
CstrSource::Tex => tex_cstr.push(value.to_string()),
}
}
parsed_indices.insert(idx);
continue;
}
let Some((key, value)) = parse_key_value(trimmed) else {
continue;
};
if key.eq_ignore_ascii_case("type") {
if CSL_TYPES.contains(&value) || CSL_TYPE_EXTENSIONS.contains(&value) {
self.ref_type = value.to_string();
parsed_indices.insert(idx);
}
} else if is_date_variable(key) {
handle_date_variable(self, key, value);
parsed_indices.insert(idx);
} else if is_name_variable(key) {
handle_name_variable(self, key, value);
parsed_indices.insert(idx);
} else if is_string_variable(key) {
handle_string_variable(self, key, value);
parsed_indices.insert(idx);
}
}
let remaining_lines: Vec<&str> = lines
.iter()
.enumerate()
.filter(|(idx, _)| !parsed_indices.contains(idx))
.map(|(_, line)| *line)
.collect();
if remaining_lines.is_empty() {
self.note = None;
} else {
self.note = Some(remaining_lines.join("\n"));
}
finish_cstr_extraction(self, direct_cstr, tex_cstr)
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
enum CstrSource {
Direct,
Tex,
}
fn parse_cstr_key_value(line: &str) -> Option<(CstrSource, &str)> {
let (key, value) = line.split_once(':')?;
let source = if key.eq_ignore_ascii_case("CSTR") {
CstrSource::Direct
} else if key.eq_ignore_ascii_case("tex.cstr") {
CstrSource::Tex
} else {
return None;
};
Some((source, value.trim()))
}
fn take_cstr_extra_values(
extra: &mut HashMap<String, serde_json::Value>,
) -> (Vec<String>, Vec<String>) {
let mut keys = extra
.keys()
.filter(|key| key.eq_ignore_ascii_case("CSTR") || key.eq_ignore_ascii_case("tex.cstr"))
.cloned()
.collect::<Vec<_>>();
keys.sort_by_key(|key| {
let precedence = if key == "CSTR" {
0
} else if key.eq_ignore_ascii_case("CSTR") {
1
} else {
2
};
(precedence, key.clone())
});
let mut direct = Vec::new();
let mut tex = Vec::new();
for key in keys {
let Some(value) = extra.remove(&key).and_then(|value| match value {
serde_json::Value::String(value) => Some(value),
_ => None,
}) else {
continue;
};
if value.trim().is_empty() {
continue;
}
if key.eq_ignore_ascii_case("CSTR") {
direct.push(value.trim().to_string());
} else {
tex.push(value.trim().to_string());
}
}
(direct, tex)
}
fn finish_cstr_extraction(
reference: &mut Reference,
direct: Vec<String>,
tex: Vec<String>,
) -> Vec<NoteFieldDiagnostic> {
let Some(kept) = direct.first().or_else(|| tex.first()).cloned() else {
return Vec::new();
};
reference
.extra
.insert("CSTR".to_string(), serde_json::Value::String(kept.clone()));
let mut ignored = direct
.iter()
.chain(&tex)
.filter(|value| *value != &kept)
.cloned()
.collect::<Vec<_>>();
ignored.sort();
ignored.dedup();
if ignored.is_empty() {
Vec::new()
} else {
vec![NoteFieldDiagnostic::ConflictingSupplementaryIdentifier {
item_id: reference.id.clone(),
identifier: "cstr".to_string(),
kept,
ignored,
}]
}
}
fn parse_key_value(line: &str) -> Option<(&str, &str)> {
if line.is_empty() {
return None;
}
let colon_pos = line.find(':')?;
#[allow(
clippy::string_slice,
reason = "colon_pos is found via find(':'), which is a valid 1-byte ASCII boundary"
)]
let key = &line[..colon_pos];
let first_char = key.chars().next()?;
if !first_char.is_ascii_alphabetic() {
return None;
}
if !key
.chars()
.skip(1)
.all(|c| c.is_ascii_alphanumeric() || c == '_' || c == '-')
{
return None;
}
#[allow(
clippy::string_slice,
reason = "colon_pos + 1 is a valid boundary after ':' (1-byte ASCII)"
)]
let value = line[colon_pos + 1..].trim();
Some((key, value))
}
fn is_date_variable(key: &str) -> bool {
matches!(
key,
"issued" | "event-date" | "original-date" | "available-date" | "accessed" | "submitted"
)
}
fn handle_date_variable(ref_obj: &mut Reference, key: &str, value: &str) {
let date = parse_date_variable(value);
match key {
"issued" if ref_obj.issued.is_none() || is_date_range(&date) => {
ref_obj.issued = Some(date);
}
"issued" if date.raw.is_some() => {
if let Some(raw) = date.raw {
ref_obj.extra.insert(
"issued-note-literal".to_string(),
serde_json::Value::String(raw),
);
}
}
"accessed" if ref_obj.accessed.is_none() => {
ref_obj.accessed = Some(date);
}
"event-date" | "original-date" | "available-date" | "submitted" => {
ref_obj.extra.insert(
key.to_string(),
serde_json::Value::String(value.to_string()),
);
}
_ => {}
}
}
fn is_date_range(date: &DateVariable) -> bool {
date.date_parts
.as_ref()
.is_some_and(|parts| parts.len() == 2)
}
fn parse_date_variable(value: &str) -> DateVariable {
let trimmed = value.trim();
let (sep_pos, sep_len) = if let Some(pos) = trimmed.find('/') {
(Some(pos), 1)
} else if let Some(pos) = trimmed.find('–') {
(Some(pos), '–'.len_utf8())
} else {
(None, 0)
};
if let Some(pos) = sep_pos {
#[allow(
clippy::string_slice,
reason = "pos is a valid byte boundary found via .find()"
)]
let start_str = trimmed[..pos].trim();
#[allow(
clippy::string_slice,
reason = "pos + sep_len is a valid byte boundary for '/' or '–'"
)]
let end_str = trimmed[pos + sep_len..].trim();
if let (Some(start_parts), Some(end_parts)) =
(parse_date_parts(start_str), parse_date_parts(end_str))
{
return DateVariable {
date_parts: Some(vec![start_parts, end_parts]),
..Default::default()
};
}
}
if let Some(parts) = parse_date_parts(trimmed) {
DateVariable {
date_parts: Some(vec![parts]),
..Default::default()
}
} else {
DateVariable {
raw: Some(trimmed.to_string()),
..Default::default()
}
}
}
fn parse_date_parts(s: &str) -> Option<Vec<i32>> {
let parts: Vec<&str> = s.split('-').collect();
if parts.is_empty() || parts.len() > 3 {
return None;
}
let mut result = Vec::new();
for part in parts {
result.push(part.trim().parse::<i32>().ok()?);
}
Some(result)
}
fn is_name_variable(key: &str) -> bool {
matches!(
key,
"editor"
| "translator"
| "interviewer"
| "director"
| "contributor"
| "recipient"
| "author"
| "collection-editor"
| "container-author"
| "editorial-director"
| "illustrator"
| "original-author"
| "reviewed-author"
| "composer"
| "narrator"
| "performer"
| "producer"
| "script-writer"
| "compiler"
)
}
fn handle_name_variable(ref_obj: &mut Reference, key: &str, value: &str) {
let name = parse_name_variable(value);
match key {
"author" if ref_obj.author.is_none() => {
ref_obj.author = Some(vec![name]);
}
"editor" if ref_obj.editor.is_none() => {
ref_obj.editor = Some(vec![name]);
}
"translator" if ref_obj.translator.is_none() => {
ref_obj.translator = Some(vec![name]);
}
"interviewer" if ref_obj.interviewer.is_none() => {
ref_obj.interviewer = Some(vec![name]);
}
"director" if ref_obj.director.is_none() => {
ref_obj.director = Some(vec![name]);
}
"contributor" if ref_obj.contributor.is_none() => {
ref_obj.contributor = Some(vec![name]);
}
"recipient" if ref_obj.recipient.is_none() => {
ref_obj.recipient = Some(vec![name]);
}
"collection-editor" | "container-author" | "editorial-director" | "illustrator"
| "original-author" | "reviewed-author" | "composer" | "narrator" | "performer"
| "producer" | "script-writer" | "compiler" => {
ref_obj.extra.insert(
key.to_string(),
serde_json::to_value(vec![name]).unwrap_or(serde_json::Value::Null),
);
}
_ => {}
}
}
fn parse_name_variable(value: &str) -> Name {
let trimmed = value.trim();
if let Some(sep_pos) = trimmed.find("||") {
#[allow(
clippy::string_slice,
reason = "sep_pos is a valid byte boundary for '||'"
)]
let family = trimmed[..sep_pos].trim().to_string();
#[allow(
clippy::string_slice,
reason = "sep_pos + 2 is a valid boundary after '||' (2-byte ASCII)"
)]
let given = trimmed[sep_pos + 2..].trim().to_string();
Name {
family: Some(family),
given: Some(given),
..Default::default()
}
} else {
Name {
literal: Some(trimmed.to_string()),
..Default::default()
}
}
}
fn is_string_variable(key: &str) -> bool {
matches!(
key,
"container-title"
| "collection-title"
| "volume-title"
| "event-title"
| "event-place"
| "event-location"
| "publisher"
| "publisher-place"
| "archive"
| "archive-place"
| "archive-location"
| "archive-collection"
| "archive_collection"
| "genre"
| "medium"
| "dimensions"
| "version"
| "section"
| "volume"
| "issue"
| "page"
| "edition"
| "number"
| "number-of-volumes"
| "number-of-pages"
| "chapter-number"
| "part-number"
| "part-title"
| "supplement-number"
| "references"
| "source"
| "status"
| "DOI"
| "ISBN"
| "ISSN"
| "URL"
| "original-title"
| "reviewed-title"
| "reviewed-genre"
| "call-number"
| "abstract"
| "language"
| "jurisdiction"
| "authority"
| "citation-number"
| "citation-label"
| "annote"
| "keyword"
| "title-short"
| "collection-number"
)
}
#[allow(
clippy::too_many_lines,
clippy::cognitive_complexity,
reason = "match statement for all CSL string variable mappings"
)]
fn handle_string_variable(ref_obj: &mut Reference, key: &str, value: &str) {
let trimmed = value.trim();
match key {
"container-title" if ref_obj.container_title.is_none() => {
ref_obj.container_title = Some(trimmed.to_string());
}
"collection-title" if ref_obj.collection_title.is_none() => {
ref_obj.collection_title = Some(trimmed.to_string());
}
"collection-number" if ref_obj.collection_number.is_none() => {
ref_obj.collection_number = Some(StringOrNumber::String(trimmed.to_string()));
}
"part-title" => {
ref_obj.extra.insert(
key.to_string(),
serde_json::Value::String(trimmed.to_string()),
);
}
"publisher-place" if ref_obj.publisher_place.is_none() => {
ref_obj.publisher_place = Some(trimmed.to_string());
}
"publisher" if ref_obj.publisher.is_none() => {
ref_obj.publisher = Some(trimmed.to_string());
}
"archive-location" if ref_obj.archive_location.is_none() => {
ref_obj.archive_location = Some(trimmed.to_string());
}
"archive-place" => {
ref_obj
.extra
.entry(key.to_string())
.or_insert_with(|| serde_json::Value::String(trimmed.to_string()));
}
"archive" if ref_obj.archive.is_none() => {
ref_obj.archive = Some(trimmed.to_string());
}
"volume" if ref_obj.volume.is_none() => {
ref_obj.volume = Some(StringOrNumber::String(trimmed.to_string()));
}
"issue" if ref_obj.issue.is_none() => {
ref_obj.issue = Some(StringOrNumber::String(trimmed.to_string()));
}
"page" if ref_obj.page.is_none() => {
ref_obj.page = Some(trimmed.to_string());
}
"edition" if ref_obj.edition.is_none() => {
ref_obj.edition = Some(StringOrNumber::String(trimmed.to_string()));
}
"number-of-pages" if ref_obj.number_of_pages.is_none() => {
ref_obj.number_of_pages = Some(StringOrNumber::String(trimmed.to_string()));
}
"number-of-volumes" if ref_obj.number_of_volumes.is_none() => {
ref_obj.number_of_volumes = Some(StringOrNumber::String(trimmed.to_string()));
}
"chapter-number" if ref_obj.chapter_number.is_none() => {
ref_obj.chapter_number = Some(trimmed.to_string());
}
"genre" if ref_obj.genre.is_none() => {
ref_obj.genre = Some(trimmed.to_string());
}
"medium" if ref_obj.medium.is_none() => {
ref_obj.medium = Some(trimmed.to_string());
}
"language" if ref_obj.language.is_none() => {
ref_obj.language = Some(trimmed.to_string());
}
"original-title" if ref_obj.original_title.is_none() => {
ref_obj.original_title = Some(trimmed.to_string());
}
"abstract" if ref_obj.abstract_text.is_none() => {
ref_obj.abstract_text = Some(trimmed.to_string());
}
"DOI" if ref_obj.doi.is_none() => {
ref_obj.doi = Some(trimmed.to_string());
}
"ISBN" if ref_obj.isbn.is_none() => {
ref_obj.isbn = Some(trimmed.to_string());
}
"ISSN" if ref_obj.issn.is_none() => {
ref_obj.issn = Some(trimmed.to_string());
}
"URL" if ref_obj.url.is_none() => {
ref_obj.url = Some(trimmed.to_string());
}
"authority" if ref_obj.authority.is_none() => {
ref_obj.authority = Some(trimmed.to_string());
}
"section" if ref_obj.section.is_none() => {
ref_obj.section = Some(trimmed.to_string());
}
"number" if ref_obj.number.is_none() => {
ref_obj.number = Some(trimmed.to_string());
}
"volume-title" | "event-title" | "event-place" | "event-location"
| "archive-collection" | "archive_collection" | "dimensions" | "part-number"
| "supplement-number" | "references" | "source" | "status" | "reviewed-title"
| "reviewed-genre" | "call-number" | "jurisdiction" | "citation-number"
| "citation-label" | "annote" | "keyword" | "title-short" | "version" => {
let key_to_store = match key {
"archive_collection" => "archive-collection",
"event-location" => "event-place",
_ => key,
};
ref_obj.extra.insert(
key_to_store.to_string(),
serde_json::Value::String(trimmed.to_string()),
);
}
_ => {}
}
}
pub type Bibliography = indexmap::IndexMap<String, Reference>;
#[cfg(test)]
#[allow(
clippy::unwrap_used,
clippy::expect_used,
clippy::panic,
clippy::indexing_slicing,
clippy::todo,
clippy::unimplemented,
clippy::unreachable,
clippy::get_unwrap,
reason = "Panicking is acceptable and often desired in tests."
)]
mod tests {
use super::*;
#[test]
fn test_parse_csl_json() {
let json = r#"{
"id": "kuhn1962",
"type": "book",
"author": [{"family": "Kuhn", "given": "Thomas S."}],
"title": "The Structure of Scientific Revolutions",
"issued": {"date-parts": [[1962]]},
"publisher": "University of Chicago Press",
"publisher-place": "Chicago"
}"#;
let reference: Reference = serde_json::from_str(json).unwrap();
assert_eq!(reference.id, "kuhn1962");
assert_eq!(reference.ref_type, "book");
assert_eq!(
reference.author.as_ref().unwrap()[0].family,
Some("Kuhn".to_string())
);
assert_eq!(reference.issued.as_ref().unwrap().year_value(), Some(1962));
}
#[test]
fn test_date_variable() {
let date = DateVariable::year(2023);
assert_eq!(date.year_value(), Some(2023));
assert_eq!(date.month_value(), None);
let date = DateVariable::year_month(2023, 6);
assert_eq!(date.year_value(), Some(2023));
assert_eq!(date.month_value(), Some(6));
}
#[test]
fn note_field_issued_range_overrides_lossy_top_level_date() {
let mut ref_obj = Reference {
id: "periodical-range".to_string(),
ref_type: "book".to_string(),
issued: Some(DateVariable::year(1957)),
note: Some("issued: 1957/1990".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(
ref_obj.issued.and_then(|date| date.date_parts),
Some(vec![vec![1957], vec![1990]])
);
}
#[test]
fn note_field_issued_literal_preserved_alongside_structured_date() {
let mut ref_obj = Reference {
id: "printing-year".to_string(),
ref_type: "book".to_string(),
issued: Some(DateVariable::year(1995)),
note: Some("issued: 1995印刷".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(
ref_obj.issued.and_then(|date| date.date_parts),
Some(vec![vec![1995]])
);
assert_eq!(
ref_obj.extra.get("issued-note-literal"),
Some(&serde_json::Value::String("1995印刷".to_string()))
);
}
#[test]
fn test_parse_note_field_type_override() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("type: article-journal".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.ref_type, "article-journal");
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_type_override_recognizes_extension_spelling() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "document".to_string(),
note: Some("type: legal-case".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.ref_type, "legal-case");
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_unrecognized_type_override_is_ignored() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("type: colection".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.ref_type, "book");
assert_eq!(ref_obj.note, Some("type: colection".to_string()));
}
#[test]
fn test_parse_note_field_string_variable() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("genre: H.R.\nstatus: enacted".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.genre, Some("H.R.".to_string()));
assert!(ref_obj.extra.contains_key("status"));
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_date() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("issued: 2020-03-15".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert!(ref_obj.issued.is_some());
let issued = ref_obj.issued.unwrap();
assert_eq!(issued.year_value(), Some(2020));
assert_eq!(issued.month_value(), Some(3));
assert_eq!(issued.day_value(), Some(15));
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_name() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("editor: Smith || John".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert!(ref_obj.editor.is_some());
let editors = ref_obj.editor.unwrap();
assert_eq!(editors.len(), 1);
assert_eq!(editors[0].family, Some("Smith".to_string()));
assert_eq!(editors[0].given, Some("John".to_string()));
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_preserves_existing() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
publisher: Some("OldPub".to_string()),
note: Some("publisher: NewPub".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.publisher, Some("OldPub".to_string()));
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_strips_parsed() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("genre: H.R.\nThis is an extra note line".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.genre, Some("H.R.".to_string()));
assert_eq!(ref_obj.note, Some("This is an extra note line".to_string()));
}
#[test]
fn test_parse_note_field_empty() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: None,
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_first_line_free_text() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("This is free text on first line\ngenre: H.R.\nstatus: enacted".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.genre, Some("H.R.".to_string()));
assert!(ref_obj.extra.contains_key("status"));
assert_eq!(
ref_obj.note,
Some("This is free text on first line".to_string())
);
}
#[test]
fn test_parse_note_field_date_range() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("issued: 2020/2021".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert!(ref_obj.issued.is_some());
let issued = ref_obj.issued.unwrap();
assert!(issued.date_parts.is_some());
let parts = issued.date_parts.unwrap();
assert_eq!(parts.len(), 2);
assert_eq!(parts[0][0], 2020);
assert_eq!(parts[1][0], 2021);
}
#[test]
fn test_parse_note_field_name_literal() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("author: United Nations".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert!(ref_obj.author.is_some());
let authors = ref_obj.author.unwrap();
assert_eq!(authors.len(), 1);
assert_eq!(authors[0].literal, Some("United Nations".to_string()));
}
#[test]
fn test_parse_note_field_uppercase_keys() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "article-journal".to_string(),
note: Some(
"DOI: 10.1000/xyz123\nISBN: 978-3-16-148410-0\nURL: https://example.org"
.to_string(),
),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.doi, Some("10.1000/xyz123".to_string()));
assert_eq!(ref_obj.isbn, Some("978-3-16-148410-0".to_string()));
assert_eq!(ref_obj.url, Some("https://example.org".to_string()));
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_unknown_key_preserved() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("genre: memoir\nfoo: bar\npublisher: MIT Press".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.genre, Some("memoir".to_string()));
assert_eq!(ref_obj.publisher, Some("MIT Press".to_string()));
assert_eq!(ref_obj.note, Some("foo: bar".to_string()));
}
#[test]
fn test_parse_note_field_archive_location() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "book".to_string(),
note: Some("archive-location: Box 5, Folder 3".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(
ref_obj.archive_location,
Some("Box 5, Folder 3".to_string())
);
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_part_title_stored_in_extra() {
let mut ref_obj = Reference {
id: "test".to_string(),
ref_type: "webpage".to_string(),
note: Some("part-title: Part title".to_string()),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(
ref_obj.extra.get("part-title"),
Some(&serde_json::Value::String("Part title".to_string()))
);
assert_eq!(ref_obj.note, None);
}
#[test]
fn test_parse_note_field_recognized_keys_after_free_text() {
let mut ref_obj = Reference {
id: "broadcast-test".to_string(),
ref_type: "broadcast".to_string(),
note: Some(
"Some free-form description with no colon\ngenre: Documentary\nevent-place: United States".to_string(),
),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.genre, Some("Documentary".to_string()));
assert_eq!(
ref_obj
.extra
.get("event-place")
.map(|v| v.as_str().unwrap_or("")),
Some("United States"),
);
assert!(
ref_obj
.note
.as_deref()
.unwrap_or("")
.contains("Some free-form description")
);
}
#[test]
fn test_parse_note_field_recognized_keys_after_midnote_free_text() {
let mut ref_obj = Reference {
id: "bill-test".to_string(),
ref_type: "bill".to_string(),
note: Some(
"genre: H.R.\nsome unrecognized prose line here\nstatus: enacted".to_string(),
),
..Default::default()
};
ref_obj.parse_note_field_hacks();
assert_eq!(ref_obj.genre, Some("H.R.".to_string()));
assert!(ref_obj.extra.contains_key("status"));
assert_eq!(
ref_obj.extra.get("status").and_then(|v| v.as_str()),
Some("enacted"),
);
assert!(
ref_obj
.note
.as_deref()
.unwrap_or("")
.contains("some unrecognized prose line")
);
}
#[test]
fn cstr_note_field_prefers_direct_value_and_reports_conflict() {
let mut reference = Reference {
id: "preprint-1".to_string(),
ref_type: "article".to_string(),
note: Some("tex.cstr: 32012.legacy\nCSTR: 32012.direct\nfree-form note".to_string()),
..Default::default()
};
let diagnostics = reference.parse_note_field_hacks_with_diagnostics();
assert_eq!(
reference
.extra
.get("CSTR")
.and_then(serde_json::Value::as_str),
Some("32012.direct")
);
assert_eq!(reference.note.as_deref(), Some("free-form note"));
assert_eq!(
diagnostics,
vec![NoteFieldDiagnostic::ConflictingSupplementaryIdentifier {
item_id: "preprint-1".to_string(),
identifier: "cstr".to_string(),
kept: "32012.direct".to_string(),
ignored: vec!["32012.legacy".to_string()],
}]
);
}
#[test]
fn tex_cstr_note_field_is_case_insensitive_and_canonicalized() {
let mut reference = Reference {
id: "preprint-2".to_string(),
ref_type: "article".to_string(),
note: Some("TeX.CsTr: 32012.tex".to_string()),
..Default::default()
};
let diagnostics = reference.parse_note_field_hacks_with_diagnostics();
assert!(diagnostics.is_empty());
assert_eq!(
reference
.extra
.get("CSTR")
.and_then(serde_json::Value::as_str),
Some("32012.tex")
);
assert!(reference.note.is_none());
}
#[test]
fn identical_cstr_spellings_are_quiet() {
let mut reference = Reference {
id: "preprint-3".to_string(),
ref_type: "article".to_string(),
note: Some("cstr: 32012.same\nTEX.CSTR: 32012.same".to_string()),
..Default::default()
};
let diagnostics = reference.parse_note_field_hacks_with_diagnostics();
assert!(diagnostics.is_empty());
assert_eq!(
reference
.extra
.get("CSTR")
.and_then(serde_json::Value::as_str),
Some("32012.same")
);
}
#[test]
fn unknown_dotted_extra_key_and_note_text_are_preserved() {
let mut reference = Reference {
id: "preprint-4".to_string(),
ref_type: "article".to_string(),
note: Some("custom.code: keep-me\nordinary note".to_string()),
..Default::default()
};
reference.parse_note_field_hacks();
assert_eq!(
reference.note.as_deref(),
Some("custom.code: keep-me\nordinary note")
);
assert!(!reference.extra.contains_key("CSTR"));
}
}