#[derive(Debug, Clone, PartialEq, Eq)]
pub struct ActRef {
pub celex: String,
pub text: String,
}
pub fn cites_article_or_annex(prose: &str) -> bool {
let has_roman_annex = prose.match_indices("Annex ").any(|(i, _)| {
prose[i + "Annex ".len()..]
.chars()
.next()
.is_some_and(|c| "IVXLCDM".contains(c))
});
let has_article = ["Art. ", "Article "].iter().any(|marker| {
prose.match_indices(marker).any(|(i, _)| {
prose[i + marker.len()..]
.chars()
.next()
.is_some_and(|c| c.is_ascii_digit())
})
});
has_roman_annex || has_article
}
pub fn act_refs(prose: &str) -> Vec<ActRef> {
let bytes = prose.as_bytes();
let mut found = Vec::new();
let mut i = 0;
while i < bytes.len() {
if !bytes[i].is_ascii_digit() {
i += 1;
continue;
}
if i > 0 && matches!(bytes[i - 1], b'-' | b'.' | b',') {
while i < bytes.len() && bytes[i].is_ascii_digit() {
i += 1;
}
continue;
}
let first_start = i;
while i < bytes.len() && bytes[i].is_ascii_digit() {
i += 1;
}
let first_len = i - first_start;
let Ok(first) = prose[first_start..i].parse::<u32>() else {
continue;
};
if i >= bytes.len() || bytes[i] != b'/' {
continue;
}
let after_slash = i + 1;
i = after_slash;
while i < bytes.len() && bytes[i].is_ascii_digit() {
i += 1;
}
if i == after_slash {
continue;
}
let second_len = i - after_slash;
let Ok(second) = prose[after_slash..i].parse::<u32>() else {
continue;
};
if i + 1 < bytes.len() && matches!(bytes[i], b'-' | b'.') && bytes[i + 1].is_ascii_digit() {
continue;
}
let mut end = i;
let mut directive_by_form = false;
if i < bytes.len() && bytes[i] == b'/' {
let suffix_start = i + 1;
let mut j = suffix_start;
while j < bytes.len() && bytes[j].is_ascii_uppercase() {
j += 1;
}
if matches!(&prose[suffix_start..j], "EU" | "EC" | "EEC") {
directive_by_form = true;
end = j;
}
}
let is_year = |y: u32| (1950..=2099).contains(&y);
let (year, number) = if first_len == 4 && is_year(first) {
(first, second)
} else if second_len == 4 && is_year(second) {
(second, first)
} else {
i = end;
continue;
};
if number == 0 || number > 9999 {
i = end;
continue;
}
let kind = nearest_kind_word(&prose[..first_start]).unwrap_or(if directive_by_form {
'L'
} else {
'R'
});
found.push(ActRef {
celex: format!("3{year}{kind}{number:04}"),
text: prose[first_start..end].to_owned(),
});
i = end;
}
found
}
fn nearest_kind_word(before: &str) -> Option<char> {
const KINDS: [(&str, char); 4] = [
("directive", 'L'),
("regulation", 'R'),
("decision", 'D'),
("recommendation", 'H'),
];
let lowered = before.to_lowercase();
KINDS
.iter()
.filter_map(|(word, sector)| {
let position = word_start(&lowered, word)?;
(lowered.len().saturating_sub(position) <= 60).then_some((position, *sector))
})
.max_by_key(|(position, _)| *position)
.map(|(_, sector)| sector)
}
fn word_start(haystack: &str, word: &str) -> Option<usize> {
let mut search_end = haystack.len();
while let Some(position) = haystack[..search_end].rfind(word) {
let preceded_by_letter = haystack[..position]
.chars()
.next_back()
.is_some_and(|c| c.is_alphanumeric());
if !preceded_by_letter {
return Some(position);
}
search_end = position;
}
None
}