use std::collections::HashSet;
use std::sync::LazyLock;
use regex::Regex;
static TOKEN: LazyLock<Regex> = LazyLock::new(|| Regex::new(r"[\p{L}\p{N}][\p{L}\p{N}_.-]*").unwrap());
const STOPWORDS: &[&str] = &[
"a", "an", "and", "are", "as", "at", "be", "been", "but", "by", "did", "do", "does", "for", "from",
"had", "has", "have", "how", "i", "if", "in", "into", "is", "it", "its", "last", "me", "my", "of", "on",
"or", "our", "so", "that", "the", "their", "then", "there", "this", "time", "to", "was", "we", "were",
"what", "when", "where", "which", "who", "why", "will", "with", "you",
];
pub fn normalize_name(value: &str) -> String {
value.split_whitespace().collect::<Vec<_>>().join(" ").to_lowercase()
}
fn words_in(text: &str) -> impl Iterator<Item = String> + '_ {
TOKEN.find_iter(text).map(|m| m.as_str().trim_end_matches(['.', '-', '_']).to_lowercase())
}
pub fn query_words(text: &str) -> impl Iterator<Item = String> + '_ {
words_in(text).filter(|t| !t.is_empty() && !STOPWORDS.contains(&t.as_str()))
}
#[derive(Debug, Clone, Default, PartialEq)]
pub struct Query {
pub words: Vec<String>,
pub phrases: Vec<String>,
pub excluded: Vec<String>,
}
impl Query {
pub fn parse(text: &str) -> Self {
let mut q = Query::default();
let mut seen = HashSet::new();
let mut rest = text;
while let Some(start) = rest.find(|c: char| !c.is_whitespace()) {
rest = &rest[start..];
let negate = rest.starts_with('-') && rest.len() > 1;
let body = if negate { &rest[1..] } else { rest };
let (chunk, phrase, next) = if let Some(inner) = body.strip_prefix('"') {
let end = inner.find('"').unwrap_or(inner.len());
(&inner[..end], true, &inner[(end + 1).min(inner.len())..])
} else {
let end = body.find(char::is_whitespace).unwrap_or(body.len());
(&body[..end], false, &body[end..])
};
rest = next;
let tokens: Vec<String> =
if phrase { words_in(chunk).collect() } else { query_words(chunk).collect() };
if tokens.is_empty() {
continue;
}
if negate {
q.excluded.push(tokens.join(" "));
} else if phrase && tokens.len() > 1 {
q.phrases.push(tokens.join(" "));
} else {
q.words.extend(tokens.into_iter().filter(|t| seen.insert(t.clone())));
}
}
q
}
pub fn is_empty(&self) -> bool {
self.words.is_empty() && self.phrases.is_empty()
}
pub fn fts(&self) -> Option<String> {
let terms: Vec<String> =
self.words.iter().chain(&self.phrases).take(64).map(|t| format!("\"{t}\"")).collect();
(!terms.is_empty()).then(|| terms.join(" OR "))
}
pub fn not_fts(&self) -> Option<String> {
let terms: Vec<String> = self.excluded.iter().take(32).map(|t| format!("\"{t}\"")).collect();
(!terms.is_empty()).then(|| terms.join(" OR "))
}
pub fn highlight(&self) -> Vec<String> {
let mut out: Vec<String> = self.words.clone();
out.extend(self.phrases.iter().flat_map(|p| p.split(' ').map(str::to_string)));
out
}
}
fn fnv1a(parts: &[&str]) -> u64 {
let mut h: u64 = 0xcbf2_9ce4_8422_2325;
for (i, part) in parts.iter().enumerate() {
if i > 0 {
h ^= 0x1f;
h = h.wrapping_mul(0x100_0000_01b3);
}
for b in part.as_bytes() {
h ^= u64::from(*b);
h = h.wrapping_mul(0x100_0000_01b3);
}
}
h
}
pub fn group_token(key: &str) -> String {
format!("g{:016x}", fnv1a(&[key]))
}
pub fn facet_token(field: &str, value: &str) -> String {
format!("f{:016x}", fnv1a(&[field, &value.trim().to_lowercase()]))
}
pub fn truncate(value: &str, max_chars: usize) -> String {
let value = value.trim();
if value.chars().count() <= max_chars {
return value.to_string();
}
let cut: String = value.chars().take(max_chars.saturating_sub(1)).collect();
format!("{}\u{2026}", cut.trim_end())
}
static ISO: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"^(\d{4})[-/](\d{1,2})[-/](\d{1,2})(?:[T ](\d{1,2}):(\d{2})(?::(\d{2}))?(?:[.,]\d+)?\s*(?:Z|[+-]\d{2}:?\d{2}|UTC)?)?$").unwrap()
});
static US: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"(?i)^(\d{1,2})/(\d{1,2})/(\d{4}|\d{2})(?:[ T,]+(\d{1,2}):(\d{2})(?::(\d{2}))?\s*(AM|PM)?)?$")
.unwrap()
});
pub fn normalize_date(raw: &str) -> Option<String> {
let s = raw.trim();
let num = |c: Option<regex::Match<'_>>| c.and_then(|m| m.as_str().parse::<u32>().ok());
let fmt = |y: u32, mo: u32, d: u32, h: Option<u32>, mi: Option<u32>, sec: Option<u32>| {
if !(1..=12).contains(&mo) || !(1..=31).contains(&d) || y < 1000 {
return None;
}
Some(match (h, mi) {
(Some(h), Some(mi)) if h < 24 && mi < 60 => {
format!("{y:04}-{mo:02}-{d:02}T{h:02}:{mi:02}:{:02}", sec.unwrap_or(0).min(59))
}
_ => format!("{y:04}-{mo:02}-{d:02}"),
})
};
if let Some(c) = ISO.captures(s) {
return fmt(
num(c.get(1))?,
num(c.get(2))?,
num(c.get(3))?,
num(c.get(4)),
num(c.get(5)),
num(c.get(6)),
);
}
if let Some(c) = US.captures(s) {
let mut y = num(c.get(3))?;
if y < 100 {
y += if y < 70 { 2000 } else { 1900 };
}
let mut h = num(c.get(4));
if let (Some(hour), Some(ampm)) = (h, c.get(7)) {
let pm = ampm.as_str().eq_ignore_ascii_case("pm");
h = Some(match (hour % 12, pm) {
(x, true) => x + 12,
(x, false) => x,
});
}
return fmt(y, num(c.get(1))?, num(c.get(2))?, h, num(c.get(5)), num(c.get(6)));
}
if s.len() >= 9 && s.len() <= 13 && s.bytes().all(|b| b.is_ascii_digit()) {
let n: i64 = s.parse().ok()?;
let secs = if s.len() == 13 { n / 1000 } else { n };
return Some(format_unix(secs).trim_end_matches('Z').to_string());
}
None
}
pub fn normalize_bound(raw: &str) -> Option<String> {
let s = raw.trim();
let b = s.as_bytes();
let digits = |r: std::ops::Range<usize>| b[r].iter().all(u8::is_ascii_digit);
match b.len() {
4 if digits(0..4) => Some(s.to_string()),
7 if digits(0..4)
&& b[4] == b'-'
&& digits(5..7)
&& (1..=12).contains(&s[5..7].parse::<u32>().ok()?) =>
{
Some(s.to_string())
}
_ => normalize_date(s),
}
}
pub fn format_unix(secs: i64) -> String {
let (days, rem) = (secs.div_euclid(86_400), secs.rem_euclid(86_400));
let z = days + 719_468;
let era = z.div_euclid(146_097);
let doe = z - era * 146_097;
let yoe = (doe - doe / 1460 + doe / 36_524 - doe / 146_096) / 365;
let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
let mp = (5 * doy + 2) / 153;
let day = doy - (153 * mp + 2) / 5 + 1;
let month = if mp < 10 { mp + 3 } else { mp - 9 };
let year = yoe + era * 400 + i64::from(month <= 2);
format!("{year:04}-{month:02}-{day:02}T{:02}:{:02}:{:02}Z", rem / 3600, rem % 3600 / 60, rem % 60)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn query_parsing_handles_phrases_exclusions_and_syntax() {
let q = Query::parse(r#"Login LOOP on app-01 "reset link" -sso -"single sign-on" OR* NEAR"#);
assert_eq!(q.words, ["login", "loop", "app-01", "near"]);
assert_eq!(q.phrases, ["reset link"]);
assert_eq!(q.excluded, ["sso", "single sign-on"]);
assert_eq!(q.fts().unwrap(), r#""login" OR "loop" OR "app-01" OR "near" OR "reset link""#);
assert_eq!(q.not_fts().unwrap(), r#""sso" OR "single sign-on""#);
assert!(Query::parse("the the").is_empty());
assert!(Query::parse(r#""unterminated phrase"#).phrases.len() == 1);
}
#[test]
fn tokens_are_single_alnum_and_case_insensitive_for_facets() {
let g = group_token("web-01 east");
assert!(g.chars().all(|c| c.is_ascii_alphanumeric()));
assert_eq!(facet_token("status", " Closed "), facet_token("status", "closed"));
assert_ne!(facet_token("status", "closed"), facet_token("state", "closed"));
}
#[test]
fn dates_normalize_to_sortable_strings() {
let n = |s: &str| normalize_date(s);
assert_eq!(n("2024-03-11T14:20:00Z").as_deref(), Some("2024-03-11T14:20:00"));
assert_eq!(n("2024-03-11 14:20:05.123+02:00").as_deref(), Some("2024-03-11T14:20:05"));
assert_eq!(n("2024/3/1").as_deref(), Some("2024-03-01"));
assert_eq!(n("3/11/2024 2:05 PM").as_deref(), Some("2024-03-11T14:05:00"));
assert_eq!(n("12/1/24 12:30 AM").as_deref(), Some("2024-12-01T00:30:00"));
assert_eq!(n("1710166800").as_deref(), Some("2024-03-11T14:20:00"));
assert_eq!(n("1710166800000").as_deref(), Some("2024-03-11T14:20:00"));
assert_eq!(n("2024-13-01"), None);
assert_eq!(n("soon"), None);
}
#[test]
fn truncate_marks_cut() {
assert_eq!(truncate("abcdef", 4), "abc\u{2026}");
assert_eq!(truncate(" abc ", 4), "abc");
}
}