1use std::collections::HashSet;
4use std::sync::LazyLock;
5
6use regex::Regex;
7
8static TOKEN: LazyLock<Regex> = LazyLock::new(|| Regex::new(r"[\p{L}\p{N}][\p{L}\p{N}_.-]*").unwrap());
9
10const STOPWORDS: &[&str] = &[
12 "a", "an", "and", "are", "as", "at", "be", "been", "but", "by", "did", "do", "does", "for", "from",
13 "had", "has", "have", "how", "i", "if", "in", "into", "is", "it", "its", "last", "me", "my", "of", "on",
14 "or", "our", "so", "that", "the", "their", "then", "there", "this", "time", "to", "was", "we", "were",
15 "what", "when", "where", "which", "who", "why", "will", "with", "you",
16];
17
18pub fn normalize_name(value: &str) -> String {
20 value.split_whitespace().collect::<Vec<_>>().join(" ").to_lowercase()
21}
22
23fn words_in(text: &str) -> impl Iterator<Item = String> + '_ {
24 TOKEN.find_iter(text).map(|m| m.as_str().trim_end_matches(['.', '-', '_']).to_lowercase())
25}
26
27pub fn query_words(text: &str) -> impl Iterator<Item = String> + '_ {
29 words_in(text).filter(|t| !t.is_empty() && !STOPWORDS.contains(&t.as_str()))
30}
31
32#[derive(Debug, Clone, Default, PartialEq)]
36pub struct Query {
37 pub words: Vec<String>,
38 pub phrases: Vec<String>,
39 pub excluded: Vec<String>,
40}
41
42impl Query {
43 pub fn parse(text: &str) -> Self {
44 let mut q = Query::default();
45 let mut seen = HashSet::new();
46 let mut rest = text;
47 while let Some(start) = rest.find(|c: char| !c.is_whitespace()) {
48 rest = &rest[start..];
49 let negate = rest.starts_with('-') && rest.len() > 1;
50 let body = if negate { &rest[1..] } else { rest };
51 let (chunk, phrase, next) = if let Some(inner) = body.strip_prefix('"') {
52 let end = inner.find('"').unwrap_or(inner.len());
53 (&inner[..end], true, &inner[(end + 1).min(inner.len())..])
54 } else {
55 let end = body.find(char::is_whitespace).unwrap_or(body.len());
56 (&body[..end], false, &body[end..])
57 };
58 rest = next;
59 let tokens: Vec<String> =
60 if phrase { words_in(chunk).collect() } else { query_words(chunk).collect() };
61 if tokens.is_empty() {
62 continue;
63 }
64 if negate {
65 q.excluded.push(tokens.join(" "));
66 } else if phrase && tokens.len() > 1 {
67 q.phrases.push(tokens.join(" "));
68 } else {
69 q.words.extend(tokens.into_iter().filter(|t| seen.insert(t.clone())));
70 }
71 }
72 q
73 }
74
75 pub fn is_empty(&self) -> bool {
76 self.words.is_empty() && self.phrases.is_empty()
77 }
78
79 pub fn fts(&self) -> Option<String> {
81 let terms: Vec<String> =
82 self.words.iter().chain(&self.phrases).take(64).map(|t| format!("\"{t}\"")).collect();
83 (!terms.is_empty()).then(|| terms.join(" OR "))
84 }
85
86 pub fn not_fts(&self) -> Option<String> {
87 let terms: Vec<String> = self.excluded.iter().take(32).map(|t| format!("\"{t}\"")).collect();
88 (!terms.is_empty()).then(|| terms.join(" OR "))
89 }
90
91 pub fn highlight(&self) -> Vec<String> {
93 let mut out: Vec<String> = self.words.clone();
94 out.extend(self.phrases.iter().flat_map(|p| p.split(' ').map(str::to_string)));
95 out
96 }
97}
98
99fn fnv1a(parts: &[&str]) -> u64 {
100 let mut h: u64 = 0xcbf2_9ce4_8422_2325;
101 for (i, part) in parts.iter().enumerate() {
102 if i > 0 {
103 h ^= 0x1f;
104 h = h.wrapping_mul(0x100_0000_01b3);
105 }
106 for b in part.as_bytes() {
107 h ^= u64::from(*b);
108 h = h.wrapping_mul(0x100_0000_01b3);
109 }
110 }
111 h
112}
113
114pub fn group_token(key: &str) -> String {
116 format!("g{:016x}", fnv1a(&[key]))
117}
118
119pub fn facet_token(field: &str, value: &str) -> String {
121 format!("f{:016x}", fnv1a(&[field, &value.trim().to_lowercase()]))
122}
123
124pub fn truncate(value: &str, max_chars: usize) -> String {
126 let value = value.trim();
127 if value.chars().count() <= max_chars {
128 return value.to_string();
129 }
130 let cut: String = value.chars().take(max_chars.saturating_sub(1)).collect();
131 format!("{}\u{2026}", cut.trim_end())
132}
133
134static ISO: LazyLock<Regex> = LazyLock::new(|| {
135 Regex::new(r"^(\d{4})[-/](\d{1,2})[-/](\d{1,2})(?:[T ](\d{1,2}):(\d{2})(?::(\d{2}))?(?:[.,]\d+)?\s*(?:Z|[+-]\d{2}:?\d{2}|UTC)?)?$").unwrap()
136});
137static US: LazyLock<Regex> = LazyLock::new(|| {
138 Regex::new(r"(?i)^(\d{1,2})/(\d{1,2})/(\d{4}|\d{2})(?:[ T,]+(\d{1,2}):(\d{2})(?::(\d{2}))?\s*(AM|PM)?)?$")
139 .unwrap()
140});
141
142pub fn normalize_date(raw: &str) -> Option<String> {
146 let s = raw.trim();
147 let num = |c: Option<regex::Match<'_>>| c.and_then(|m| m.as_str().parse::<u32>().ok());
148 let fmt = |y: u32, mo: u32, d: u32, h: Option<u32>, mi: Option<u32>, sec: Option<u32>| {
149 if !(1..=12).contains(&mo) || !(1..=31).contains(&d) || y < 1000 {
150 return None;
151 }
152 Some(match (h, mi) {
153 (Some(h), Some(mi)) if h < 24 && mi < 60 => {
154 format!("{y:04}-{mo:02}-{d:02}T{h:02}:{mi:02}:{:02}", sec.unwrap_or(0).min(59))
155 }
156 _ => format!("{y:04}-{mo:02}-{d:02}"),
157 })
158 };
159 if let Some(c) = ISO.captures(s) {
160 return fmt(
161 num(c.get(1))?,
162 num(c.get(2))?,
163 num(c.get(3))?,
164 num(c.get(4)),
165 num(c.get(5)),
166 num(c.get(6)),
167 );
168 }
169 if let Some(c) = US.captures(s) {
170 let mut y = num(c.get(3))?;
171 if y < 100 {
172 y += if y < 70 { 2000 } else { 1900 };
173 }
174 let mut h = num(c.get(4));
175 if let (Some(hour), Some(ampm)) = (h, c.get(7)) {
176 let pm = ampm.as_str().eq_ignore_ascii_case("pm");
177 h = Some(match (hour % 12, pm) {
178 (x, true) => x + 12,
179 (x, false) => x,
180 });
181 }
182 return fmt(y, num(c.get(1))?, num(c.get(2))?, h, num(c.get(5)), num(c.get(6)));
183 }
184 if s.len() >= 9 && s.len() <= 13 && s.bytes().all(|b| b.is_ascii_digit()) {
185 let n: i64 = s.parse().ok()?;
186 let secs = if s.len() == 13 { n / 1000 } else { n };
187 return Some(format_unix(secs).trim_end_matches('Z').to_string());
188 }
189 None
190}
191
192pub fn normalize_bound(raw: &str) -> Option<String> {
195 let s = raw.trim();
196 let b = s.as_bytes();
197 let digits = |r: std::ops::Range<usize>| b[r].iter().all(u8::is_ascii_digit);
198 match b.len() {
199 4 if digits(0..4) => Some(s.to_string()),
200 7 if digits(0..4)
201 && b[4] == b'-'
202 && digits(5..7)
203 && (1..=12).contains(&s[5..7].parse::<u32>().ok()?) =>
204 {
205 Some(s.to_string())
206 }
207 _ => normalize_date(s),
208 }
209}
210
211pub fn format_unix(secs: i64) -> String {
213 let (days, rem) = (secs.div_euclid(86_400), secs.rem_euclid(86_400));
214 let z = days + 719_468;
215 let era = z.div_euclid(146_097);
216 let doe = z - era * 146_097;
217 let yoe = (doe - doe / 1460 + doe / 36_524 - doe / 146_096) / 365;
218 let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
219 let mp = (5 * doy + 2) / 153;
220 let day = doy - (153 * mp + 2) / 5 + 1;
221 let month = if mp < 10 { mp + 3 } else { mp - 9 };
222 let year = yoe + era * 400 + i64::from(month <= 2);
223 format!("{year:04}-{month:02}-{day:02}T{:02}:{:02}:{:02}Z", rem / 3600, rem % 3600 / 60, rem % 60)
224}
225
226#[cfg(test)]
227mod tests {
228 use super::*;
229
230 #[test]
231 fn query_parsing_handles_phrases_exclusions_and_syntax() {
232 let q = Query::parse(r#"Login LOOP on app-01 "reset link" -sso -"single sign-on" OR* NEAR"#);
233 assert_eq!(q.words, ["login", "loop", "app-01", "near"]);
234 assert_eq!(q.phrases, ["reset link"]);
235 assert_eq!(q.excluded, ["sso", "single sign-on"]);
236 assert_eq!(q.fts().unwrap(), r#""login" OR "loop" OR "app-01" OR "near" OR "reset link""#);
237 assert_eq!(q.not_fts().unwrap(), r#""sso" OR "single sign-on""#);
238 assert!(Query::parse("the the").is_empty());
239 assert!(Query::parse(r#""unterminated phrase"#).phrases.len() == 1);
240 }
241
242 #[test]
243 fn tokens_are_single_alnum_and_case_insensitive_for_facets() {
244 let g = group_token("web-01 east");
245 assert!(g.chars().all(|c| c.is_ascii_alphanumeric()));
246 assert_eq!(facet_token("status", " Closed "), facet_token("status", "closed"));
247 assert_ne!(facet_token("status", "closed"), facet_token("state", "closed"));
248 }
249
250 #[test]
251 fn dates_normalize_to_sortable_strings() {
252 let n = |s: &str| normalize_date(s);
253 assert_eq!(n("2024-03-11T14:20:00Z").as_deref(), Some("2024-03-11T14:20:00"));
254 assert_eq!(n("2024-03-11 14:20:05.123+02:00").as_deref(), Some("2024-03-11T14:20:05"));
255 assert_eq!(n("2024/3/1").as_deref(), Some("2024-03-01"));
256 assert_eq!(n("3/11/2024 2:05 PM").as_deref(), Some("2024-03-11T14:05:00"));
257 assert_eq!(n("12/1/24 12:30 AM").as_deref(), Some("2024-12-01T00:30:00"));
258 assert_eq!(n("1710166800").as_deref(), Some("2024-03-11T14:20:00"));
259 assert_eq!(n("1710166800000").as_deref(), Some("2024-03-11T14:20:00"));
260 assert_eq!(n("2024-13-01"), None);
261 assert_eq!(n("soon"), None);
262 }
263
264 #[test]
265 fn truncate_marks_cut() {
266 assert_eq!(truncate("abcdef", 4), "abc\u{2026}");
267 assert_eq!(truncate(" abc ", 4), "abc");
268 }
269}