Skip to main content

leviathan/
text.rs

1//! Text helpers shared by the indexer and the query layer.
2
3use std::collections::HashSet;
4use std::sync::LazyLock;
5
6use regex::Regex;
7
8static TOKEN: LazyLock<Regex> = LazyLock::new(|| Regex::new(r"[\p{L}\p{N}][\p{L}\p{N}_.-]*").unwrap());
9
10/// Words that match nearly every record and only slow BM25 down.
11const STOPWORDS: &[&str] = &[
12    "a", "an", "and", "are", "as", "at", "be", "been", "but", "by", "did", "do", "does", "for", "from",
13    "had", "has", "have", "how", "i", "if", "in", "into", "is", "it", "its", "last", "me", "my", "of", "on",
14    "or", "our", "so", "that", "the", "their", "then", "there", "this", "time", "to", "was", "we", "were",
15    "what", "when", "where", "which", "who", "why", "will", "with", "you",
16];
17
18/// Lowercase and collapse whitespace: the group-name comparison key.
19pub fn normalize_name(value: &str) -> String {
20    value.split_whitespace().collect::<Vec<_>>().join(" ").to_lowercase()
21}
22
23fn words_in(text: &str) -> impl Iterator<Item = String> + '_ {
24    TOKEN.find_iter(text).map(|m| m.as_str().trim_end_matches(['.', '-', '_']).to_lowercase())
25}
26
27/// Lowercased searchable words, stopwords removed.
28pub fn query_words(text: &str) -> impl Iterator<Item = String> + '_ {
29    words_in(text).filter(|t| !t.is_empty() && !STOPWORDS.contains(&t.as_str()))
30}
31
32/// A free-text query: words (ranked, any may match), `"quoted phrases"`, and
33/// `-excluded` words or phrases. Everything is re-quoted for FTS5, so query
34/// syntax in user input is inert.
35#[derive(Debug, Clone, Default, PartialEq)]
36pub struct Query {
37    pub words: Vec<String>,
38    pub phrases: Vec<String>,
39    pub excluded: Vec<String>,
40}
41
42impl Query {
43    pub fn parse(text: &str) -> Self {
44        let mut q = Query::default();
45        let mut seen = HashSet::new();
46        let mut rest = text;
47        while let Some(start) = rest.find(|c: char| !c.is_whitespace()) {
48            rest = &rest[start..];
49            let negate = rest.starts_with('-') && rest.len() > 1;
50            let body = if negate { &rest[1..] } else { rest };
51            let (chunk, phrase, next) = if let Some(inner) = body.strip_prefix('"') {
52                let end = inner.find('"').unwrap_or(inner.len());
53                (&inner[..end], true, &inner[(end + 1).min(inner.len())..])
54            } else {
55                let end = body.find(char::is_whitespace).unwrap_or(body.len());
56                (&body[..end], false, &body[end..])
57            };
58            rest = next;
59            let tokens: Vec<String> =
60                if phrase { words_in(chunk).collect() } else { query_words(chunk).collect() };
61            if tokens.is_empty() {
62                continue;
63            }
64            if negate {
65                q.excluded.push(tokens.join(" "));
66            } else if phrase && tokens.len() > 1 {
67                q.phrases.push(tokens.join(" "));
68            } else {
69                q.words.extend(tokens.into_iter().filter(|t| seen.insert(t.clone())));
70            }
71        }
72        q
73    }
74
75    pub fn is_empty(&self) -> bool {
76        self.words.is_empty() && self.phrases.is_empty()
77    }
78
79    /// Positive FTS5 expression: any word or phrase.
80    pub fn fts(&self) -> Option<String> {
81        let terms: Vec<String> =
82            self.words.iter().chain(&self.phrases).take(64).map(|t| format!("\"{t}\"")).collect();
83        (!terms.is_empty()).then(|| terms.join(" OR "))
84    }
85
86    pub fn not_fts(&self) -> Option<String> {
87        let terms: Vec<String> = self.excluded.iter().take(32).map(|t| format!("\"{t}\"")).collect();
88        (!terms.is_empty()).then(|| terms.join(" OR "))
89    }
90
91    /// Words to highlight in snippets.
92    pub fn highlight(&self) -> Vec<String> {
93        let mut out: Vec<String> = self.words.clone();
94        out.extend(self.phrases.iter().flat_map(|p| p.split(' ').map(str::to_string)));
95        out
96    }
97}
98
99fn fnv1a(parts: &[&str]) -> u64 {
100    let mut h: u64 = 0xcbf2_9ce4_8422_2325;
101    for (i, part) in parts.iter().enumerate() {
102        if i > 0 {
103            h ^= 0x1f;
104            h = h.wrapping_mul(0x100_0000_01b3);
105        }
106        for b in part.as_bytes() {
107            h ^= u64::from(*b);
108            h = h.wrapping_mul(0x100_0000_01b3);
109        }
110    }
111    h
112}
113
114/// Single FTS token that stands for one group inside the index.
115pub fn group_token(key: &str) -> String {
116    format!("g{:016x}", fnv1a(&[key]))
117}
118
119/// Single FTS token for `field = value` (case-insensitive).
120pub fn facet_token(field: &str, value: &str) -> String {
121    format!("f{:016x}", fnv1a(&[field, &value.trim().to_lowercase()]))
122}
123
124/// Truncate on a char boundary, marking the cut with an ellipsis.
125pub fn truncate(value: &str, max_chars: usize) -> String {
126    let value = value.trim();
127    if value.chars().count() <= max_chars {
128        return value.to_string();
129    }
130    let cut: String = value.chars().take(max_chars.saturating_sub(1)).collect();
131    format!("{}\u{2026}", cut.trim_end())
132}
133
134static ISO: LazyLock<Regex> = LazyLock::new(|| {
135    Regex::new(r"^(\d{4})[-/](\d{1,2})[-/](\d{1,2})(?:[T ](\d{1,2}):(\d{2})(?::(\d{2}))?(?:[.,]\d+)?\s*(?:Z|[+-]\d{2}:?\d{2}|UTC)?)?$").unwrap()
136});
137static US: LazyLock<Regex> = LazyLock::new(|| {
138    Regex::new(r"(?i)^(\d{1,2})/(\d{1,2})/(\d{4}|\d{2})(?:[ T,]+(\d{1,2}):(\d{2})(?::(\d{2}))?\s*(AM|PM)?)?$")
139        .unwrap()
140});
141
142/// Sortable `YYYY-MM-DD[THH:MM:SS]` from common date spellings: ISO 8601 /
143/// RFC 3339 (offset ignored), `YYYY/MM/DD`, US `M/D/YYYY [h:mm[:ss] [AM|PM]]`,
144/// and Unix epoch seconds or milliseconds. `None` when unrecognized.
145pub fn normalize_date(raw: &str) -> Option<String> {
146    let s = raw.trim();
147    let num = |c: Option<regex::Match<'_>>| c.and_then(|m| m.as_str().parse::<u32>().ok());
148    let fmt = |y: u32, mo: u32, d: u32, h: Option<u32>, mi: Option<u32>, sec: Option<u32>| {
149        if !(1..=12).contains(&mo) || !(1..=31).contains(&d) || y < 1000 {
150            return None;
151        }
152        Some(match (h, mi) {
153            (Some(h), Some(mi)) if h < 24 && mi < 60 => {
154                format!("{y:04}-{mo:02}-{d:02}T{h:02}:{mi:02}:{:02}", sec.unwrap_or(0).min(59))
155            }
156            _ => format!("{y:04}-{mo:02}-{d:02}"),
157        })
158    };
159    if let Some(c) = ISO.captures(s) {
160        return fmt(
161            num(c.get(1))?,
162            num(c.get(2))?,
163            num(c.get(3))?,
164            num(c.get(4)),
165            num(c.get(5)),
166            num(c.get(6)),
167        );
168    }
169    if let Some(c) = US.captures(s) {
170        let mut y = num(c.get(3))?;
171        if y < 100 {
172            y += if y < 70 { 2000 } else { 1900 };
173        }
174        let mut h = num(c.get(4));
175        if let (Some(hour), Some(ampm)) = (h, c.get(7)) {
176            let pm = ampm.as_str().eq_ignore_ascii_case("pm");
177            h = Some(match (hour % 12, pm) {
178                (x, true) => x + 12,
179                (x, false) => x,
180            });
181        }
182        return fmt(y, num(c.get(1))?, num(c.get(2))?, h, num(c.get(5)), num(c.get(6)));
183    }
184    if s.len() >= 9 && s.len() <= 13 && s.bytes().all(|b| b.is_ascii_digit()) {
185        let n: i64 = s.parse().ok()?;
186        let secs = if s.len() == 13 { n / 1000 } else { n };
187        return Some(format_unix(secs).trim_end_matches('Z').to_string());
188    }
189    None
190}
191
192/// A `--since/--until` bound: any [`normalize_date`] input, or a bare year
193/// (`2024`) or month (`2024-03`).
194pub fn normalize_bound(raw: &str) -> Option<String> {
195    let s = raw.trim();
196    let b = s.as_bytes();
197    let digits = |r: std::ops::Range<usize>| b[r].iter().all(u8::is_ascii_digit);
198    match b.len() {
199        4 if digits(0..4) => Some(s.to_string()),
200        7 if digits(0..4)
201            && b[4] == b'-'
202            && digits(5..7)
203            && (1..=12).contains(&s[5..7].parse::<u32>().ok()?) =>
204        {
205            Some(s.to_string())
206        }
207        _ => normalize_date(s),
208    }
209}
210
211/// Unix seconds as `YYYY-MM-DDTHH:MM:SSZ` (Howard Hinnant's civil_from_days).
212pub fn format_unix(secs: i64) -> String {
213    let (days, rem) = (secs.div_euclid(86_400), secs.rem_euclid(86_400));
214    let z = days + 719_468;
215    let era = z.div_euclid(146_097);
216    let doe = z - era * 146_097;
217    let yoe = (doe - doe / 1460 + doe / 36_524 - doe / 146_096) / 365;
218    let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
219    let mp = (5 * doy + 2) / 153;
220    let day = doy - (153 * mp + 2) / 5 + 1;
221    let month = if mp < 10 { mp + 3 } else { mp - 9 };
222    let year = yoe + era * 400 + i64::from(month <= 2);
223    format!("{year:04}-{month:02}-{day:02}T{:02}:{:02}:{:02}Z", rem / 3600, rem % 3600 / 60, rem % 60)
224}
225
226#[cfg(test)]
227mod tests {
228    use super::*;
229
230    #[test]
231    fn query_parsing_handles_phrases_exclusions_and_syntax() {
232        let q = Query::parse(r#"Login LOOP on app-01 "reset link" -sso -"single sign-on" OR* NEAR"#);
233        assert_eq!(q.words, ["login", "loop", "app-01", "near"]);
234        assert_eq!(q.phrases, ["reset link"]);
235        assert_eq!(q.excluded, ["sso", "single sign-on"]);
236        assert_eq!(q.fts().unwrap(), r#""login" OR "loop" OR "app-01" OR "near" OR "reset link""#);
237        assert_eq!(q.not_fts().unwrap(), r#""sso" OR "single sign-on""#);
238        assert!(Query::parse("the the").is_empty());
239        assert!(Query::parse(r#""unterminated phrase"#).phrases.len() == 1);
240    }
241
242    #[test]
243    fn tokens_are_single_alnum_and_case_insensitive_for_facets() {
244        let g = group_token("web-01 east");
245        assert!(g.chars().all(|c| c.is_ascii_alphanumeric()));
246        assert_eq!(facet_token("status", " Closed "), facet_token("status", "closed"));
247        assert_ne!(facet_token("status", "closed"), facet_token("state", "closed"));
248    }
249
250    #[test]
251    fn dates_normalize_to_sortable_strings() {
252        let n = |s: &str| normalize_date(s);
253        assert_eq!(n("2024-03-11T14:20:00Z").as_deref(), Some("2024-03-11T14:20:00"));
254        assert_eq!(n("2024-03-11 14:20:05.123+02:00").as_deref(), Some("2024-03-11T14:20:05"));
255        assert_eq!(n("2024/3/1").as_deref(), Some("2024-03-01"));
256        assert_eq!(n("3/11/2024 2:05 PM").as_deref(), Some("2024-03-11T14:05:00"));
257        assert_eq!(n("12/1/24 12:30 AM").as_deref(), Some("2024-12-01T00:30:00"));
258        assert_eq!(n("1710166800").as_deref(), Some("2024-03-11T14:20:00"));
259        assert_eq!(n("1710166800000").as_deref(), Some("2024-03-11T14:20:00"));
260        assert_eq!(n("2024-13-01"), None);
261        assert_eq!(n("soon"), None);
262    }
263
264    #[test]
265    fn truncate_marks_cut() {
266        assert_eq!(truncate("abcdef", 4), "abc\u{2026}");
267        assert_eq!(truncate(" abc ", 4), "abc");
268    }
269}