use std::sync::OnceLock;
use regex::Regex;
#[must_use]
pub fn is_js_whitespace(c: char) -> bool {
matches!(
c,
'\t' | '\n' | '\u{0B}' | '\u{0C}' | '\r' | ' ' | '\u{00A0}' | '\u{1680}' | '\u{2000}'
..='\u{200A}'
| '\u{2028}'
| '\u{2029}'
| '\u{202F}'
| '\u{205F}'
| '\u{3000}'
| '\u{FEFF}'
)
}
pub fn split_ws(text: &str) -> impl Iterator<Item = &str> {
text.split(is_js_whitespace).filter(|w| !w.is_empty())
}
#[must_use]
pub fn has_word_char(s: &str) -> bool {
static RE: OnceLock<Regex> = OnceLock::new();
RE.get_or_init(|| Regex::new(r"[\p{L}\p{N}]").expect("valid regex"))
.is_match(s)
}
fn quote_token(token: &str) -> String {
format!("\"{}\"", token.replace('"', "\"\""))
}
#[must_use]
pub fn sanitize_fts_query(input: &str) -> String {
let chars: Vec<char> = input.chars().collect();
let n = chars.len();
let mut out: Vec<String> = Vec::new();
let mut i = 0;
while i < n {
let c = chars[i];
if is_js_whitespace(c) {
i += 1;
continue;
}
if c == '"' {
i += 1;
let start = i;
while i < n && chars[i] != '"' {
i += 1;
}
let phrase: String = chars[start..i].iter().collect();
if i < n {
i += 1; }
if has_word_char(&phrase) {
out.push(quote_token(&phrase));
}
continue;
}
let start = i;
while i < n && !is_js_whitespace(chars[i]) && chars[i] != '"' {
i += 1;
}
let word: String = chars[start..i].iter().collect();
let (core, prefix) = match word.strip_suffix('*') {
Some(core) => (core.to_owned(), true),
None => (word, false),
};
if has_word_char(&core) {
let mut t = quote_token(&core);
if prefix {
t.push('*');
}
out.push(t);
}
}
out.join(" ")
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn readme_table() {
assert_eq!(
sanitize_fts_query("guides/onboarding"),
"\"guides/onboarding\""
);
assert_eq!(
sanitize_fts_query("\"exact phrase\" other*"),
"\"exact phrase\" \"other\"*"
);
assert_eq!(sanitize_fts_query("---"), "");
assert_eq!(sanitize_fts_query(""), "");
assert_eq!(sanitize_fts_query(" \t\n"), "");
}
#[test]
fn phrases_and_quotes() {
assert_eq!(sanitize_fts_query("\"open phrase"), "\"open phrase\"");
assert_eq!(
sanitize_fts_query("ab\"cd ef\"gh"),
"\"ab\" \"cd ef\" \"gh\""
);
assert_eq!(sanitize_fts_query("\"\" \"--\" x"), "\"x\"");
assert_eq!(sanitize_fts_query("\"a b*\""), "\"a b*\"");
}
#[test]
fn prefix_and_word_chars() {
assert_eq!(sanitize_fts_query("foo*"), "\"foo\"*");
assert_eq!(sanitize_fts_query("* ** ***"), "");
assert_eq!(sanitize_fts_query("foo**"), "\"foo*\"*");
assert_eq!(sanitize_fts_query("héllo 9 ¿?"), "\"héllo\" \"9\"");
assert_eq!(
sanitize_fts_query("AND OR NOT:x (y)"),
"\"AND\" \"OR\" \"NOT:x\" \"(y)\""
);
}
#[test]
fn js_whitespace_set() {
assert!(is_js_whitespace('\u{FEFF}'));
assert!(is_js_whitespace('\u{00A0}'));
assert!(is_js_whitespace('\u{3000}'));
assert!(!is_js_whitespace('\u{0085}'));
assert!(!is_js_whitespace('\u{200B}'));
assert_eq!(
sanitize_fts_query("a\u{00A0}b\u{FEFF}c"),
"\"a\" \"b\" \"c\""
);
assert_eq!(sanitize_fts_query("a\u{0085}b"), "\"a\u{0085}b\"");
assert_eq!(
split_ws(" a b\u{3000}c ").collect::<Vec<_>>(),
["a", "b", "c"]
);
}
}