const STOPWORDS: &[&str] = &[
"a",
"an",
"and",
"are",
"as",
"async",
"at",
"await",
"be",
"break",
"by",
"class",
"const",
"continue",
"crate",
"def",
"default",
"dyn",
"elif",
"else",
"enum",
"export",
"extends",
"extern",
"false",
"fn",
"for",
"from",
"function",
"if",
"impl",
"implements",
"import",
"in",
"interface",
"is",
"it",
"lambda",
"let",
"loop",
"match",
"mod",
"move",
"mut",
"new",
"none",
"not",
"null",
"of",
"on",
"or",
"pass",
"pub",
"ref",
"return",
"self",
"static",
"struct",
"super",
"that",
"the",
"this",
"to",
"trait",
"true",
"type",
"undefined",
"unsafe",
"use",
"var",
"void",
"where",
"while",
"with",
"yield",
];
pub fn is_stopword(token: &str) -> bool {
STOPWORDS.binary_search(&token).is_ok()
}
fn split_words(text: &str) -> Vec<&str> {
let mut out = Vec::new();
for segment in text.split(|c: char| !c.is_alphanumeric()) {
if segment.is_empty() {
continue;
}
let chars: Vec<(usize, char)> = segment.char_indices().collect();
let mut start = 0;
for i in 1..chars.len() {
let (pos, cur) = chars[i];
let prev = chars[i - 1].1;
let next = chars.get(i + 1).map(|&(_, c)| c);
let boundary = (prev.is_lowercase() && cur.is_uppercase())
|| (prev.is_alphabetic() && cur.is_ascii_digit())
|| (prev.is_ascii_digit() && cur.is_alphabetic())
|| (prev.is_uppercase()
&& cur.is_uppercase()
&& next.is_some_and(|n| n.is_lowercase()));
if boundary {
out.push(&segment[start..pos]);
start = pos;
}
}
out.push(&segment[start..]);
}
out
}
fn normalize(word: &str) -> String {
let mut lower = word.to_lowercase();
if lower.chars().count() > 3 && lower.ends_with('s') {
lower.pop();
}
lower
}
fn push_parts(text: &str, out: &mut Vec<String>) {
for word in split_words(text) {
let lower = word.to_lowercase();
if is_stopword(&lower) {
continue;
}
out.push(normalize(word));
}
}
fn whole_token(word: &str) -> Option<String> {
let whole: String = word
.chars()
.filter(|c| c.is_alphanumeric() || *c == '_')
.collect::<String>()
.to_lowercase();
(!whole.is_empty()).then_some(whole)
}
pub fn tokenize_identifier(ident: &str) -> Vec<String> {
let mut out = Vec::new();
push_parts(ident, &mut out);
if let Some(whole) = whole_token(ident)
&& !out.contains(&whole)
{
out.push(whole);
}
out
}
pub fn tokenize_text(text: &str) -> Vec<String> {
let mut out = Vec::new();
push_parts(text, &mut out);
out
}
pub fn tokenize_query(query: &str) -> Vec<String> {
let mut out: Vec<String> = Vec::new();
for word in query.split_whitespace() {
let mut parts = Vec::new();
push_parts(word, &mut parts);
if let Some(whole) = whole_token(word)
&& !is_stopword(&whole)
{
parts.push(whole);
}
for p in parts {
if !out.contains(&p) {
out.push(p);
}
}
}
if out.is_empty() {
for raw in query.split(|c: char| !c.is_alphanumeric()) {
if raw.is_empty() {
continue;
}
let lower = raw.to_lowercase();
if !out.contains(&lower) {
out.push(lower);
}
}
}
out
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn stopwords_are_sorted() {
let mut sorted = STOPWORDS.to_vec();
sorted.sort_unstable();
assert_eq!(sorted, STOPWORDS);
}
#[test]
fn splits_camel_case_and_acronyms() {
assert_eq!(
tokenize_identifier("HTTPServerError"),
vec!["http", "server", "error", "httpservererror"]
);
assert_eq!(
tokenize_identifier("parseJSONConfig"),
vec!["parse", "json", "config", "parsejsonconfig"]
);
}
#[test]
fn splits_snake_case_and_drops_stopwords() {
assert_eq!(
tokenize_identifier("retry_with_backoff"),
vec!["retry", "backoff", "retry_with_backoff"]
);
}
#[test]
fn splits_digits() {
assert_eq!(tokenize_text("sha256Digest"), vec!["sha", "256", "digest"]);
assert_eq!(tokenize_text("utf8"), vec!["utf", "8"]);
}
#[test]
fn strips_trailing_s_on_long_tokens() {
assert_eq!(tokenize_text("Edges bus"), vec!["edge", "bus"]);
}
#[test]
fn text_tokens_drop_keywords() {
assert_eq!(
tokenize_text("pub fn enforce_limit(&self, payload: &str)"),
vec!["enforce", "limit", "payload", "str"]
);
}
#[test]
fn whole_identifier_is_never_stopword_filtered() {
assert_eq!(tokenize_identifier("default"), vec!["default"]);
assert_eq!(tokenize_identifier("to"), vec!["to"]);
}
#[test]
fn query_falls_back_when_all_stopwords() {
assert_eq!(tokenize_query("impl"), vec!["impl"]);
assert_eq!(tokenize_query("to the"), vec!["to", "the"]);
assert_eq!(tokenize_query("default"), vec!["default"]);
}
#[test]
fn query_drops_stopwords_when_others_remain() {
assert_eq!(tokenize_query("the retry"), vec!["retry"]);
}
#[test]
fn query_dedups_and_keeps_whole_identifier() {
assert_eq!(
tokenize_query("retry_with_backoff backoff"),
vec!["retry", "backoff", "retry_with_backoff"]
);
}
}