scryer-engine 0.2.0

Tree-sitter and stack-graphs AST indexing engine for Scryer code intelligence
//! Deterministic tokenizer for code identifiers and prose.
//!
//! Identifiers are split on non-alphanumerics, camelCase / acronym boundaries and
//! letter↔digit boundaries (`HTTPServerError` → `http`, `server`, `error`), lowercased,
//! lightly normalised (trailing `s` stripped from tokens longer than 3 chars) and filtered
//! against a small stopword list of language keywords and common English words.

/// Language keywords (Rust, Python, TypeScript) and common English words that carry no
/// search signal. Must stay sorted: lookups use `binary_search`.
const STOPWORDS: &[&str] = &[
    "a",
    "an",
    "and",
    "are",
    "as",
    "async",
    "at",
    "await",
    "be",
    "break",
    "by",
    "class",
    "const",
    "continue",
    "crate",
    "def",
    "default",
    "dyn",
    "elif",
    "else",
    "enum",
    "export",
    "extends",
    "extern",
    "false",
    "fn",
    "for",
    "from",
    "function",
    "if",
    "impl",
    "implements",
    "import",
    "in",
    "interface",
    "is",
    "it",
    "lambda",
    "let",
    "loop",
    "match",
    "mod",
    "move",
    "mut",
    "new",
    "none",
    "not",
    "null",
    "of",
    "on",
    "or",
    "pass",
    "pub",
    "ref",
    "return",
    "self",
    "static",
    "struct",
    "super",
    "that",
    "the",
    "this",
    "to",
    "trait",
    "true",
    "type",
    "undefined",
    "unsafe",
    "use",
    "var",
    "void",
    "where",
    "while",
    "with",
    "yield",
];

/// Returns true if `token` (already lowercased) is a stopword.
pub fn is_stopword(token: &str) -> bool {
    STOPWORDS.binary_search(&token).is_ok()
}

/// Splits `text` into raw sub-words: on non-alphanumerics, camelCase, acronym and
/// letter↔digit boundaries. Case is preserved.
fn split_words(text: &str) -> Vec<&str> {
    let mut out = Vec::new();
    for segment in text.split(|c: char| !c.is_alphanumeric()) {
        if segment.is_empty() {
            continue;
        }
        let chars: Vec<(usize, char)> = segment.char_indices().collect();
        let mut start = 0;
        for i in 1..chars.len() {
            let (pos, cur) = chars[i];
            let prev = chars[i - 1].1;
            let next = chars.get(i + 1).map(|&(_, c)| c);
            let boundary = (prev.is_lowercase() && cur.is_uppercase())
                || (prev.is_alphabetic() && cur.is_ascii_digit())
                || (prev.is_ascii_digit() && cur.is_alphabetic())
                || (prev.is_uppercase()
                    && cur.is_uppercase()
                    && next.is_some_and(|n| n.is_lowercase()));
            if boundary {
                out.push(&segment[start..pos]);
                start = pos;
            }
        }
        out.push(&segment[start..]);
    }
    out
}

/// Lowercases and applies the light plural normalisation.
fn normalize(word: &str) -> String {
    let mut lower = word.to_lowercase();
    if lower.chars().count() > 3 && lower.ends_with('s') {
        lower.pop();
    }
    lower
}

/// Pushes the non-stopword, normalised sub-words of `text` onto `out`.
fn push_parts(text: &str, out: &mut Vec<String>) {
    for word in split_words(text) {
        let lower = word.to_lowercase();
        if is_stopword(&lower) {
            continue;
        }
        out.push(normalize(word));
    }
}

/// The whole identifier as a single lowercased token (identifier characters only).
fn whole_token(word: &str) -> Option<String> {
    let whole: String = word
        .chars()
        .filter(|c| c.is_alphanumeric() || *c == '_')
        .collect::<String>()
        .to_lowercase();
    (!whole.is_empty()).then_some(whole)
}

/// Document-side tokens for an identifier (e.g. a symbol name).
///
/// Emits the sub-words plus the whole lowercased identifier. The whole-identifier token is
/// never stopword-filtered, so a symbol named `default` stays retrievable by its own name.
pub fn tokenize_identifier(ident: &str) -> Vec<String> {
    let mut out = Vec::new();
    push_parts(ident, &mut out);
    if let Some(whole) = whole_token(ident)
        && !out.contains(&whole)
    {
        out.push(whole);
    }
    out
}

/// Document-side tokens for free text: signatures, docstrings, qualified paths, file paths.
pub fn tokenize_text(text: &str) -> Vec<String> {
    let mut out = Vec::new();
    push_parts(text, &mut out);
    out
}

/// Query-side tokens: deduplicated sub-words plus whole-word tokens for each
/// whitespace-separated query word, all stopword-filtered.
///
/// If stopword removal leaves nothing (e.g. `"impl"`, `"to the"`), falls back to the raw
/// lowercased split on non-alphanumerics with stopwords kept, so the query still searches.
pub fn tokenize_query(query: &str) -> Vec<String> {
    let mut out: Vec<String> = Vec::new();
    for word in query.split_whitespace() {
        let mut parts = Vec::new();
        push_parts(word, &mut parts);
        if let Some(whole) = whole_token(word)
            && !is_stopword(&whole)
        {
            parts.push(whole);
        }
        for p in parts {
            if !out.contains(&p) {
                out.push(p);
            }
        }
    }
    if out.is_empty() {
        for raw in query.split(|c: char| !c.is_alphanumeric()) {
            if raw.is_empty() {
                continue;
            }
            let lower = raw.to_lowercase();
            if !out.contains(&lower) {
                out.push(lower);
            }
        }
    }
    out
}

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn stopwords_are_sorted() {
        let mut sorted = STOPWORDS.to_vec();
        sorted.sort_unstable();
        assert_eq!(sorted, STOPWORDS);
    }

    #[test]
    fn splits_camel_case_and_acronyms() {
        assert_eq!(
            tokenize_identifier("HTTPServerError"),
            vec!["http", "server", "error", "httpservererror"]
        );
        assert_eq!(
            tokenize_identifier("parseJSONConfig"),
            vec!["parse", "json", "config", "parsejsonconfig"]
        );
    }

    #[test]
    fn splits_snake_case_and_drops_stopwords() {
        assert_eq!(
            tokenize_identifier("retry_with_backoff"),
            vec!["retry", "backoff", "retry_with_backoff"]
        );
    }

    #[test]
    fn splits_digits() {
        assert_eq!(tokenize_text("sha256Digest"), vec!["sha", "256", "digest"]);
        assert_eq!(tokenize_text("utf8"), vec!["utf", "8"]);
    }

    #[test]
    fn strips_trailing_s_on_long_tokens() {
        assert_eq!(tokenize_text("Edges bus"), vec!["edge", "bus"]);
    }

    #[test]
    fn text_tokens_drop_keywords() {
        assert_eq!(
            tokenize_text("pub fn enforce_limit(&self, payload: &str)"),
            vec!["enforce", "limit", "payload", "str"]
        );
    }

    #[test]
    fn whole_identifier_is_never_stopword_filtered() {
        assert_eq!(tokenize_identifier("default"), vec!["default"]);
        assert_eq!(tokenize_identifier("to"), vec!["to"]);
    }

    #[test]
    fn query_falls_back_when_all_stopwords() {
        assert_eq!(tokenize_query("impl"), vec!["impl"]);
        assert_eq!(tokenize_query("to the"), vec!["to", "the"]);
        assert_eq!(tokenize_query("default"), vec!["default"]);
    }

    #[test]
    fn query_drops_stopwords_when_others_remain() {
        assert_eq!(tokenize_query("the retry"), vec!["retry"]);
    }

    #[test]
    fn query_dedups_and_keeps_whole_identifier() {
        assert_eq!(
            tokenize_query("retry_with_backoff backoff"),
            vec!["retry", "backoff", "retry_with_backoff"]
        );
    }
}