Skip to main content

khive_text/
identifier.rs

1//! Script/identifier detection and splitting for code-aware tokenization.
2
3/// Return whether text has an ASCII letter and an identifier boundary but no whitespace.
4///
5/// See `crates/khive-text/docs/api/identifier-and-script.md`.
6pub fn is_identifier(text: &str) -> bool {
7    if text.chars().any(|c| c.is_whitespace()) {
8        return false;
9    }
10    if !text.chars().any(|c| c.is_ascii_alphabetic()) {
11        return false;
12    }
13
14    let chars: Vec<char> = text.chars().collect();
15    let n = chars.len();
16
17    // Single-char identifiers: just a letter, no boundary possible
18    if n < 2 {
19        return false;
20    }
21
22    // Check for explicit separator characters
23    for &c in &chars {
24        if matches!(c, '_' | '-' | '/' | '.' | ':') {
25            return true;
26        }
27    }
28
29    // Check for camelCase (lowercase followed by uppercase) or digit-letter boundaries
30    for i in 0..n - 1 {
31        let a = chars[i];
32        let b = chars[i + 1];
33        if (a.is_ascii_lowercase() && b.is_ascii_uppercase())
34            || (a.is_ascii_alphabetic() && b.is_ascii_digit())
35            || (a.is_ascii_digit() && b.is_ascii_alphabetic())
36        {
37            return true;
38        }
39    }
40
41    false
42}
43
44/// Split `text` on separators (`_`, `-`, `.`, `/`, `::`) and camelCase/digit boundaries,
45/// returning lowercase parts of at least `min_part_len` characters.
46///
47/// See `crates/khive-text/docs/api/identifier-and-script.md`.
48pub fn split_identifier(text: &str, min_part_len: usize) -> Vec<String> {
49    // Collect chars, then split on separators / boundaries
50    let chars: Vec<char> = text.chars().collect();
51    let mut parts: Vec<String> = Vec::new();
52    let mut current = String::new();
53
54    let n = chars.len();
55    let mut i = 0;
56    while i < n {
57        let c = chars[i];
58
59        // Skip separator characters
60        if matches!(c, '_' | '-' | '/' | '.') {
61            if !current.is_empty() {
62                parts.push(current.clone());
63                current.clear();
64            }
65            i += 1;
66            continue;
67        }
68
69        // Handle '::' as a two-char separator
70        if c == ':' && i + 1 < n && chars[i + 1] == ':' {
71            if !current.is_empty() {
72                parts.push(current.clone());
73                current.clear();
74            }
75            i += 2;
76            continue;
77        }
78        // Lone ':' — treat as separator too
79        if c == ':' {
80            if !current.is_empty() {
81                parts.push(current.clone());
82                current.clear();
83            }
84            i += 1;
85            continue;
86        }
87
88        // Detect camelCase: lower→upper boundary
89        if !current.is_empty() && c.is_ascii_uppercase() {
90            let prev = chars[i - 1];
91            if prev.is_ascii_lowercase() {
92                // e.g. camelCase: split before 'C'
93                parts.push(current.clone());
94                current.clear();
95            } else if prev.is_ascii_uppercase() && i + 1 < n && chars[i + 1].is_ascii_lowercase() {
96                // e.g. "GPTModel": push accumulated "GP", keep "T" for new part
97                parts.push(current.clone());
98                current.clear();
99            }
100        }
101
102        // Detect digit→letter and letter→digit transitions
103        if !current.is_empty() {
104            let prev = chars[i - 1];
105            if (prev.is_ascii_digit() && c.is_ascii_alphabetic())
106                || (prev.is_ascii_alphabetic() && c.is_ascii_digit())
107            {
108                parts.push(current.clone());
109                current.clear();
110            }
111        }
112
113        current.push(c);
114        i += 1;
115    }
116
117    if !current.is_empty() {
118        parts.push(current);
119    }
120
121    // Lowercase and filter by min_part_len (character count, not byte length)
122    let min_part_len = min_part_len.max(1);
123    parts
124        .into_iter()
125        .map(|p| p.to_lowercase())
126        .filter(|p| p.chars().count() >= min_part_len)
127        .collect()
128}
129
130#[cfg(test)]
131mod tests {
132    use super::*;
133
134    #[test]
135    fn camel_case_is_identifier() {
136        assert!(is_identifier("camelCase"));
137        assert!(is_identifier("myVariableName"));
138    }
139
140    #[test]
141    fn snake_case_is_identifier() {
142        assert!(is_identifier("snake_case"));
143        assert!(is_identifier("MY_CONST"));
144    }
145
146    #[test]
147    fn hyphenated_is_identifier() {
148        assert!(is_identifier("kebab-case"));
149        assert!(is_identifier("GPT-4"));
150    }
151
152    #[test]
153    fn path_separators_are_identifier() {
154        assert!(is_identifier("a/b"));
155        assert!(is_identifier("pkg::Type"));
156    }
157
158    #[test]
159    fn plain_word_is_not_identifier() {
160        // "plain" has no boundary — single run of lowercase, no separator
161        assert!(!is_identifier("plain"));
162    }
163
164    #[test]
165    fn has_whitespace_is_not_identifier() {
166        assert!(!is_identifier("hello world"));
167        assert!(!is_identifier("foo bar"));
168    }
169
170    #[test]
171    fn no_ascii_letter_not_identifier() {
172        assert!(!is_identifier("123"));
173        assert!(!is_identifier("_"));
174    }
175
176    #[test]
177    fn camel_case_split() {
178        assert_eq!(split_identifier("camelCase", 1), vec!["camel", "case"]);
179    }
180
181    #[test]
182    fn snake_case_split() {
183        assert_eq!(split_identifier("snake_case", 1), vec!["snake", "case"]);
184    }
185
186    #[test]
187    fn gpt4_split() {
188        assert_eq!(split_identifier("GPT-4", 1), vec!["gpt", "4"]);
189    }
190
191    #[test]
192    fn lora_split() {
193        assert_eq!(split_identifier("LoRA", 1), vec!["lo", "ra"]);
194    }
195
196    #[test]
197    fn plain_single_part() {
198        assert_eq!(split_identifier("plain", 1), vec!["plain"]);
199    }
200
201    #[test]
202    fn bm25f_split() {
203        // "BM25F": B+M (uppercase run), 2+5 (digits), F (letter)
204        let parts = split_identifier("BM25F", 1);
205        assert_eq!(parts, vec!["bm", "25", "f"]);
206    }
207
208    #[test]
209    fn double_colon_separator() {
210        assert_eq!(split_identifier("pkg::Type", 1), vec!["pkg", "type"]);
211    }
212
213    #[test]
214    fn dot_separator() {
215        assert_eq!(
216            split_identifier("com.example.Foo", 1),
217            vec!["com", "example", "foo"]
218        );
219    }
220
221    #[test]
222    fn slash_separator() {
223        assert_eq!(split_identifier("a/b/c", 1), vec!["a", "b", "c"]);
224    }
225
226    #[test]
227    fn min_part_len_filters_short() {
228        // "GPT-4": parts ["gpt", "4"] — "4" has len 1, filtered when min=2
229        let parts = split_identifier("GPT-4", 2);
230        assert_eq!(parts, vec!["gpt"]);
231    }
232
233    #[test]
234    fn split_identifier_filters_unicode_min_part_len_by_chars() {
235        // "\u{4F60}" (你) is 1 char but 3 UTF-8 bytes; must be filtered by
236        // char count, not byte length, when min_part_len == 2.
237        let parts = split_identifier("foo_\u{4F60}", 2);
238        assert_eq!(parts, vec!["foo"]);
239
240        // A two-character non-ASCII part is kept at min_part_len == 2.
241        let parts = split_identifier("foo_\u{4F60}\u{597D}", 2);
242        assert_eq!(parts, vec!["foo", "\u{4F60}\u{597D}"]);
243    }
244
245    #[test]
246    fn mixed_case_and_digits() {
247        // "v2Api" -> ["v", "2", "api"] with min=1; min=2 drops "v" and "2"
248        let parts = split_identifier("v2Api", 2);
249        assert_eq!(parts, vec!["api"]);
250    }
251
252    #[test]
253    fn all_uppercase_run_then_lower() {
254        // "XMLParser" -> "XML" split before "P" (uppercase before lowercase)
255        // => ["xml", "parser"]
256        let parts = split_identifier("XMLParser", 1);
257        assert_eq!(parts, vec!["xml", "parser"]);
258    }
259}