Skip to main content

khive_text/
lang.rs

1//! Script/alphabet identification for per-language routing decisions.
2
3use std::collections::HashMap;
4
5/// Returns true if `c` falls within standard CJK Unicode blocks (Unified, Extension A/B, Compatibility, Hiragana, Katakana, Hangul).
6#[inline]
7pub fn is_cjk_char(c: char) -> bool {
8    matches!(c,
9        '\u{3040}'..='\u{309F}'     // Hiragana
10        | '\u{30A0}'..='\u{30FF}'   // Katakana
11        | '\u{3400}'..='\u{4DBF}'   // CJK Extension A
12        | '\u{4E00}'..='\u{9FFF}'   // CJK Unified Ideographs
13        | '\u{F900}'..='\u{FAFF}'   // CJK Compatibility Ideographs
14        | '\u{AC00}'..='\u{D7AF}'   // Hangul Syllables
15        | '\u{20000}'..='\u{2A6DF}' // CJK Extension B
16    )
17}
18
19/// Returns true when more than 15% of the characters in `text` are CJK.
20pub fn contains_cjk(text: &str) -> bool {
21    let chars: Vec<char> = text.chars().collect();
22    if chars.is_empty() {
23        return false;
24    }
25    let cjk_count = chars.iter().filter(|&&c| is_cjk_char(c)).count();
26    (cjk_count as f64 / chars.len() as f64) > 0.15
27}
28
29/// Lightweight script profile computed over a single string.
30#[derive(Debug, Clone, PartialEq)]
31pub struct ScriptProfile {
32    /// Fraction of characters that are CJK (0.0-1.0).
33    pub cjk_fraction: f64,
34    /// Fraction of characters that are ASCII letters (0.0-1.0).
35    pub latin_fraction: f64,
36    /// Total character count (not byte count).
37    pub char_count: usize,
38}
39
40impl ScriptProfile {
41    /// Analyze `text` and return a ScriptProfile.
42    pub fn analyze(text: &str) -> Self {
43        let chars: Vec<char> = text.chars().collect();
44        let n = chars.len();
45        if n == 0 {
46            return Self {
47                cjk_fraction: 0.0,
48                latin_fraction: 0.0,
49                char_count: 0,
50            };
51        }
52        let cjk = chars.iter().filter(|&&c| is_cjk_char(c)).count();
53        let latin = chars.iter().filter(|&&c| c.is_ascii_alphabetic()).count();
54        Self {
55            cjk_fraction: cjk as f64 / n as f64,
56            latin_fraction: latin as f64 / n as f64,
57            char_count: n,
58        }
59    }
60
61    /// True when CJK fraction exceeds 15%.
62    pub fn is_cjk_dominant(&self) -> bool {
63        self.cjk_fraction > 0.15
64    }
65}
66
67/// Returns true when `query` is worth sending to a retrieval backend.
68/// Rejects empty, symbol-only, single ASCII letter, and repeated-char (>80%) gibberish.
69pub fn is_meaningful_query(query: &str) -> bool {
70    let trimmed = query.trim();
71    if trimmed.is_empty() {
72        return false;
73    }
74
75    let non_ws: Vec<char> = trimmed.chars().filter(|c| !c.is_whitespace()).collect();
76    if non_ws.is_empty() {
77        return false;
78    }
79
80    // All non-whitespace are ASCII but not alphanumeric => symbols only
81    if non_ws
82        .iter()
83        .all(|c| c.is_ascii() && !c.is_ascii_alphanumeric())
84    {
85        return false;
86    }
87
88    // Single ASCII letter
89    if non_ws.len() == 1 && non_ws[0].is_ascii_alphabetic() {
90        return false;
91    }
92
93    // Repeated-char gibberish: dominant char > 80% of non-ws chars.
94    // Skip when total == 1 — a single character cannot exhibit gibberish repetition.
95    let total = non_ws.len();
96    if total > 1 {
97        let mut counts: HashMap<char, usize> = HashMap::new();
98        for c in &non_ws {
99            *counts.entry(*c).or_insert(0) += 1;
100        }
101        if let Some(&max_count) = counts.values().max() {
102            if max_count as f64 / total as f64 > 0.80 {
103                return false;
104            }
105        }
106    }
107
108    true
109}
110
111#[cfg(test)]
112mod tests {
113    use super::*;
114
115    #[test]
116    fn cjk_ideograph() {
117        assert!(is_cjk_char('使'));
118        assert!(is_cjk_char('用'));
119        assert!(is_cjk_char('进'));
120    }
121
122    #[test]
123    fn hiragana_katakana() {
124        assert!(is_cjk_char('あ'));
125        assert!(is_cjk_char('ア'));
126    }
127
128    #[test]
129    fn hangul() {
130        assert!(is_cjk_char('가'));
131    }
132
133    #[test]
134    fn latin_ascii_not_cjk() {
135        assert!(!is_cjk_char('a'));
136        assert!(!is_cjk_char('Z'));
137        assert!(!is_cjk_char('0'));
138        assert!(!is_cjk_char('-'));
139    }
140
141    #[test]
142    fn cjk_extension_a_boundary() {
143        assert!(is_cjk_char('\u{3400}')); // first Extension A
144        assert!(is_cjk_char('\u{4DBF}')); // last Extension A
145        assert!(!is_cjk_char('\u{33FF}')); // just before
146        assert!(!is_cjk_char('\u{4DC0}')); // just after Extension A
147    }
148
149    #[test]
150    fn unified_ideographs_boundary() {
151        assert!(is_cjk_char('\u{4E00}')); // first CJK Unified
152        assert!(is_cjk_char('\u{9FFF}')); // last CJK Unified
153        assert!(!is_cjk_char('\u{A000}')); // just after
154    }
155
156    #[test]
157    fn compatibility_ideographs_boundary() {
158        assert!(is_cjk_char('\u{F900}')); // first Compatibility
159        assert!(is_cjk_char('\u{FAFF}')); // last Compatibility
160        assert!(!is_cjk_char('\u{F8FF}')); // just before
161        assert!(!is_cjk_char('\u{FB00}')); // just after
162    }
163
164    #[test]
165    fn hiragana_boundary() {
166        assert!(is_cjk_char('\u{3040}')); // first Hiragana
167        assert!(is_cjk_char('\u{309F}')); // last Hiragana
168        assert!(!is_cjk_char('\u{303F}')); // just before
169    }
170
171    #[test]
172    fn katakana_boundary() {
173        assert!(is_cjk_char('\u{30A0}')); // first Katakana
174        assert!(is_cjk_char('\u{30FF}')); // last Katakana
175        assert!(!is_cjk_char('\u{3100}')); // just after
176    }
177
178    #[test]
179    fn hangul_boundary() {
180        assert!(is_cjk_char('\u{AC00}')); // first Hangul Syllable
181        assert!(is_cjk_char('\u{D7AF}')); // last Hangul Syllable
182        assert!(!is_cjk_char('\u{D7B0}')); // just after
183    }
184
185    // --- contains_cjk ---
186
187    #[test]
188    fn empty_string_no_cjk() {
189        assert!(!contains_cjk(""));
190    }
191
192    #[test]
193    fn all_latin_no_cjk() {
194        assert!(!contains_cjk("hello world"));
195    }
196
197    #[test]
198    fn all_cjk() {
199        assert!(contains_cjk("你好世界"));
200    }
201
202    #[test]
203    fn mixed_above_threshold() {
204        // 2 CJK in 5 chars = 40% > 15%
205        assert!(contains_cjk("abc你好"));
206    }
207
208    #[test]
209    fn mixed_below_threshold() {
210        // 1 CJK in 10 chars = 10% ≤ 15%
211        assert!(!contains_cjk("abcdefghi你"));
212    }
213
214    #[test]
215    fn exactly_15_percent_is_false() {
216        // 3 CJK in 20 chars = 15.0%, not > 15%
217        let text = "你好世abcdefghijklmnopq"; // 3 CJK + 17 latin = 20 chars
218        assert_eq!(text.chars().count(), 20);
219        assert!(!contains_cjk(text));
220    }
221
222    // --- ScriptProfile ---
223
224    #[test]
225    fn profile_pure_latin() {
226        let p = ScriptProfile::analyze("hello");
227        assert_eq!(p.char_count, 5);
228        assert_eq!(p.cjk_fraction, 0.0);
229        assert_eq!(p.latin_fraction, 1.0);
230        assert!(!p.is_cjk_dominant());
231    }
232
233    #[test]
234    fn profile_pure_cjk() {
235        let p = ScriptProfile::analyze("你好");
236        assert_eq!(p.char_count, 2);
237        assert_eq!(p.cjk_fraction, 1.0);
238        assert_eq!(p.latin_fraction, 0.0);
239        assert!(p.is_cjk_dominant());
240    }
241
242    #[test]
243    fn profile_empty() {
244        let p = ScriptProfile::analyze("");
245        assert_eq!(p.char_count, 0);
246        assert_eq!(p.cjk_fraction, 0.0);
247        assert!(!p.is_cjk_dominant());
248    }
249
250    #[test]
251    fn profile_mixed() {
252        // "hi你" — 3 chars: 2 latin, 1 CJK => 33% CJK > 15%
253        let p = ScriptProfile::analyze("hi你");
254        assert_eq!(p.char_count, 3);
255        assert!((p.cjk_fraction - 1.0 / 3.0).abs() < 1e-9);
256        assert!((p.latin_fraction - 2.0 / 3.0).abs() < 1e-9);
257        assert!(p.is_cjk_dominant());
258    }
259
260    // --- is_meaningful_query ---
261
262    #[test]
263    fn empty_not_meaningful() {
264        assert!(!is_meaningful_query(""));
265        assert!(!is_meaningful_query("   "));
266    }
267
268    #[test]
269    fn symbols_only_not_meaningful() {
270        assert!(!is_meaningful_query("!!!"));
271        assert!(!is_meaningful_query("@#$%"));
272        assert!(!is_meaningful_query("..."));
273    }
274
275    #[test]
276    fn single_latin_char_not_meaningful() {
277        assert!(!is_meaningful_query("a"));
278        assert!(!is_meaningful_query("Z"));
279    }
280
281    #[test]
282    fn repeated_char_gibberish_not_meaningful() {
283        assert!(!is_meaningful_query("aaaaaaa")); // 100%
284        assert!(!is_meaningful_query("aaaaab")); // 5/6 ≈ 83% > 80%
285    }
286
287    #[test]
288    fn repeated_char_below_threshold_is_meaningful() {
289        assert!(is_meaningful_query("aaab")); // 3/4 = 75% ≤ 80%
290    }
291
292    #[test]
293    fn normal_queries_are_meaningful() {
294        assert!(is_meaningful_query("rust programming"));
295        assert!(is_meaningful_query("你好世界"));
296        assert!(is_meaningful_query("BM25"));
297        assert!(is_meaningful_query("ab"));
298    }
299
300    #[test]
301    fn single_digit_is_meaningful() {
302        // Only single ASCII letter is blocked, not digit
303        assert!(is_meaningful_query("5"));
304    }
305}