Skip to main content

khive_text/
lang.rs

1//! Script/alphabet identification for per-language routing decisions.
2
3use std::collections::HashMap;
4
5/// Returns true if `c` falls within standard CJK Unicode blocks (Unified, Extension A/B, Compatibility, Hiragana, Katakana, Hangul).
6#[inline]
7pub fn is_cjk_char(c: char) -> bool {
8    matches!(c,
9        '\u{3040}'..='\u{309F}'     // Hiragana
10        | '\u{30A0}'..='\u{30FF}'   // Katakana
11        | '\u{3400}'..='\u{4DBF}'   // CJK Extension A
12        | '\u{4E00}'..='\u{9FFF}'   // CJK Unified Ideographs
13        | '\u{F900}'..='\u{FAFF}'   // CJK Compatibility Ideographs
14        | '\u{AC00}'..='\u{D7AF}'   // Hangul Syllables
15        | '\u{20000}'..='\u{2A6DF}' // CJK Extension B
16    )
17}
18
19/// Returns true when more than 15% of the characters in `text` are CJK.
20pub fn contains_cjk(text: &str) -> bool {
21    let chars: Vec<char> = text.chars().collect();
22    if chars.is_empty() {
23        return false;
24    }
25    let cjk_count = chars.iter().filter(|&&c| is_cjk_char(c)).count();
26    (cjk_count as f64 / chars.len() as f64) > 0.15
27}
28
29/// Lightweight script profile computed over a single string.
30#[derive(Debug, Clone, PartialEq)]
31pub struct ScriptProfile {
32    /// Fraction of characters that are CJK (0.0-1.0).
33    pub cjk_fraction: f64,
34    /// Fraction of characters that are ASCII letters (0.0-1.0).
35    pub latin_fraction: f64,
36    /// Total character count (not byte count).
37    pub char_count: usize,
38}
39
40impl ScriptProfile {
41    /// Analyze `text` and return a ScriptProfile.
42    pub fn analyze(text: &str) -> Self {
43        let chars: Vec<char> = text.chars().collect();
44        let n = chars.len();
45        if n == 0 {
46            return Self {
47                cjk_fraction: 0.0,
48                latin_fraction: 0.0,
49                char_count: 0,
50            };
51        }
52        let cjk = chars.iter().filter(|&&c| is_cjk_char(c)).count();
53        let latin = chars.iter().filter(|&&c| c.is_ascii_alphabetic()).count();
54        Self {
55            cjk_fraction: cjk as f64 / n as f64,
56            latin_fraction: latin as f64 / n as f64,
57            char_count: n,
58        }
59    }
60
61    /// True when CJK fraction exceeds 15%.
62    pub fn is_cjk_dominant(&self) -> bool {
63        self.cjk_fraction > 0.15
64    }
65}
66
67/// Returns true when `query` is worth sending to a retrieval backend.
68/// Rejects empty, symbol-only, single ASCII letter, and repeated-char (>80%) gibberish.
69pub fn is_meaningful_query(query: &str) -> bool {
70    let trimmed = query.trim();
71    if trimmed.is_empty() {
72        return false;
73    }
74
75    let non_ws: Vec<char> = trimmed.chars().filter(|c| !c.is_whitespace()).collect();
76    if non_ws.is_empty() {
77        return false;
78    }
79
80    // Symbol/punctuation/emoji-only queries are not meaningful, including Unicode symbols.
81    if !non_ws.iter().any(|c| c.is_alphanumeric()) {
82        return false;
83    }
84
85    // Single ASCII letter
86    if non_ws.len() == 1 && non_ws[0].is_ascii_alphabetic() {
87        return false;
88    }
89
90    // Repeated-char gibberish: dominant char > 80% of non-ws chars.
91    // Skip when total == 1 — a single character cannot exhibit gibberish repetition.
92    let total = non_ws.len();
93    if total > 1 {
94        let mut counts: HashMap<char, usize> = HashMap::new();
95        for c in &non_ws {
96            *counts.entry(*c).or_insert(0) += 1;
97        }
98        if let Some(&max_count) = counts.values().max() {
99            if max_count as f64 / total as f64 > 0.80 {
100                return false;
101            }
102        }
103    }
104
105    true
106}
107
108#[cfg(test)]
109mod tests {
110    use super::*;
111
112    #[test]
113    fn cjk_ideograph() {
114        assert!(is_cjk_char('使'));
115        assert!(is_cjk_char('用'));
116        assert!(is_cjk_char('进'));
117    }
118
119    #[test]
120    fn hiragana_katakana() {
121        assert!(is_cjk_char('あ'));
122        assert!(is_cjk_char('ア'));
123    }
124
125    #[test]
126    fn hangul() {
127        assert!(is_cjk_char('가'));
128    }
129
130    #[test]
131    fn latin_ascii_not_cjk() {
132        assert!(!is_cjk_char('a'));
133        assert!(!is_cjk_char('Z'));
134        assert!(!is_cjk_char('0'));
135        assert!(!is_cjk_char('-'));
136    }
137
138    #[test]
139    fn cjk_extension_a_boundary() {
140        assert!(is_cjk_char('\u{3400}')); // first Extension A
141        assert!(is_cjk_char('\u{4DBF}')); // last Extension A
142        assert!(!is_cjk_char('\u{33FF}')); // just before
143        assert!(!is_cjk_char('\u{4DC0}')); // just after Extension A
144    }
145
146    #[test]
147    fn unified_ideographs_boundary() {
148        assert!(is_cjk_char('\u{4E00}')); // first CJK Unified
149        assert!(is_cjk_char('\u{9FFF}')); // last CJK Unified
150        assert!(!is_cjk_char('\u{A000}')); // just after
151    }
152
153    #[test]
154    fn compatibility_ideographs_boundary() {
155        assert!(is_cjk_char('\u{F900}')); // first Compatibility
156        assert!(is_cjk_char('\u{FAFF}')); // last Compatibility
157        assert!(!is_cjk_char('\u{F8FF}')); // just before
158        assert!(!is_cjk_char('\u{FB00}')); // just after
159    }
160
161    #[test]
162    fn hiragana_boundary() {
163        assert!(is_cjk_char('\u{3040}')); // first Hiragana
164        assert!(is_cjk_char('\u{309F}')); // last Hiragana
165        assert!(!is_cjk_char('\u{303F}')); // just before
166    }
167
168    #[test]
169    fn katakana_boundary() {
170        assert!(is_cjk_char('\u{30A0}')); // first Katakana
171        assert!(is_cjk_char('\u{30FF}')); // last Katakana
172        assert!(!is_cjk_char('\u{3100}')); // just after
173    }
174
175    #[test]
176    fn hangul_boundary() {
177        assert!(is_cjk_char('\u{AC00}')); // first Hangul Syllable
178        assert!(is_cjk_char('\u{D7AF}')); // last Hangul Syllable
179        assert!(!is_cjk_char('\u{D7B0}')); // just after
180    }
181
182    // --- contains_cjk ---
183
184    #[test]
185    fn empty_string_no_cjk() {
186        assert!(!contains_cjk(""));
187    }
188
189    #[test]
190    fn all_latin_no_cjk() {
191        assert!(!contains_cjk("hello world"));
192    }
193
194    #[test]
195    fn all_cjk() {
196        assert!(contains_cjk("你好世界"));
197    }
198
199    #[test]
200    fn mixed_above_threshold() {
201        // 2 CJK in 5 chars = 40% > 15%
202        assert!(contains_cjk("abc你好"));
203    }
204
205    #[test]
206    fn mixed_below_threshold() {
207        // 1 CJK in 10 chars = 10% ≤ 15%
208        assert!(!contains_cjk("abcdefghi你"));
209    }
210
211    #[test]
212    fn exactly_15_percent_is_false() {
213        // 3 CJK in 20 chars = 15.0%, not > 15%
214        let text = "你好世abcdefghijklmnopq"; // 3 CJK + 17 latin = 20 chars
215        assert_eq!(text.chars().count(), 20);
216        assert!(!contains_cjk(text));
217    }
218
219    // --- ScriptProfile ---
220
221    #[test]
222    fn profile_pure_latin() {
223        let p = ScriptProfile::analyze("hello");
224        assert_eq!(p.char_count, 5);
225        assert_eq!(p.cjk_fraction, 0.0);
226        assert_eq!(p.latin_fraction, 1.0);
227        assert!(!p.is_cjk_dominant());
228    }
229
230    #[test]
231    fn profile_pure_cjk() {
232        let p = ScriptProfile::analyze("你好");
233        assert_eq!(p.char_count, 2);
234        assert_eq!(p.cjk_fraction, 1.0);
235        assert_eq!(p.latin_fraction, 0.0);
236        assert!(p.is_cjk_dominant());
237    }
238
239    #[test]
240    fn profile_empty() {
241        let p = ScriptProfile::analyze("");
242        assert_eq!(p.char_count, 0);
243        assert_eq!(p.cjk_fraction, 0.0);
244        assert!(!p.is_cjk_dominant());
245    }
246
247    #[test]
248    fn profile_mixed() {
249        // "hi你" — 3 chars: 2 latin, 1 CJK => 33% CJK > 15%
250        let p = ScriptProfile::analyze("hi你");
251        assert_eq!(p.char_count, 3);
252        assert!((p.cjk_fraction - 1.0 / 3.0).abs() < 1e-9);
253        assert!((p.latin_fraction - 2.0 / 3.0).abs() < 1e-9);
254        assert!(p.is_cjk_dominant());
255    }
256
257    // --- is_meaningful_query ---
258
259    #[test]
260    fn empty_not_meaningful() {
261        assert!(!is_meaningful_query(""));
262        assert!(!is_meaningful_query("   "));
263    }
264
265    #[test]
266    fn symbols_only_not_meaningful() {
267        assert!(!is_meaningful_query("!!!"));
268        assert!(!is_meaningful_query("@#$%"));
269        assert!(!is_meaningful_query("..."));
270    }
271
272    #[test]
273    fn single_latin_char_not_meaningful() {
274        assert!(!is_meaningful_query("a"));
275        assert!(!is_meaningful_query("Z"));
276    }
277
278    #[test]
279    fn repeated_char_gibberish_not_meaningful() {
280        assert!(!is_meaningful_query("aaaaaaa")); // 100%
281        assert!(!is_meaningful_query("aaaaab")); // 5/6 ≈ 83% > 80%
282    }
283
284    #[test]
285    fn repeated_char_below_threshold_is_meaningful() {
286        assert!(is_meaningful_query("aaab")); // 3/4 = 75% ≤ 80%
287    }
288
289    #[test]
290    fn normal_queries_are_meaningful() {
291        assert!(is_meaningful_query("rust programming"));
292        assert!(is_meaningful_query("你好世界"));
293        assert!(is_meaningful_query("BM25"));
294        assert!(is_meaningful_query("ab"));
295    }
296
297    #[test]
298    fn single_digit_is_meaningful() {
299        // Only single ASCII letter is blocked, not digit
300        assert!(is_meaningful_query("5"));
301    }
302
303    #[test]
304    fn unicode_symbol_only_queries_are_not_meaningful() {
305        assert!(!is_meaningful_query("\u{FF0C}")); // fullwidth comma
306        assert!(!is_meaningful_query("\u{3001}")); // ideographic comma
307        assert!(!is_meaningful_query("\u{FF01}\u{FF1F}")); // fullwidth ! ?
308        assert!(!is_meaningful_query("\u{1F600}")); // emoji-only
309
310        assert!(is_meaningful_query("\u{4F60}\u{597D}\u{4E16}\u{754C}")); // 你好世界
311        assert!(is_meaningful_query("BM25"));
312        assert!(is_meaningful_query("5"));
313    }
314}