string-analyze 0.1.1

Find key strings from cluttered text
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
//! 字符串匹配与启发式特征分析工具模块。
//!
//! 本模块提供了底层的高性能字符串搜索算法(子串、字符集、正则),
//! 以及针对特定编码(如 Base64、Hex)的启发式(Heuristics)校验函数。
//! 这些工具被广泛应用于扫描规则的执行流中。

#![allow(unused)]
use crate::State;
use regex::Regex;

/// 匹配目标抽象特征。
///
/// 允许底层匹配函数(如 `has_keyword`)既可以作为一个单纯的布尔检查器(传入 `&str`),
/// 也可以作为一个范围捕获器(传入 `&mut State`),从而避免重复编写匹配逻辑。
pub trait MatchTarget<'a> {
    /// 获取当前正在处理的完整字符串。
    fn get_input(&self) -> &'a str;
    /// 清空之前记录的所有匹配区间。
    fn clear_ranges(&mut self);
    /// 记录一个新的命中区间 `[start, end)`。
    fn push_range(&mut self, start: usize, end: usize);
}

// 针对只读字符串的空实现(仅用于布尔判断,丢弃捕获的区间)
impl<'a> MatchTarget<'a> for &'a str {
    #[inline]
    fn get_input(&self) -> &'a str {
        self
    }
    #[inline]
    fn clear_ranges(&mut self) {}
    #[inline]
    fn push_range(&mut self, _start: usize, _end: usize) {}
}

// 针对规则状态的实现(会将匹配到的区间真实记录到 State 中)
impl<'a> MatchTarget<'a> for &mut State<'a> {
    #[inline]
    fn get_input(&self) -> &'a str {
        self.input
    }
    #[inline]
    fn clear_ranges(&mut self) {
        self.ranges.clear();
    }
    #[inline]
    fn push_range(&mut self, start: usize, end: usize) {
        self.ranges.push((start, end));
    }
}

/// 扫描输入文本是否包含指定的子串关键词。
///
/// 匹配到的所有位置会被写入 `target`。
///
/// # 参数
/// - `target`: 匹配目标载体(`&str` 或 `&mut State`)。
/// - `keyword`: 要查找的子串。
/// - `ignore_case`: 是否忽略大小写(仅支持 ASCII 忽略大小写,以保证性能)。
#[inline]
pub fn has_keyword<'a>(mut target: impl MatchTarget<'a>, keyword: &str, ignore_case: bool) -> bool {
    target.clear_ranges();
    let input = target.get_input();
    let k_len = keyword.len();
    if k_len == 0 || input.len() < k_len {
        return false;
    }

    let mut found = false;

    if !ignore_case {
        let mut start = 0;
        while let Some(idx) = input[start..].find(keyword) {
            let abs_idx = start + idx;
            target.push_range(abs_idx, abs_idx + k_len);
            found = true;
            start = abs_idx + k_len;
        }
    } else {
        // 忽略大小写模式,采用 ASCII 窗口滑动匹配
        let input_bytes = input.as_bytes();
        let keyword_bytes = keyword.as_bytes();
        let mut i = 0;
        while i <= input_bytes.len() - k_len {
            let window = &input_bytes[i..i + k_len];
            if window.eq_ignore_ascii_case(keyword_bytes) {
                // 确保匹配边界在有效的 UTF-8 字符上
                if input.is_char_boundary(i) && input.is_char_boundary(i + k_len) {
                    target.push_range(i, i + k_len);
                    found = true;
                }
                i += k_len;
            } else {
                i += 1;
            }
        }
    }
    found
}

/// 扫描输入文本是否包含指定的字符集合。
///
/// # 参数
/// - `chars`: 需要查找的字符数组。
/// - `any`: 如果为 `true`,只要出现 `chars` 中的任意一个字符即算匹配成功;
///   如果为 `false`,则必须包含 `chars` 中的**所有**不同字符才算成功。
/// - `ignore_case`: 是否忽略 ASCII 大小写。
#[inline]
pub fn has_chars<'a>(
    mut target: impl MatchTarget<'a>,
    chars: &[char],
    any: bool,
    ignore_case: bool,
) -> bool {
    target.clear_ranges();
    let input = target.get_input();

    if chars.is_empty() || input.is_empty() {
        return false;
    }

    let mut target_map = [false; 256];
    let mut target_count = 0;

    // 1. 构建目标字符掩码表 (仅支持 ASCII)
    for &c in chars {
        if c.is_ascii() {
            let b = if ignore_case {
                c.to_ascii_lowercase() as usize
            } else {
                c as usize
            };
            if !target_map[b] {
                target_map[b] = true;
                target_count += 1; // 统计有多少种独立的目标字符
            }
        }
    }

    if target_count == 0 {
        return false;
    }

    let mut found_map = [false; 256];
    let mut distinct_found = 0;
    let mut any_found = false;

    // 2. 扫描输入字符串并记录命中的单字节位置
    for (i, b) in input.bytes().enumerate() {
        if !b.is_ascii() {
            continue;
        }
        let val = if ignore_case {
            b.to_ascii_lowercase() as usize
        } else {
            b as usize
        };

        if target_map[val] {
            // 标记当前命中字符的区间
            target.push_range(i, i + 1);
            any_found = true;

            // 如果要求包含所有指定字符 (!any),需记录当前找到了几种独立字符
            if !any && !found_map[val] {
                found_map[val] = true;
                distinct_found += 1;
            }
        }
    }

    // 3. 校验最终结果
    if any {
        any_found
    } else {
        let all_found = distinct_found == target_count;
        if !all_found {
            // 如果没有找齐所有指定的字符,视为失败,清空捕获区间
            target.clear_ranges();
        }
        all_found
    }
}

/// 执行正则表达式扫描,将所有命中区间记录进目标载体。
#[inline]
pub fn has_regex<'a>(mut target: impl MatchTarget<'a>, re: &Regex) -> bool {
    target.clear_ranges();
    let input = target.get_input();
    let mut found = false;

    for mat in re.find_iter(input) {
        target.push_range(mat.start(), mat.end());
        found = true;
    }
    found
}

/// 计算字符串中大写字母占所有字母(大小写)的比例。
///
/// 忽略非字母字符。如果字符串没有英文字母,返回 `0.0`。
#[inline]
pub fn upper_prob(input: &str) -> f64 {
    let mut upper_count = 0;
    let mut letter_count = 0;

    for b in input.bytes() {
        if b.is_ascii_uppercase() {
            upper_count += 1;
            letter_count += 1;
        } else if b.is_ascii_lowercase() {
            letter_count += 1;
        }
    }

    if letter_count == 0 {
        return 0.0;
    }
    (upper_count as f64) / (letter_count as f64)
}

/// 快速检查字符串是否包含 ASCII 大写字母。
#[inline]
pub fn has_upper(input: &str) -> bool {
    input.bytes().any(|b| b.is_ascii_uppercase())
}

/// 快速检查字符串是否包含 ASCII 小写字母。
#[inline]
pub fn has_lower(input: &str) -> bool {
    input.bytes().any(|b| b.is_ascii_lowercase())
}

/// 启发式判断字符串是否可能是一段真实的 Base64 编码数据 (支持 URL-Safe & 无填充格式)
///
/// # 过滤规则 (Heuristics)
/// 1. **长度限制**: 必须大于 8 个字符。
/// 2. **字节对齐**: 真实 Base64 剔除 '=' 后,长度对 4 取模绝对不能为 1。
/// 3. **填充符校验**: `=` 最多出现 2 次。
/// 4. **非法结尾**: 有效载荷不能以 `+`, `/`, `-`, `_` 结尾(基于编码填充位的数学特性)。
/// 5. **混入度校验**: 必须包含数字或特殊字符(`+`, `/`, `-`, `_`, `=`),防止将纯字母当成 Base64。
/// 6. **连续性检查**: 不能有连续 8 个大写或小写字母,防止英文长单词或全大写常量误报。
/// 7. **大小写比例**: 大写字母在总字母中的比例必须介于 `0.25` 到 `0.75` 之间。
pub fn is_base64(s: &str) -> bool {
    let len = s.len();

    // 1. 基础长度限制,降噪
    if len <= 8 {
        return false;
    }

    // 3. 去除填充符并检查
    let trimmed = s.trim_end_matches('=');
    let padding_count = len - trimmed.len();

    if padding_count > 2 {
        return false;
    }

    // 2. 字节对齐检查 (核心修改)
    // 无论是标准还是 URL-Safe,无 '=' 状态下的有效载荷长度 % 4 只能是 0, 2, 3。
    if trimmed.len() % 4 == 1 {
        return false;
    }

    // 4. Base64 有效载荷通常不能以特殊符号结尾
    // (结尾字符的低位必须是 0 作为隐式填充,因此对应字典表中的字符不可能是这几个)
    if trimmed.ends_with('/')
        || trimmed.ends_with('+')
        || trimmed.ends_with('-')
        || trimmed.ends_with('_')
    {
        return false;
    }

    let mut upper_count = 0;
    let mut letter_count = 0;

    let mut has_digit = false;
    let mut has_special = padding_count > 0;

    let mut consecutive_upper = 0;
    let mut consecutive_lower = 0;

    // 单次遍历优化 (O(N))
    for b in trimmed.bytes() {
        if b.is_ascii_uppercase() {
            upper_count += 1;
            letter_count += 1;
            consecutive_upper += 1;
            consecutive_lower = 0;

            if consecutive_upper >= 8 {
                return false;
            }
        } else if b.is_ascii_lowercase() {
            letter_count += 1;
            consecutive_lower += 1;
            consecutive_upper = 0;

            if consecutive_lower >= 8 {
                return false;
            }
        } else {
            // 打断连续性
            consecutive_upper = 0;
            consecutive_lower = 0;

            if b.is_ascii_digit() {
                has_digit = true;
            } else if b == b'+' || b == b'/' || b == b'-' || b == b'_' {
                // 将 - 和 _ 纳入特殊字符检测
                has_special = true;
            }
        }
    }

    // 5. 必须包含至少一个数字或特殊符号
    if !has_digit && !has_special {
        return false;
    }

    // 避免除零异常
    if letter_count == 0 {
        return false;
    }

    // 7. 校验大小写字母的分布比例
    let prob = (upper_count as f64) / (letter_count as f64);
    if prob <= 0.25 || prob >= 0.75 {
        return false;
    }

    true
}

/// 统计字符串中各 ASCII 字符的出现频率,并按降序排序。
///
/// # 返回值
/// 返回一个向量,元素为 `(字符, 出现次数)`,按次数降序排列,次数相同时按字符升序排列。
#[inline]
pub fn sort_ascii_counts(input: &str, ignore_case: bool) -> Vec<(char, usize)> {
    let mut counts = [0usize; 256];

    for b in input.bytes() {
        if b.is_ascii() {
            let val = if ignore_case {
                b.to_ascii_lowercase() as usize
            } else {
                b as usize
            };
            counts[val] += 1;
        }
    }

    let mut result: Vec<(char, usize)> = counts
        .iter()
        .enumerate()
        .filter(|&(_, &count)| count > 0)
        .map(|(b, &count)| (b as u8 as char, count))
        .collect();

    // 优先按次数降序,次数相同则按字符 ASCII 码升序,保证稳定性
    result.sort_unstable_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(&b.0)));

    result
}

/// 启发式判断字符串是否为一段连续或格式化的十六进制 (Hex) 数据。
///
/// 旨在精准提取代码或内存转储中的 Hex 序列,如 `\xAA\xBB`, `0x1a, 0x2b`, `AABBCC` 等。
///
/// # 过滤规则 (Heuristics)
/// 1. 过滤短文本,总长度需大于 10。
/// 2. 支持并提取常见前缀 (`0x`, `0X`, `\x`, `\X`, `%`)。
/// 3. **一致性检查**: 一个 Hex 串中的字母**不允许大小写混杂** (要么全是 `a-f`,要么全是 `A-F`)。
/// 4. **分隔符推导**: 如果字节之间有间隔,会自动推导分隔符(如空格、逗号等),并要求整个序列遵循该分隔符。
/// 5. 分隔符本身不能包含字母或数字,以防止错误截断。
pub fn is_hex(input: &str) -> bool {
    let mut s = input.trim();
    if s.is_empty() || !s.is_ascii() || s.len() <= 10 {
        return false;
    }

    // 处理全局仅带有一次前缀的情况 (例如 0xAABBCC),剥离前缀当做纯 Hex 处理
    if (s.starts_with("0x") || s.starts_with("0X") || s.starts_with("\\x") || s.starts_with("\\X"))
        && s[2..].find(&s[0..2]).is_none()
    {
        s = &s[2..];
    }

    // 提取重复性前缀
    let prefix = if s.starts_with("0x") {
        "0x"
    } else if s.starts_with("0X") {
        "0X"
    } else if s.starts_with("\\x") {
        "\\x"
    } else if s.starts_with("\\X") {
        "\\X"
    } else if s.starts_with("%") {
        "%"
    } else {
        ""
    };

    let mut idx = prefix.len();
    let bytes = s.as_bytes();

    if idx + 2 > bytes.len() {
        return false;
    }

    let mut upper_hex = false;
    let mut lower_hex = false;

    // 内部校验是否为有效的 Hex 字符,并记录大小写状态
    #[inline]
    fn check_hex(b: u8, upper: &mut bool, lower: &mut bool) -> bool {
        if b.is_ascii_digit() {
            return true;
        }
        if (b'a'..=b'f').contains(&b) {
            *lower = true;
            return true;
        }
        if (b'A'..=b'F').contains(&b) {
            *upper = true;
            return true;
        }
        false
    }

    if !check_hex(bytes[idx], &mut upper_hex, &mut lower_hex)
        || !check_hex(bytes[idx + 1], &mut upper_hex, &mut lower_hex)
    {
        return false;
    }
    idx += 2;

    // 推导字节之间的分隔符(如: 0xAA[分隔符]0xBB)
    let separator = if idx == bytes.len() {
        ""
    } else {
        if !prefix.is_empty() {
            if let Some(next_prefix_idx) = s[idx..].find(prefix) {
                &s[idx..idx + next_prefix_idx]
            } else {
                return false;
            }
        } else {
            let mut sep_len = 0;
            while idx + sep_len < bytes.len() {
                let b = bytes[idx + sep_len];
                if b.is_ascii_hexdigit() {
                    break;
                }
                sep_len += 1;
            }
            &s[idx..idx + sep_len]
        }
    };

    // 严禁分隔符包含字母和数字,避免解析错乱
    if !separator.is_empty() && separator.bytes().any(|b| b.is_ascii_alphanumeric()) {
        return false;
    }

    let mut curr_idx = 0;
    let mut count = 0;

    // 遍历循环验证整条数据链
    while curr_idx < bytes.len() {
        if !s[curr_idx..].starts_with(prefix) {
            return false;
        }
        curr_idx += prefix.len();

        if curr_idx + 2 > bytes.len() {
            return false;
        }

        if !check_hex(bytes[curr_idx], &mut upper_hex, &mut lower_hex)
            || !check_hex(bytes[curr_idx + 1], &mut upper_hex, &mut lower_hex)
        {
            return false;
        }

        // 核心规则:十六进制字符串中的字母大小写必须保持一致
        if upper_hex && lower_hex {
            return false;
        }

        curr_idx += 2;
        count += 1;

        if curr_idx == bytes.len() {
            break;
        }

        if !s[curr_idx..].starts_with(separator) {
            return false;
        }
        curr_idx += separator.len();

        if curr_idx == bytes.len() {
            return false; // 结尾不能带着悬空的分隔符
        }
    }
    count > 0
}

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn test() {
        has_keyword("", "", true);
        has_keyword(&mut State::new(""), "", true);
    }
}