Skip to main content

string_analyze/
rules.rs

1//! 规则引擎核心定义与宏构建系统。
2//!
3//! 该模块定义了扫描规则的底层结构(如 `RawRule`、`Assertion`),
4//! 并利用 `inventory` 库实现了规则的分布式注册。
5//! 模块内提供的 `lazy_rule!` 宏是定义扫描规则的核心工具。
6
7pub mod b64file;
8pub mod coding;
9pub mod core;
10pub mod hexfile;
11
12use std::collections::HashSet;
13use std::sync::LazyLock;
14
15use crate::entropy::composite_entropy;
16use crate::{AnyRule, Hit, RuleResult, State};
17
18/// 上下文断言 (Lookaround Assertion)。
19///
20/// 用于在命中主体内容后,检查其前置或后置文本是否符合特定条件。
21pub struct Assertion {
22    /// 检查的偏移量/最大窗口大小(字节数)。
23    pub offset: usize,
24    /// 期望的匹配结果:`true` 为正向断言(必须匹配),`false` 为负向断言(必须不匹配)。
25    pub expected: bool,
26    /// 用于断言的具体匹配逻辑(正则表达式或闭包)。
27    pub find: Find,
28}
29
30/// 匹配查找器。
31///
32/// 封装了底层的匹配引擎,支持正则表达式或自定义的 Rust 闭包。
33pub enum Find {
34    /// 基于 `regex::Regex` 的正则匹配。
35    Regex(regex::Regex),
36    /// 自定义状态处理闭包,修改 `State` 并返回是否匹配成功。
37    Fn(fn(&mut State) -> bool),
38}
39
40impl Find {
41    /// 在给定的状态中执行查找,过滤并保留命中的范围。
42    pub fn find(&self, state: &mut State) -> bool {
43        match self {
44            Find::Regex(regex) => has_regex(state, regex),
45            Find::Fn(f) => f(state),
46        }
47    }
48
49    /// 检查给定的字符串是否满足匹配条件(不保留命中范围,仅作布尔判断)。
50    pub fn is_match(&self, input: &str) -> bool {
51        match self {
52            Find::Regex(regex) => regex.is_match(input),
53            Find::Fn(_) => {
54                let mut temp_state = State::new(input);
55                self.find(&mut temp_state)
56            }
57        }
58    }
59}
60
61/// 原始规则定义。
62///
63/// 包含了触发一次敏感信息扫描所需的所有静态配置(正则、描述、熵值要求、断言等)。
64/// 最终会被转换并装载进 `AnyRule` 的执行流中。
65pub struct RawRule {
66    /// 基础匹配查找器(正则表达式或闭包)。
67    pub find: Find,
68    /// 规则的名称或描述。
69    pub describe: &'static str,
70    /// 规则重要程度(0-100),用于计算最终的风险评分。
71    pub importance: u8,
72    /// 最低复合熵值要求,低于此值的命中将被过滤(用于过滤死板的测试数据)。
73    pub min_entropy: Option<f64>,
74    /// 向前断言 (Lookbehind):检查匹配文本之前的上下文。
75    pub before_assert: Option<Assertion>,
76    /// 向后断言 (Lookahead):检查匹配文本之后的上下文。
77    pub after_assert: Option<Assertion>,
78    /// 组合逻辑控制:`true` 表示必须同时满足前后断言 (AND),`false` 表示满足其一即可 (OR)。默认为 `true`。
79    pub is_and_assert: bool,
80    /// 最终的自定义校验闭包,参数为 `(&匹配片段, &完整文本)`,返回 `true` 表示校验通过。
81    pub check: Option<fn(&str, &str) -> bool>,
82}
83
84impl RawRule {
85    /// 执行自定义代码校验逻辑。
86    pub fn check_custom(&self, state: &mut State) {
87        if let Some(check_fn) = self.check {
88            state.retain(|input, (start, end)| check_fn(&input[start..end], input));
89        }
90    }
91
92    /// 执行上下文断言校验逻辑。
93    fn check_assertions(&self, state: &mut State) {
94        state.retain(|input, (start, end)| {
95            // 1. 校验向前断言 (Lookbehind)
96            let before_pass = self.before_assert.as_ref().map(|assert| {
97                let mut check_start = start.saturating_sub(assert.offset);
98                // 确保字符串切片落在安全的 UTF-8 字符边界上
99                while check_start > 0 && !input.is_char_boundary(check_start) {
100                    check_start -= 1;
101                }
102                let slice = &input[check_start..start];
103                assert.find.is_match(slice) == assert.expected
104            });
105
106            // 2. 校验向后断言 (Lookahead)
107            let after_pass = self.after_assert.as_ref().map(|assert| {
108                let mut check_end = (end + assert.offset).min(input.len());
109                // 确保字符串切片落在安全的 UTF-8 字符边界上
110                while check_end < input.len() && !input.is_char_boundary(check_end) {
111                    check_end += 1;
112                }
113                let slice = &input[end..check_end];
114                assert.find.is_match(slice) == assert.expected
115            });
116
117            // 3. 结合两者的组合逻辑 (AND / OR)
118            match (before_pass, after_pass) {
119                (Some(b), Some(a)) => {
120                    if self.is_and_assert {
121                        b && a // 两个断言都必须满足
122                    } else {
123                        b || a // 满足任意一个断言即可
124                    }
125                }
126                (Some(b), None) => b,
127                (None, Some(a)) => a,
128                (None, None) => true,
129            }
130        });
131    }
132}
133
134/// 用于 `inventory` 收集的规则注册表项。
135pub struct RuleEntry {
136    pub name: &'static str,
137    pub module_path: &'static str,
138    pub importance: u8,
139    pub get_rule: fn() -> &'static RawRule,
140}
141
142inventory::collect!(RuleEntry);
143
144/// 核心规则定义宏。
145///
146/// 用于快速构建 `RawRule` 并自动注册到 `inventory` 中。
147/// 支持多种参数组合,提供从基础匹配到复杂断言校验的极简 DSL 语法。
148///
149/// # 参数排列顺序(可选参数依次向后追加)
150/// 1. `$name` (标识符) = `$regex` / `|$state| { ... }` : 规则名称与基础匹配逻辑
151/// 2. `$describe` (字符串字面量): 规则描述
152/// 3. `$importance` (数字): 严重程度 (0-100)
153/// 4. `$min_entropy` (浮点数,可选): 最低熵值要求
154/// 5. `$asserts` (元组,可选): 上下文断言定义 `(before_assert, after_assert, is_and_optional)`
155/// 6. `$check` (闭包,可选): 自定义校验代码 `|slice, input| { bool }`
156///
157/// # 断言元组语法
158/// - 断言格式: `(offset, expected, regex_or_closure)`
159/// - 空断言: `None`
160///
161/// # 示例
162/// ```rust,ignore
163/// use string_analyze::lazy_rule;
164/// use string_analyze::entropy::entropy;
165/// lazy_rule!(
166///     RE_FLAG = r#"(?i)\b[a-z0-9_.-]{0,20}(?:flag|ctf)[a-z0-9\s-]{0,4}\{[^\{\}\n=\t()]{4,256}\}"#,
167///     "发现标准 flag",
168///     100
169/// );
170/// lazy_rule!(
171///     HIGH_ENTROPY = |s| if s.input.len() <= 512 && entropy(s.input.as_bytes()) >= 5.1 { s.ranges.push((0,s.input.len())); true } else { false } ,
172///     "发现 高熵数据",
173///     30
174/// );
175/// lazy_rule!(
176///     RE_BRAINFUCK = r#"(?:[<>+\-.\[\],]{20,})"#,
177///     "发现疑似 Brainfuck 代码",
178///     40,
179///     |slice, _input| {
180///         let op_types = ['>', '<', '+', '-', '.', ',', '[', ']']
181///             .iter()
182///             .filter(|&&op| slice.contains(op))
183///             .count();
184///         op_types >= 2
185///     }
186/// );
187/// lazy_rule!(
188///     RE_PATH_UNIX = r#"(?:~|(?:\.\./|\./)|/)[a-zA-Z0-9._-]+(?:/[a-zA-Z0-9._-]+)+"#,
189///     "发现 Unix 路径",
190///     1,
191///     ((8, false, r"(?i)(?:[a-z]:?|https?:/?/?)$"), ) // (前断言, 后断言)
192/// );
193/// ```
194#[macro_export]
195macro_rules! lazy_rule {
196
197    // ==========================================
198    // 基础公有匹配模式
199    // ==========================================
200
201    // 匹配闭包
202    ($name:ident = |$state:ident| $find:expr, $describe:literal, $importance:literal $(, $($rest:tt)+)?) => {
203        $crate::rules::lazy_rule!(@parse_args
204            $name,
205            $crate::rules::Find::Fn(|$state: &mut $crate::State| $find),
206            $describe,
207            $importance
208            $(, $($rest)+)?
209        );
210    };
211    // 匹配正则表达式
212    ($name:ident = $re:literal, $describe:literal, $importance:literal $(, $($rest:tt)+)?) => {
213        $crate::rules::lazy_rule!(@parse_args
214            $name,
215            $crate::rules::Find::Regex(regex::Regex::new($re).expect(concat!("Invalid regex in ", stringify!($name)))),
216            $describe,
217            $importance
218            $(, $($rest)+)?
219        );
220    };
221
222    // ==========================================
223    // 中间宏:用来构建单一的 Assertion
224    // ==========================================
225
226    (@build_assert None) => { None };
227
228    (@build_assert ($offset:expr, $expected:expr, $re:literal)) => {
229        Some($crate::rules::Assertion {
230            offset: $offset,
231            expected: $expected,
232            find: $crate::rules::Find::Regex(regex::Regex::new($re).expect("Invalid regex in assertion")),
233        })
234    };
235
236    (@build_assert ($offset:expr, $expected:expr, |$state:ident| $func:expr)) => {
237        Some($crate::rules::Assertion {
238            offset: $offset,
239            expected: $expected,
240            find: $crate::rules::Find::Fn(|$state: &mut $crate::State| $func),
241        })
242    };
243
244    // ==========================================
245    // 中间宏:用来构建 (before, after, is_and) 断言元组
246    // ==========================================
247
248    // 1. 显式指定 is_and 参数: ($before, $after, false/true)
249    (@build_assert_tuple ($before_assert:tt, $after_assert:tt, $is_and:expr)) => {
250        (
251            $crate::rules::lazy_rule!(@build_assert $before_assert),
252            $crate::rules::lazy_rule!(@build_assert $after_assert),
253            $is_and,
254        )
255    };
256    // 2. 隐式关系(默认为与 / true): ($before, $after)
257    (@build_assert_tuple ($before_assert:tt, $after_assert:tt)) => {
258        $crate::rules::lazy_rule!(@build_assert_tuple ($before_assert, $after_assert, true))
259    };
260    // 3. 仅向前断言: ($before, )
261    (@build_assert_tuple ($before_assert:tt, )) => {
262        (
263            $crate::rules::lazy_rule!(@build_assert $before_assert),
264            None,
265            true,
266        )
267    };
268    // 4. 仅向后断言: (, $after)
269    (@build_assert_tuple (, $after_assert:tt)) => {
270        (
271            None,
272            $crate::rules::lazy_rule!(@build_assert $after_assert),
273            true,
274        )
275    };
276
277
278    // ==========================================
279    // 根基解析逻辑
280    // ==========================================
281    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, $min_entropy:expr, $asserts:expr, $check:expr $(,)?) => {
282        pub static $name: std::sync::LazyLock<$crate::rules::RawRule> = std::sync::LazyLock::new(|| {
283            let asserts = $asserts;
284            $crate::rules::RawRule {
285                find: $filter,
286                describe: $describe,
287                importance: $importance,
288                min_entropy: $min_entropy,
289                before_assert: asserts.0,
290                after_assert: asserts.1,
291                is_and_assert: asserts.2,
292                check: $check,
293            }
294        });
295        inventory::submit! {
296            $crate::rules::RuleEntry {
297                name: stringify!($name),
298                module_path: module_path!(),
299                importance: $importance,
300                get_rule: || &$name
301            }
302        }
303    };
304
305    // ==========================================
306    // 参数组合重载解析
307    // ==========================================
308
309    // === 6 字段: 包含熵, 断言元组, 自定义 check ===
310    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, $min_entropy:expr, ($($asserts:tt)+), |$slice:pat_param,$input:pat_param| $check:expr $(,)?) => {
311        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, Some($min_entropy),
312            $crate::rules::lazy_rule!(@build_assert_tuple ($($asserts)+)),
313            Some(|$slice,$input| $check)
314        );
315    };
316
317    // === 5 字段组合 ===
318    // 无熵, 断言元组, 自定义 check
319    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, ($($asserts:tt)+), |$slice:pat_param,$input:pat_param| $check:expr $(,)?) => {
320        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, None,
321            $crate::rules::lazy_rule!(@build_assert_tuple ($($asserts)+)),
322            Some(|$slice,$input| $check)
323        );
324    };
325    // 包含熵, 断言元组, 无 check
326    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, $min_entropy:expr, ($($asserts:tt)+) $(,)?) => {
327        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, Some($min_entropy),
328            $crate::rules::lazy_rule!(@build_assert_tuple ($($asserts)+)),
329            None
330        );
331    };
332    // 包含熵, 无断言元组, 有 check
333    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, $min_entropy:expr, |$slice:pat_param,$input:pat_param| $check:expr $(,)?) => {
334        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, Some($min_entropy), (None, None, true), Some(|$slice,$input| $check));
335    };
336
337    // === 4 字段组合 ===
338    // 无熵, 无断言元组, 有 check
339    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, |$slice:pat_param,$input:pat_param| $check:expr $(,)?) => {
340        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, None, (None, None, true), Some(|$slice,$input| $check));
341    };
342    // 无熵, 有断言元组, 无 check
343    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, ($($asserts:tt)+) $(,)?) => {
344        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, None,
345            $crate::rules::lazy_rule!(@build_assert_tuple ($($asserts)+)),
346            None
347        );
348    };
349    // 有熵, 无断言元组, 无 check
350    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, $min_entropy:expr $(,)?) => {
351        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, Some($min_entropy), (None, None, true), None);
352    };
353
354    // === 3 字段 (最简版本) ===
355    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal $(,)?) => {
356        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, None, (None, None, true), None);
357    };
358}
359
360use crate::tool::has_regex;
361pub use lazy_rule;
362
363/// 全局预编译的所有规则实例。
364/// 利用 `LazyLock` 在首次访问时加载并编译 `inventory` 收集到的规则。
365pub static ALL_RULES: LazyLock<Vec<AnyRule>> = LazyLock::new(|| get_rules(|_, _| true));
366
367/// 获取被编译和组装好的可执行规则集合 (`AnyRule`)。
368///
369/// # 参数
370/// - `filter`: 闭包函数 `(模块路径, 重要程度) -> bool`。用于根据模块或严重程度过滤不需要的规则。
371///
372/// # 执行流 (Pipeline)
373/// 构建的 `AnyRule` 将按照以下管道依次处理文本:
374/// 1. **基础提取**:执行正则或基础闭包查找,圈定候选范围。
375/// 2. **熵值过滤**(可选):移除复合熵低于 `min_entropy` 的候选区间(过滤死板字符)。
376/// 3. **上下文断言**(可选):检查并过滤前后文不符合预期的区间。
377/// 4. **自定义校验**(可选):执行用户自定义的特定逻辑检查。
378pub fn get_rules<F>(filter: F) -> Vec<AnyRule>
379where
380    F: Fn(&str, u8) -> bool,
381{
382    let mut rules = Vec::new();
383    let mut seen_names = HashSet::new();
384
385    for entry in inventory::iter::<RuleEntry> {
386        // 1. 应用模块和严重程度过滤
387        if filter(entry.module_path, entry.importance) {
388            // 避免同名规则重复加载
389            if !seen_names.insert(entry.name) {
390                continue;
391            }
392
393            let r: &'static RawRule = (entry.get_rule)();
394
395            // 2. 组装 AnyRule 执行流
396            rules.push(
397                AnyRule::new(|state| Hit {
398                    describe: r.describe.into(),
399                    importance: r.importance,
400                    data: RuleResult::from(state),
401                })
402                // Flow 1: 基础查找
403                .add_flow(move |state| r.find.find(state))
404                // Flow 2: 复合熵值校验
405                .add_flow(move |state| {
406                    if let Some(min_ent) = r.min_entropy {
407                        state.retain(|input, (start, end)| {
408                            composite_entropy(&input.as_bytes()[start..end]) >= min_ent
409                        });
410                    }
411                    !state.ranges.is_empty()
412                })
413                // Flow 3: 上下文断言校验 (Lookaround)
414                .add_flow(move |state| {
415                    r.check_assertions(state);
416                    !state.ranges.is_empty()
417                })
418                // Flow 4: 自定义代码校验
419                .add_flow(move |state| {
420                    r.check_custom(state);
421                    !state.ranges.is_empty()
422                }),
423            );
424        }
425    }
426
427    rules
428}