string-analyze 0.1.0

Find key strings from cluttered text
Documentation
//string_analyze/src/rules.rs

pub mod b64file;
pub mod core;
pub mod hexfile;
pub mod coding;

use std::collections::HashSet;
use std::sync::LazyLock;

use crate::entropy::composite_entropy;
use crate::{AnyRule, Hit, RuleResult, State};

pub struct Assertion {
    pub offset: usize,
    pub expected: bool,
    pub find: Find,
}

pub enum Find {
    Regex(regex::Regex),
    Fn(fn(&mut State) -> bool),
}

impl Find {
    pub fn find(&self, state: &mut State) -> bool {
        match self {
            Find::Regex(regex) => has_regex(state, regex),
            Find::Fn(f) => f(state),
        }
    }
    pub fn is_match(&self, input: &str) -> bool {
        match self {
            Find::Regex(regex) => regex.is_match(input),
            Find::Fn(_) => {
                let mut temp_state = State::new(input);
                self.find(&mut temp_state)
            }
        }
    }
}

pub struct RawRule {
    pub find: Find,
    pub describe: &'static str,
    pub importance: u8,
    pub min_entropy: Option<f64>,
    pub before_assert: Option<Assertion>, // 向前断言
    pub after_assert: Option<Assertion>,  // 向后断言
    pub is_and_assert: bool,              // true 为“与”(AND),false 为“或”(OR),默认为 true
    pub check: Option<fn(&str, &str) -> bool>,
}

impl RawRule {
    pub fn check_custom(&self, state: &mut State) {
        if let Some(check_fn) = self.check {
            state.retain(|input, (start, end)| check_fn(&input[start..end], input));
        }
    }

    fn check_assertions(&self, state: &mut State) {
        state.retain(|input, (start, end)| {
            // 1. 校验向前断言 (Lookbehind)
            let before_pass = self.before_assert.as_ref().map(|assert| {
                let mut check_start = start.saturating_sub(assert.offset);
                while check_start > 0 && !input.is_char_boundary(check_start) {
                    check_start -= 1;
                }
                let slice = &input[check_start..start];
                assert.find.is_match(slice) == assert.expected
            });

            // 2. 校验向后断言 (Lookahead)
            let after_pass = self.after_assert.as_ref().map(|assert| {
                let mut check_end = (end + assert.offset).min(input.len());
                while check_end < input.len() && !input.is_char_boundary(check_end) {
                    check_end += 1;
                }
                let slice = &input[end..check_end];
                assert.find.is_match(slice) == assert.expected
            });

            // 3. 结合两者的组合逻辑 (AND / OR)
            match (before_pass, after_pass) {
                (Some(b), Some(a)) => {
                    if self.is_and_assert {
                        b && a // 两个断言都必须满足
                    } else {
                        b || a // 满足任意一个断言即可
                    }
                }
                (Some(b), None) => b,
                (None, Some(a)) => a,
                (None, None) => true,
            }
        });
    }
}
pub struct RuleEntry {
    pub name: &'static str,
    pub module_path: &'static str,
    pub importance: u8,
    pub get_rule: fn() -> &'static RawRule,
}

inventory::collect!(RuleEntry);

macro_rules! lazy_rule {

    // ==========================================
    // 基础公有匹配模式
    // ==========================================

    // 匹配闭包
    ($name:ident = |$state:ident| $find:expr, $describe:literal, $importance:literal $(, $($rest:tt)+)?) => {
        $crate::rules::lazy_rule!(@parse_args
            $name,
            $crate::rules::Find::Fn(|$state: &mut $crate::State| $find),
            $describe,
            $importance
            $(, $($rest)+)?
        );
    };
    // 匹配正则表达式
    ($name:ident = $re:literal, $describe:literal, $importance:literal $(, $($rest:tt)+)?) => {
        $crate::rules::lazy_rule!(@parse_args
            $name,
            $crate::rules::Find::Regex(regex::Regex::new($re).expect(concat!("Invalid regex in ", stringify!($name)))),
            $describe,
            $importance
            $(, $($rest)+)?
        );
    };

    // ==========================================
    // 中间宏:用来构建单一的 Assertion
    // ==========================================

    (@build_assert None) => { None };

    (@build_assert ($offset:expr, $expected:expr, $re:literal)) => {
        Some($crate::rules::Assertion {
            offset: $offset,
            expected: $expected,
            find: $crate::rules::Find::Regex(regex::Regex::new($re).expect("Invalid regex in assertion")),
        })
    };

    (@build_assert ($offset:expr, $expected:expr, |$state:ident| $func:expr)) => {
        Some($crate::rules::Assertion {
            offset: $offset,
            expected: $expected,
            find: $crate::rules::Find::Fn(|$state: &mut $crate::State| $func),
        })
    };

    // ==========================================
    // 中间宏:用来构建 (before, after, is_and) 断言元组
    // ==========================================

    // 1. 显式指定 is_and 参数: ($before, $after, false/true)
    (@build_assert_tuple ($before_assert:tt, $after_assert:tt, $is_and:expr)) => {
        (
            $crate::rules::lazy_rule!(@build_assert $before_assert),
            $crate::rules::lazy_rule!(@build_assert $after_assert),
            $is_and,
        )
    };
    // 2. 隐式关系(默认为与 / true): ($before, $after)
    (@build_assert_tuple ($before_assert:tt, $after_assert:tt)) => {
        $crate::rules::lazy_rule!(@build_assert_tuple ($before_assert, $after_assert, true))
    };
    // 3. 仅向前断言: ($before, )
    (@build_assert_tuple ($before_assert:tt, )) => {
        (
            $crate::rules::lazy_rule!(@build_assert $before_assert),
            None,
            true,
        )
    };
    // 4. 仅向后断言: (, $after)
    (@build_assert_tuple (, $after_assert:tt)) => {
        (
            None,
            $crate::rules::lazy_rule!(@build_assert $after_assert),
            true,
        )
    };


    // ==========================================
    // 根基解析逻辑
    // ==========================================
    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, $min_entropy:expr, $asserts:expr, $check:expr $(,)?) => {
        pub static $name: std::sync::LazyLock<$crate::rules::RawRule> = std::sync::LazyLock::new(|| {
            let asserts = $asserts;
            $crate::rules::RawRule {
                find: $filter,
                describe: $describe,
                importance: $importance,
                min_entropy: $min_entropy,
                before_assert: asserts.0,
                after_assert: asserts.1,
                is_and_assert: asserts.2,
                check: $check,
            }
        });
        inventory::submit! {
            $crate::rules::RuleEntry {
                name: stringify!($name),
                module_path: module_path!(),
                importance: $importance,
                get_rule: || &$name
            }
        }
    };

    // ==========================================
    // 参数组合重载解析
    // ==========================================

    // === 6 字段: 包含熵, 断言元组, 自定义 check ===
    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, $min_entropy:expr, ($($asserts:tt)+), |$slice:pat_param,$input:pat_param| $check:expr $(,)?) => {
        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, Some($min_entropy),
            $crate::rules::lazy_rule!(@build_assert_tuple ($($asserts)+)),
            Some(|$slice,$input| $check)
        );
    };

    // === 5 字段组合 ===
    // 无熵, 断言元组, 自定义 check
    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, ($($asserts:tt)+), |$slice:pat_param,$input:pat_param| $check:expr $(,)?) => {
        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, None,
            $crate::rules::lazy_rule!(@build_assert_tuple ($($asserts)+)),
            Some(|$slice,$input| $check)
        );
    };
    // 包含熵, 断言元组, 无 check
    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, $min_entropy:expr, ($($asserts:tt)+) $(,)?) => {
        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, Some($min_entropy),
            $crate::rules::lazy_rule!(@build_assert_tuple ($($asserts)+)),
            None
        );
    };
    // 包含熵, 无断言元组, 有 check
    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, $min_entropy:expr, |$slice:pat_param,$input:pat_param| $check:expr $(,)?) => {
        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, Some($min_entropy), (None, None, true), Some(|$slice,$input| $check));
    };

    // === 4 字段组合 ===
    // 无熵, 无断言元组, 有 check
    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, |$slice:pat_param,$input:pat_param| $check:expr $(,)?) => {
        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, None, (None, None, true), Some(|$slice,$input| $check));
    };
    // 无熵, 有断言元组, 无 check
    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, ($($asserts:tt)+) $(,)?) => {
        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, None,
            $crate::rules::lazy_rule!(@build_assert_tuple ($($asserts)+)),
            None
        );
    };
    // 有熵, 无断言元组, 无 check
    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal, $min_entropy:expr $(,)?) => {
        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, Some($min_entropy), (None, None, true), None);
    };

    // === 3 字段 (最简版本) ===
    (@parse_args $name:ident, $filter:expr, $describe:literal, $importance:literal $(,)?) => {
        $crate::rules::lazy_rule!(@parse_args $name, $filter, $describe, $importance, None, (None, None, true), None);
    };
}

use crate::tool::has_regex;
pub(crate) use lazy_rule;

pub static ALL_RULES: LazyLock<Vec<AnyRule>> = LazyLock::new(|| get_rules(|_, _| true));

pub fn get_rules<F>(filter: F) -> Vec<AnyRule>
where
    F: Fn(&str, u8) -> bool,
{
    let mut rules = Vec::new();
    let mut seen_names = HashSet::new();

    for entry in inventory::iter::<RuleEntry> {
        if filter(entry.module_path, entry.importance) {

            if !seen_names.insert(entry.name) {
                continue;
            }

            let r: &'static RawRule = (entry.get_rule)();

            rules.push(
                AnyRule::new(|state| Hit {
                    describe: r.describe.into(),
                    importance: r.importance,
                    data: RuleResult::from(state),
                })
                .add_flow(move |state| r.find.find(state))
                .add_flow(move |state| {
                    if let Some(min_ent) = r.min_entropy {
                        state.retain(|input, (start, end)| {
                            composite_entropy((&input[start..end]).as_bytes()) >= min_ent
                        });
                    }
                    !state.ranges.is_empty()
                })
                .add_flow(move |state| {
                    r.check_assertions(state);
                    !state.ranges.is_empty()
                })
                .add_flow(move |state| {
                    r.check_custom(state);
                    !state.ranges.is_empty()
                }),
            );
        }
    }

    rules
}