string_analyze/rules/core/
pii.rs1use crate::rules::lazy_rule;
3
4fn luhn_check(s: &str) -> bool {
5 let digits: Vec<u32> = s
6 .chars()
7 .filter(|c| c.is_ascii_digit())
8 .filter_map(|c| c.to_digit(10))
9 .collect();
10
11 if digits.len() < 13 || digits.len() > 19 {
12 return false;
13 }
14
15 let checksum: u32 = digits
16 .iter()
17 .rev()
18 .enumerate()
19 .map(|(idx, &d)| {
20 if idx % 2 == 1 {
21 let doubled = d * 2;
22 if doubled > 9 { doubled - 9 } else { doubled }
23 } else {
24 d
25 }
26 })
27 .sum();
28
29 checksum % 10 == 0
30}
31fn is_valid_bin(card: &str) -> bool {
32 if card.len() < 6 {
33 return false;
34 }
35 const VALID_BINS: &[&str] = &[
36 "62", "4", "5", "35", "37", "6011", "622", ];
44
45 VALID_BINS.iter().any(|&bin| card.starts_with(bin))
46}
47
48fn id_card_check(id: &str) -> bool {
49 if id.len() != 18 {
50 return false;
51 }
52
53 let weights = [7, 9, 10, 5, 8, 4, 2, 1, 6, 3, 7, 9, 10, 5, 8, 4, 2];
54 let check_codes = ['1', '0', 'X', '9', '8', '7', '6', '5', '4', '3', '2'];
55
56 let sum: u32 = id[..17]
57 .chars()
58 .filter_map(|c| c.to_digit(10))
59 .enumerate()
60 .map(|(i, d)| d * weights[i])
61 .sum();
62
63 let expected = check_codes[(sum % 11) as usize];
64 let actual = id.chars().nth(17).unwrap_or(' ').to_ascii_uppercase();
65
66 expected == actual
67}
68
69lazy_rule!(
70 RE_CREDIT_CARD = r"\b[1-9]\d{12,18}\b",
71 "发现 疑似信用卡号",
72 90,
73 |input,_| is_valid_bin(input) && luhn_check(input)
74);
75
76lazy_rule!(
77 RE_CN_ID_CARD =
78 r#"[1-9]\d{5}(?:19|20)\d{2}(?:0[1-9]|1[0-2])(?:0[1-9]|[12]\d|3[01])\d{3}[\dXx]"#,
79 "发现 疑似中国大陆身份证号",
80 90,
81 |input,_| id_card_check(input)
82);
83
84lazy_rule!(
85 RE_CN_PHONE = r#"\b1[3-9]\d{9}\b"#,
86 "发现 疑似手机号",
87 60,
88 ((10, false, r"(?i)(order|id|stamp|time)"), )
89);
90lazy_rule!(
91 RE_PASSPORT = r#"\b[A-Z]{1,2}[0-9]{6,9}\b"#,
92 "发现 疑似护照号",
93 90
94);
95lazy_rule!(
96 RE_HKID = r#"\b[A-Z]{1,2}\d{6}\([0-9A]\)"#,
97 "发现 香港身份证号",
98 90
99);
100lazy_rule!(
101 RE_EMAIL = r#"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b"#,
102 "发现 邮箱地址",
103 60
104);
105
106fn usci_check(s: &str) -> bool {
107 if s.len() != 18 {
108 return false;
109 }
110 let dict = "0123456789ABCDEFGHJKLMNPQRTUWXY";
111 let weights = [
112 1, 3, 9, 27, 19, 26, 16, 17, 20, 29, 25, 14, 12, 5, 15, 14, 12,
113 ];
114
115 let mut sum = 0;
116 for (i, c) in s[..17].chars().enumerate() {
117 if let Some(val) = dict.find(c) {
118 sum += val * weights[i];
119 } else {
120 return false;
121 }
122 }
123
124 let check_val = 31 - (sum % 31);
125 let check_char = if check_val == 31 {
126 '0'
127 } else {
128 dict.chars().nth(check_val).unwrap()
129 };
130
131 s.ends_with(check_char)
132}
133
134lazy_rule!(
135 RE_CN_USCI = r#"[0-9A-HJ-NPQRTUWXY]{2}\d{6}[0-9A-HJ-NPQRTUWXY]{10}"#,
136 "发现 统一社会信用代码",
137 90,
138 |input,_| usci_check(input)
139);
140
141lazy_rule!(
142 RE_TW_ID = r"\b[A-Z][12]\d{8}\b",
143 "发现 台湾身份证号",
144 90,
145 |input,_| tw_id_check(input)
146);
147
148fn tw_id_check(id: &str) -> bool {
149 if id.len() != 10 {
150 return false;
151 }
152
153 let first_num = match id.chars().next() {
155 Some('A') => 10,
156 Some('B') => 11,
157 Some('C') => 12,
158 Some('D') => 13,
159 Some('E') => 14,
160 Some('F') => 15,
161 Some('G') => 16,
162 Some('H') => 17,
163 Some('I') => 34,
164 Some('J') => 18,
165 Some('K') => 19,
166 Some('L') => 20,
167 Some('M') => 21,
168 Some('N') => 22,
169 Some('O') => 35,
170 Some('P') => 23,
171 Some('Q') => 24,
172 Some('R') => 25,
173 Some('S') => 26,
174 Some('T') => 27,
175 Some('U') => 28,
176 Some('V') => 29,
177 Some('W') => 32,
178 Some('X') => 30,
179 Some('Y') => 31,
180 Some('Z') => 33,
181 _ => return false,
182 };
183
184 let weights = [1, 9, 8, 7, 6, 5, 4, 3, 2, 1, 1];
185 let sum: u32 = (first_num / 10) * weights[0]
186 + (first_num % 10) * weights[1]
187 + id[1..]
188 .chars()
189 .filter_map(|c| c.to_digit(10))
190 .zip(&weights[2..])
191 .map(|(d, &w)| d * w)
192 .sum::<u32>();
193
194 sum % 10 == 0
195}
196
197lazy_rule!(RE_MO_ID = r"\b[1578]\d{6}\(\d\)\b", "发现 澳门身份证号", 90);