systemprompt_security/policy/secrets/
entropy.rs1use std::ops::Range;
23
24use base64::Engine as _;
25use base64::engine::general_purpose::{STANDARD_NO_PAD, URL_SAFE_NO_PAD};
26use regex::Regex;
27
28pub const DEFAULT_MIN_LEN: usize = 32;
29
30pub const DEFAULT_THRESHOLD: f64 = 0.80;
31
32const TOKEN_DELIMITERS: &str = "\"'`()[]{}<>,;:";
33const TOKEN_CHARSET_EXTRA: &str = "+/=_-";
34const ENTROPY_CEILING_SYMBOLS: usize = 64;
35
36const MIN_STRUCTURED_LEN: usize = 16;
37const MAX_PAYLOAD_PREFIX_LEN: usize = 10;
38const DIGEST_LENGTHS: [(&str, usize); 3] = [("sha256", 32), ("sha384", 48), ("sha512", 64)];
39const MAX_FIELD_NUMBER: u64 = 64;
40const MAX_NESTING_DEPTH: u32 = 4;
41const MIN_NESTED_PAYLOAD_LEN: usize = 4;
42const TEXT_RATIO_NUMERATOR: usize = 9;
43const TEXT_RATIO_DENOMINATOR: usize = 10;
44
45#[derive(Debug, Clone)]
49pub struct EntropyConfig {
50 pub enabled: bool,
51 pub min_len: usize,
52 pub threshold: f64,
53 pub allowlist: Vec<Regex>,
54}
55
56impl Default for EntropyConfig {
57 fn default() -> Self {
58 Self {
59 enabled: true,
60 min_len: DEFAULT_MIN_LEN,
61 threshold: DEFAULT_THRESHOLD,
62 allowlist: Vec::new(),
63 }
64 }
65}
66
67#[must_use]
68pub fn find_high_entropy_token<'a>(text: &'a str, config: &EntropyConfig) -> Option<&'a str> {
69 high_entropy_spans(text, config)
70 .next()
71 .map(|(_, token)| token)
72}
73
74pub(super) fn high_entropy_spans<'a, 'c>(
75 text: &'a str,
76 config: &'c EntropyConfig,
77) -> impl Iterator<Item = (Range<usize>, &'a str)> + use<'a, 'c> {
78 config
79 .enabled
80 .then(|| token_spans(text).filter(move |(_, token)| is_credential_shaped(token, config)))
81 .into_iter()
82 .flatten()
83}
84
85fn is_token_delimiter(c: char) -> bool {
86 c.is_whitespace() || TOKEN_DELIMITERS.contains(c)
87}
88
89fn token_spans(text: &str) -> impl Iterator<Item = (Range<usize>, &str)> {
90 let mut cursor = 0;
91 std::iter::from_fn(move || {
92 let start = cursor + text[cursor..].find(|c| !is_token_delimiter(c))?;
93 let end = text[start..]
94 .find(is_token_delimiter)
95 .map_or(text.len(), |len| start + len);
96 cursor = end;
97 Some((start..end, &text[start..end]))
98 })
99}
100
101fn is_credential_shaped(token: &str, config: &EntropyConfig) -> bool {
102 has_credential_shape(token, config)
103 && !config.allowlist.iter().any(|re| re.is_match(token))
104 && !is_verified_digest(token)
105 && !is_structured_payload(token)
106 && !is_filesystem_path(token, config)
107}
108
109fn has_credential_shape(token: &str, config: &EntropyConfig) -> bool {
110 token.len() >= config.min_len
111 && token
112 .chars()
113 .all(|c| c.is_ascii_alphanumeric() || TOKEN_CHARSET_EXTRA.contains(c))
114 && token.chars().any(|c| c.is_ascii_uppercase())
115 && token.chars().any(|c| c.is_ascii_lowercase())
116 && token.chars().any(|c| c.is_ascii_digit())
117 && entropy_ratio(token) >= config.threshold
118}
119
120fn is_filesystem_path(token: &str, config: &EntropyConfig) -> bool {
121 token.starts_with('/')
122 && token.matches('/').count() >= 2
123 && !token.contains(['+', '='])
124 && !token
125 .split('/')
126 .any(|segment| has_credential_shape(segment, config))
127}
128
129fn is_verified_digest(token: &str) -> bool {
132 let Some((prefix, payload)) = token.split_once('-') else {
133 return false;
134 };
135 DIGEST_LENGTHS
136 .iter()
137 .find(|(algo, _)| algo.eq_ignore_ascii_case(prefix))
138 .is_some_and(|&(_, digest_len)| {
139 decode_base64(payload).is_some_and(|bytes| bytes.len() == digest_len)
140 })
141}
142
143fn shannon_entropy(s: &str) -> f64 {
144 let mut counts = [0u32; 256];
145 let bytes = s.as_bytes();
146 for &b in bytes {
147 counts[usize::from(b)] += 1;
148 }
149 let len = bytes.len() as f64;
150 counts
151 .iter()
152 .filter(|&&c| c > 0)
153 .map(|&c| {
154 let p = f64::from(c) / len;
155 -p * p.log2()
156 })
157 .sum()
158}
159
160fn entropy_ratio(s: &str) -> f64 {
161 let symbols = s.len().min(ENTROPY_CEILING_SYMBOLS);
162 let ceiling = (symbols as f64).log2();
163 if ceiling <= 0.0 {
164 return 0.0;
165 }
166 shannon_entropy(s) / ceiling
167}
168
169fn is_structured_payload(token: &str) -> bool {
170 decoded_payload(token).is_some_and(|bytes| {
171 bytes.len() >= MIN_STRUCTURED_LEN && (is_mostly_text(&bytes) || is_protobuf(&bytes, 0))
172 })
173}
174
175fn decoded_payload(token: &str) -> Option<Vec<u8>> {
176 decode_base64(token).or_else(|| {
177 let (prefix, payload) = token.split_once('-')?;
178 let plausible_prefix = prefix.len() <= MAX_PAYLOAD_PREFIX_LEN
179 && prefix.chars().all(|c| c.is_ascii_alphanumeric());
180 plausible_prefix.then(|| decode_base64(payload)).flatten()
181 })
182}
183
184fn decode_base64(token: &str) -> Option<Vec<u8>> {
185 let body = token.trim_end_matches('=');
186 STANDARD_NO_PAD
187 .decode(body)
188 .or_else(|_| URL_SAFE_NO_PAD.decode(body))
189 .ok()
190}
191
192fn is_mostly_text(bytes: &[u8]) -> bool {
193 if bytes.is_empty() {
194 return false;
195 }
196 let printable = bytes
197 .iter()
198 .filter(|&&b| matches!(b, b'\t' | b'\n' | b'\r') || (0x20..0x7f).contains(&b))
199 .count();
200 printable * TEXT_RATIO_DENOMINATOR >= bytes.len() * TEXT_RATIO_NUMERATOR
201}
202
203fn is_protobuf(bytes: &[u8], depth: u32) -> bool {
204 if depth > MAX_NESTING_DEPTH || bytes.len() < MIN_STRUCTURED_LEN {
205 return false;
206 }
207 parse_protobuf(bytes, depth).is_some_and(|parsed| parsed.fields >= 2 && parsed.nested_structure)
208}
209
210struct ParsedMessage {
211 fields: usize,
212 nested_structure: bool,
213}
214
215fn parse_protobuf(bytes: &[u8], depth: u32) -> Option<ParsedMessage> {
216 let mut cursor = 0usize;
217 let mut parsed = ParsedMessage {
218 fields: 0,
219 nested_structure: false,
220 };
221 while cursor < bytes.len() {
222 let (tag, after_tag) = read_varint(bytes, cursor)?;
223 let field_number = tag >> 3;
224 if field_number == 0 || field_number > MAX_FIELD_NUMBER {
225 return None;
226 }
227 cursor = match tag & 7 {
228 0 => read_varint(bytes, after_tag)?.1,
229 1 => advance(bytes, after_tag, 8)?,
230 5 => advance(bytes, after_tag, 4)?,
231 2 => {
232 let (len, after_len) = read_varint(bytes, after_tag)?;
233 let len = usize::try_from(len).ok()?;
234 let end = advance(bytes, after_len, len)?;
235 let payload = bytes.get(after_len..end)?;
236 if payload.len() >= MIN_NESTED_PAYLOAD_LEN
237 && (is_mostly_text(payload) || is_protobuf(payload, depth + 1))
238 {
239 parsed.nested_structure = true;
240 }
241 end
242 },
243 _ => return None,
244 };
245 parsed.fields += 1;
246 }
247 Some(parsed)
248}
249
250fn advance(bytes: &[u8], cursor: usize, by: usize) -> Option<usize> {
251 let end = cursor.checked_add(by)?;
252 (end <= bytes.len()).then_some(end)
253}
254
255fn read_varint(bytes: &[u8], start: usize) -> Option<(u64, usize)> {
256 let mut value = 0u64;
257 let mut shift = 0u32;
258 let mut cursor = start;
259 loop {
260 let byte = *bytes.get(cursor)?;
261 cursor += 1;
262 value |= u64::from(byte & 0x7f) << shift;
263 if byte & 0x80 == 0 {
264 return Some((value, cursor));
265 }
266 shift += 7;
267 if shift >= 64 {
268 return None;
269 }
270 }
271}