Skip to main content

systemprompt_security/policy/secrets/
entropy.rs

1//! The high-entropy backstop for credentials carrying no vendor prefix.
2//!
3//! A random base64 blob pasted into a prompt can match no configured signature
4//! but still read as machine-generated key
5//! material. Randomness alone cannot say so: a serialised protobuf, a base64
6//! JSON envelope, and a 32-byte key are all dense mixed-case base64 of similar
7//! measured entropy. [`is_structured_payload`] supplies the missing
8//! discriminator — key material decodes to bytes with no readable structure,
9//! whereas a tool result decodes to text or to a self-consistent wire format.
10//!
11//! `/` is a legal token character, so an absolute path is scored as one token
12//! rather than split on its separators. A macOS `$TMPDIR`
13//! (`/var/folders/<12>/<30>/T/`) clears every check that way — the random
14//! segment supplies the entropy, and the `T` supplies the uppercase the shape
15//! test demands — so [`is_filesystem_path`] exempts path-shaped tokens. Claude
16//! Code carries those paths in its system prompt, and without the exemption
17//! every request from an affected Mac is denied.
18//!
19//! Copyright (c) systemprompt.io — Business Source License 1.1.
20//! See <https://systemprompt.io> for licensing details.
21
22use std::ops::Range;
23
24use base64::Engine as _;
25use base64::engine::general_purpose::{STANDARD_NO_PAD, URL_SAFE_NO_PAD};
26use regex::Regex;
27
28pub const DEFAULT_MIN_LEN: usize = 32;
29
30pub const DEFAULT_THRESHOLD: f64 = 0.80;
31
32const TOKEN_DELIMITERS: &str = "\"'`()[]{}<>,;:";
33const TOKEN_CHARSET_EXTRA: &str = "+/=_-";
34const ENTROPY_CEILING_SYMBOLS: usize = 64;
35
36const MIN_STRUCTURED_LEN: usize = 16;
37const MAX_PAYLOAD_PREFIX_LEN: usize = 10;
38const DIGEST_LENGTHS: [(&str, usize); 3] = [("sha256", 32), ("sha384", 48), ("sha512", 64)];
39const MAX_FIELD_NUMBER: u64 = 64;
40const MAX_NESTING_DEPTH: u32 = 4;
41const MIN_NESTED_PAYLOAD_LEN: usize = 4;
42const TEXT_RATIO_NUMERATOR: usize = 9;
43const TEXT_RATIO_DENOMINATOR: usize = 10;
44
45/// Tunables for the heuristic, read from the `secret_scan` policy's `entropy`
46/// block. [`Default`] reproduces the built-in behaviour, which is what every
47/// caller outside the policy chain gets.
48#[derive(Debug, Clone)]
49pub struct EntropyConfig {
50    pub enabled: bool,
51    pub min_len: usize,
52    pub threshold: f64,
53    pub allowlist: Vec<Regex>,
54}
55
56impl Default for EntropyConfig {
57    fn default() -> Self {
58        Self {
59            enabled: true,
60            min_len: DEFAULT_MIN_LEN,
61            threshold: DEFAULT_THRESHOLD,
62            allowlist: Vec::new(),
63        }
64    }
65}
66
67#[must_use]
68pub fn find_high_entropy_token<'a>(text: &'a str, config: &EntropyConfig) -> Option<&'a str> {
69    high_entropy_spans(text, config)
70        .next()
71        .map(|(_, token)| token)
72}
73
74pub(super) fn high_entropy_spans<'a, 'c>(
75    text: &'a str,
76    config: &'c EntropyConfig,
77) -> impl Iterator<Item = (Range<usize>, &'a str)> + use<'a, 'c> {
78    config
79        .enabled
80        .then(|| token_spans(text).filter(move |(_, token)| is_credential_shaped(token, config)))
81        .into_iter()
82        .flatten()
83}
84
85fn is_token_delimiter(c: char) -> bool {
86    c.is_whitespace() || TOKEN_DELIMITERS.contains(c)
87}
88
89fn token_spans(text: &str) -> impl Iterator<Item = (Range<usize>, &str)> {
90    let mut cursor = 0;
91    std::iter::from_fn(move || {
92        let start = cursor + text[cursor..].find(|c| !is_token_delimiter(c))?;
93        let end = text[start..]
94            .find(is_token_delimiter)
95            .map_or(text.len(), |len| start + len);
96        cursor = end;
97        Some((start..end, &text[start..end]))
98    })
99}
100
101fn is_credential_shaped(token: &str, config: &EntropyConfig) -> bool {
102    has_credential_shape(token, config)
103        && !config.allowlist.iter().any(|re| re.is_match(token))
104        && !is_verified_digest(token)
105        && !is_structured_payload(token)
106        && !is_filesystem_path(token, config)
107}
108
109fn has_credential_shape(token: &str, config: &EntropyConfig) -> bool {
110    token.len() >= config.min_len
111        && token
112            .chars()
113            .all(|c| c.is_ascii_alphanumeric() || TOKEN_CHARSET_EXTRA.contains(c))
114        && token.chars().any(|c| c.is_ascii_uppercase())
115        && token.chars().any(|c| c.is_ascii_lowercase())
116        && token.chars().any(|c| c.is_ascii_digit())
117        && entropy_ratio(token) >= config.threshold
118}
119
120fn is_filesystem_path(token: &str, config: &EntropyConfig) -> bool {
121    token.starts_with('/')
122        && token.matches('/').count() >= 2
123        && !token.contains(['+', '='])
124        && !token
125            .split('/')
126            .any(|segment| has_credential_shape(segment, config))
127}
128
129// Why: SRI digests are public integrity metadata despite their high-entropy
130// base64 payloads.
131fn is_verified_digest(token: &str) -> bool {
132    let Some((prefix, payload)) = token.split_once('-') else {
133        return false;
134    };
135    DIGEST_LENGTHS
136        .iter()
137        .find(|(algo, _)| algo.eq_ignore_ascii_case(prefix))
138        .is_some_and(|&(_, digest_len)| {
139            decode_base64(payload).is_some_and(|bytes| bytes.len() == digest_len)
140        })
141}
142
143fn shannon_entropy(s: &str) -> f64 {
144    let mut counts = [0u32; 256];
145    let bytes = s.as_bytes();
146    for &b in bytes {
147        counts[usize::from(b)] += 1;
148    }
149    let len = bytes.len() as f64;
150    counts
151        .iter()
152        .filter(|&&c| c > 0)
153        .map(|&c| {
154            let p = f64::from(c) / len;
155            -p * p.log2()
156        })
157        .sum()
158}
159
160fn entropy_ratio(s: &str) -> f64 {
161    let symbols = s.len().min(ENTROPY_CEILING_SYMBOLS);
162    let ceiling = (symbols as f64).log2();
163    if ceiling <= 0.0 {
164        return 0.0;
165    }
166    shannon_entropy(s) / ceiling
167}
168
169fn is_structured_payload(token: &str) -> bool {
170    decoded_payload(token).is_some_and(|bytes| {
171        bytes.len() >= MIN_STRUCTURED_LEN && (is_mostly_text(&bytes) || is_protobuf(&bytes, 0))
172    })
173}
174
175fn decoded_payload(token: &str) -> Option<Vec<u8>> {
176    decode_base64(token).or_else(|| {
177        let (prefix, payload) = token.split_once('-')?;
178        let plausible_prefix = prefix.len() <= MAX_PAYLOAD_PREFIX_LEN
179            && prefix.chars().all(|c| c.is_ascii_alphanumeric());
180        plausible_prefix.then(|| decode_base64(payload)).flatten()
181    })
182}
183
184fn decode_base64(token: &str) -> Option<Vec<u8>> {
185    let body = token.trim_end_matches('=');
186    STANDARD_NO_PAD
187        .decode(body)
188        .or_else(|_| URL_SAFE_NO_PAD.decode(body))
189        .ok()
190}
191
192fn is_mostly_text(bytes: &[u8]) -> bool {
193    if bytes.is_empty() {
194        return false;
195    }
196    let printable = bytes
197        .iter()
198        .filter(|&&b| matches!(b, b'\t' | b'\n' | b'\r') || (0x20..0x7f).contains(&b))
199        .count();
200    printable * TEXT_RATIO_DENOMINATOR >= bytes.len() * TEXT_RATIO_NUMERATOR
201}
202
203fn is_protobuf(bytes: &[u8], depth: u32) -> bool {
204    if depth > MAX_NESTING_DEPTH || bytes.len() < MIN_STRUCTURED_LEN {
205        return false;
206    }
207    parse_protobuf(bytes, depth).is_some_and(|parsed| parsed.fields >= 2 && parsed.nested_structure)
208}
209
210struct ParsedMessage {
211    fields: usize,
212    nested_structure: bool,
213}
214
215fn parse_protobuf(bytes: &[u8], depth: u32) -> Option<ParsedMessage> {
216    let mut cursor = 0usize;
217    let mut parsed = ParsedMessage {
218        fields: 0,
219        nested_structure: false,
220    };
221    while cursor < bytes.len() {
222        let (tag, after_tag) = read_varint(bytes, cursor)?;
223        let field_number = tag >> 3;
224        if field_number == 0 || field_number > MAX_FIELD_NUMBER {
225            return None;
226        }
227        cursor = match tag & 7 {
228            0 => read_varint(bytes, after_tag)?.1,
229            1 => advance(bytes, after_tag, 8)?,
230            5 => advance(bytes, after_tag, 4)?,
231            2 => {
232                let (len, after_len) = read_varint(bytes, after_tag)?;
233                let len = usize::try_from(len).ok()?;
234                let end = advance(bytes, after_len, len)?;
235                let payload = bytes.get(after_len..end)?;
236                if payload.len() >= MIN_NESTED_PAYLOAD_LEN
237                    && (is_mostly_text(payload) || is_protobuf(payload, depth + 1))
238                {
239                    parsed.nested_structure = true;
240                }
241                end
242            },
243            _ => return None,
244        };
245        parsed.fields += 1;
246    }
247    Some(parsed)
248}
249
250fn advance(bytes: &[u8], cursor: usize, by: usize) -> Option<usize> {
251    let end = cursor.checked_add(by)?;
252    (end <= bytes.len()).then_some(end)
253}
254
255fn read_varint(bytes: &[u8], start: usize) -> Option<(u64, usize)> {
256    let mut value = 0u64;
257    let mut shift = 0u32;
258    let mut cursor = start;
259    loop {
260        let byte = *bytes.get(cursor)?;
261        cursor += 1;
262        value |= u64::from(byte & 0x7f) << shift;
263        if byte & 0x80 == 0 {
264            return Some((value, cursor));
265        }
266        shift += 7;
267        if shift >= 64 {
268            return None;
269        }
270    }
271}