use super::{
contains_bounded_word, contains_word, SecretMatch, COMPOUND_TRIGGER_WORDS, TRIGGER_WORDS,
};
pub(super) fn is_assignment_label_gap(c: char) -> bool {
matches!(c, '"' | '\'' | '`')
}
pub(super) const LOOKUP_KEY_LABELS: &[&str] = &[
"association_key",
"cache_key",
"composite_key",
"dedup_key",
"foreign_key",
"idempotency_key",
"index_key",
"lookup_key",
"map_key",
"natural_key",
"partition_key",
"primary_key",
"range_key",
"routing_key",
"row_key",
"search_key",
"shard_key",
"sort_key",
"surrogate_key",
"unique_key",
];
pub(super) fn is_lookup_key_label(label: &str) -> bool {
LOOKUP_KEY_LABELS.contains(&label)
}
fn compound_trigger(low_text: &str) -> Option<&'static str> {
COMPOUND_TRIGGER_WORDS.iter().copied().find(|needle| {
let mut start = 0;
while let Some(rel) = low_text[start..].find(needle) {
let abs = start + rel;
let before_ok = abs == 0
|| low_text[..abs]
.chars()
.next_back()
.is_none_or(|c| !c.is_ascii_alphanumeric());
if before_ok {
return true;
}
start = abs + needle.len();
}
false
})
}
pub(super) fn find_trigger(text: &str, credential_label_only: bool) -> Option<&'static str> {
let low = text.to_ascii_lowercase();
TRIGGER_WORDS
.iter()
.copied()
.find(|tw| contains_bounded_word(&low, tw))
.or_else(|| compound_trigger(&low))
.or_else(|| {
((!credential_label_only && has_standalone_token(&low)) || has_token_assignment(&low))
.then_some("token")
})
.or_else(|| assignment_credential_trigger(&low))
}
pub(super) fn assignment_credential_trigger(low_text: &str) -> Option<&'static str> {
low_text.char_indices().find_map(|(index, ch)| {
if !matches!(ch, '=' | ':') {
return None;
}
let before =
low_text[..index].trim_end_matches(|c: char| !c.is_ascii_alphanumeric() && c != '_');
let label = before
.rsplit(|c: char| !c.is_ascii_alphanumeric() && c != '_')
.next()
.unwrap_or_default();
COMPOUND_TRIGGER_WORDS
.iter()
.copied()
.find(|needle| label.contains(needle))
.or_else(|| {
TRIGGER_WORDS.iter().copied().find(|tw| {
(*tw != "key"
|| !is_lookup_key_label(label)
|| !low_text[before.len()..index]
.chars()
.all(is_assignment_label_gap))
&& contains_bounded_word(label, tw)
})
})
.or_else(|| (label == "token").then_some("token"))
})
}
pub(super) fn inline_credential_trigger(raw_token: &str) -> Option<&'static str> {
let low = raw_token.to_ascii_lowercase();
assignment_credential_trigger(&low).or_else(|| {
if !low.contains(['/', '-', '.']) && low.contains('_') {
COMPOUND_TRIGGER_WORDS
.iter()
.copied()
.find(|needle| low.contains(needle))
.or_else(|| {
TRIGGER_WORDS
.iter()
.copied()
.find(|tw| contains_bounded_word(&low, tw))
})
} else {
None
}
})
}
fn has_standalone_token(low_window: &str) -> bool {
contains_word(low_window, "token", true)
}
fn has_token_assignment(low_window: &str) -> bool {
let needle = "token";
let mut start = 0;
while let Some(rel) = low_window[start..].find(needle) {
let abs = start + rel;
let before_ok = abs == 0
|| low_window[..abs]
.chars()
.next_back()
.is_none_or(|c| !c.is_ascii_alphanumeric() && c != '_');
let after_end = abs + needle.len();
let after_char = low_window[after_end..].chars().next();
let after_is_assign = matches!(after_char, Some('=') | Some(':'));
if before_ok && after_is_assign {
return true;
}
start = abs + needle.len().max(1);
}
false
}
pub(super) fn is_pure_hex(token: &str) -> bool {
let hex_part = token
.strip_prefix("0x")
.or(token.strip_prefix("0X"))
.unwrap_or(token);
hex_part.len() >= 8 && hex_part.len() <= 128 && hex_part.bytes().all(|b| b.is_ascii_hexdigit())
}
pub(super) fn is_base64_content_hash(token: &str) -> bool {
const VENDOR_PREFIXES: &[&str] = &[
"sk-",
"rk_live_",
"fm2_",
"vercel_",
"xoxb-",
"xoxa-",
"xoxp-",
"xoxr-",
"xoxs-",
"ghp_",
"gho_",
"ghu_",
"ghs_",
"ghr_",
"github_pat_",
"AKIA",
"ASIA",
"AGE-SECRET-KEY-",
"FlyV1",
];
if VENDOR_PREFIXES.iter().any(|p| token.starts_with(p)) {
return false;
}
let body = if let Some(rest) = token.strip_prefix("sha") {
let dash = rest.find('-').unwrap_or(rest.len());
let digits = &rest[..dash];
if !digits.is_empty() && digits.bytes().all(|b| b.is_ascii_digit()) && dash < rest.len() {
&rest[dash + 1..] } else {
return false; }
} else {
return false; };
let stripped = body.trim_end_matches('=');
let pad_removed = body.len() - stripped.len();
if pad_removed > 2 {
return false;
}
let n = stripped.len();
if n != 43 && n != 64 && !(86..=88).contains(&n) {
return false;
}
stripped
.bytes()
.all(|b| b.is_ascii_alphanumeric() || b == b'+' || b == b'/' || b == b'-' || b == b'_')
}
const STRUCTURAL_SEPARATORS: [char; 4] = ['/', '-', '_', '.'];
const MAX_RUN_LEN: usize = 24;
const DENSITY_EXEMPT_LETTER_LEN: usize = 4;
const MAX_CASE_TRANSITION_DENSITY: f64 = 0.3;
pub(super) fn is_structured_identifier(token: &str) -> bool {
if !token.contains(|c: char| STRUCTURAL_SEPARATORS.contains(&c)) {
return false;
}
let runs: Vec<&str> = token
.split(|c: char| !c.is_ascii_alphanumeric())
.filter(|r| !r.is_empty())
.collect();
runs.len() >= 2 && runs.iter().all(|run| is_word_shaped_run(run))
}
fn is_word_shaped_run(run: &str) -> bool {
if run.is_empty() || run.len() > MAX_RUN_LEN {
return false;
}
let bytes = run.as_bytes();
if bytes.iter().all(|b| b.is_ascii_digit()) {
return true;
}
let letter_end = bytes
.iter()
.position(|b| !b.is_ascii_alphabetic())
.unwrap_or(bytes.len());
if letter_end == 0 {
return false;
}
if !bytes[letter_end..].iter().all(|b| b.is_ascii_digit()) {
return false;
}
case_transition_density_ok(&run[..letter_end])
}
fn case_transition_density_ok(letters: &str) -> bool {
let chars: Vec<char> = letters.chars().collect();
if chars.len() <= DENSITY_EXEMPT_LETTER_LEN {
return true;
}
let transitions = chars
.windows(2)
.filter(|w| w[0].is_ascii_uppercase() != w[1].is_ascii_uppercase())
.count();
let density = transitions as f64 / (chars.len() - 1) as f64;
density <= MAX_CASE_TRANSITION_DENSITY
}
pub(super) fn is_uuid_canonical(s: &str) -> bool {
let b = s.as_bytes();
if b.len() != 36 {
return false;
}
b[8] == b'-'
&& b[13] == b'-'
&& b[18] == b'-'
&& b[23] == b'-'
&& b[..8].iter().all(|c| c.is_ascii_hexdigit())
&& b[9..13].iter().all(|c| c.is_ascii_hexdigit())
&& b[14..18].iter().all(|c| c.is_ascii_hexdigit())
&& b[19..23].iter().all(|c| c.is_ascii_hexdigit())
&& b[24..].iter().all(|c| c.is_ascii_hexdigit())
}
pub(super) fn strip_delimiters(s: &str) -> &str {
s.trim_matches(|c| matches!(c, '"' | '\'' | '`' | ':' | '=' | ',' | ';'))
}
fn strip_wrappers(s: &str) -> &str {
s.trim_matches(|c: char| {
matches!(
c,
'{' | '}' | '(' | ')' | '[' | ']' | '"' | '\'' | '`' | '.' | ',' | ';'
)
})
}
pub(super) fn wrapper_strip_repeated(token: &str) -> &str {
let mut cur = token;
loop {
let next = strip_wrappers(cur);
if next == cur {
return cur;
}
cur = next;
}
}
pub(super) fn value_candidates(token: &str) -> impl Iterator<Item = &str> {
let cur = wrapper_strip_repeated(token);
std::iter::once(cur).chain(cur.char_indices().filter_map(move |(i, c)| {
if c == '=' || c == ':' {
let after = strip_wrappers(&cur[i + c.len_utf8()..]);
if !after.is_empty() {
return Some(after);
}
}
None
}))
}
pub(super) fn extract_token(s: &str) -> &str {
let end = s
.find(|c: char| c.is_whitespace() || c == '\n' || c == '\r')
.unwrap_or(s.len());
&s[..end]
}
pub(super) fn shannon_entropy(bytes: &[u8]) -> f64 {
if bytes.is_empty() {
return 0.0;
}
let mut counts = [0u32; 256];
for &b in bytes {
counts[b as usize] += 1;
}
let len = bytes.len() as f64;
counts
.iter()
.filter(|&&c| c > 0)
.map(|&c| {
let p = c as f64 / len;
-p * p.log2()
})
.sum()
}
pub(super) fn build_match(detector: &'static str, candidate: &str) -> SecretMatch {
let chars: Vec<char> = candidate.chars().collect();
let preview: String = chars.iter().take(6).collect();
let masked = format!("{}...{}chars", preview, chars.len());
SecretMatch {
detector,
trigger: None,
masked,
location: None,
}
}