use crate::{
CandidateEvidence, Location, SensitiveCandidate, SensitiveCandidateKind,
validators::utils::is_obvious_placeholder,
};
const GROUP_LEN: usize = 4;
const GROUP_COUNT: usize = 4;
const SEPARATOR_COUNT: usize = GROUP_COUNT - 1;
const RECOVERY_LIKE_LEN: usize = GROUP_LEN * GROUP_COUNT + SEPARATOR_COUNT;
pub(crate) fn detect_sensitive_candidates(source: &str) -> Vec<SensitiveCandidate> {
let bytes = source.as_bytes();
let mut spans = Vec::new();
let mut separator = GROUP_LEN;
while separator < bytes.len() {
if bytes[separator] != b'-' {
separator += 1;
continue;
}
let start = separator - GROUP_LEN;
if start + RECOVERY_LIKE_LEN <= bytes.len() && is_recovery_like_at(bytes, start) {
let end = start + RECOVERY_LIKE_LEN;
spans.push((start, end));
separator = end.saturating_add(GROUP_LEN);
} else {
separator += 1;
}
}
materialize_candidates(source, spans)
}
fn materialize_candidates(source: &str, spans: Vec<(usize, usize)>) -> Vec<SensitiveCandidate> {
let mut candidates = Vec::with_capacity(spans.len());
let mut cursor = 0;
let mut line = 1;
let mut column = 1;
for (start, end) in spans {
advance_position(source, &mut cursor, start, &mut line, &mut column);
let mut location = Location::from_span(start, end);
location.set_position(line, column);
candidates.push(SensitiveCandidate::new(
SensitiveCandidateKind::RecoveryLikeCode,
location,
CandidateEvidence::Structural,
));
}
candidates
}
fn is_recovery_like_at(bytes: &[u8], start: usize) -> bool {
let end = start + RECOVERY_LIKE_LEN;
if !has_token_boundaries(bytes, start, end) {
return false;
}
let token = &bytes[start..end];
if !has_grouped_shape(token) {
return false;
}
let mut compact = [0_u8; GROUP_LEN * GROUP_COUNT];
let mut compact_index = 0;
for &byte in token {
if byte != b'-' {
compact[compact_index] = byte;
compact_index += 1;
}
}
if compact.iter().all(u8::is_ascii_digit) {
return false;
}
if compact.iter().all(u8::is_ascii_hexdigit) {
return false;
}
let compact_str =
core::str::from_utf8(&compact).expect("validated recovery-like bytes are ASCII");
!is_obvious_placeholder(compact_str)
}
fn has_grouped_shape(token: &[u8]) -> bool {
debug_assert_eq!(token.len(), RECOVERY_LIKE_LEN);
for (index, &byte) in token.iter().enumerate() {
let separator = matches!(index, 4 | 9 | 14);
if separator {
if byte != b'-' {
return false;
}
} else if !(byte.is_ascii_uppercase() || byte.is_ascii_digit()) {
return false;
}
}
true
}
fn has_token_boundaries(bytes: &[u8], start: usize, end: usize) -> bool {
let boundary_byte = |byte: u8| byte.is_ascii_alphanumeric() || byte == b'-';
let before_is_clear = start == 0 || !boundary_byte(bytes[start - 1]);
let after_is_clear = end == bytes.len() || !boundary_byte(bytes[end]);
before_is_clear && after_is_clear
}
fn advance_position(
source: &str,
cursor: &mut usize,
target: usize,
line: &mut usize,
column: &mut usize,
) {
debug_assert!(*cursor <= target);
debug_assert!(source.is_char_boundary(*cursor));
debug_assert!(source.is_char_boundary(target));
for character in source[*cursor..target].chars() {
if character == '\n' {
*line += 1;
*column = 1;
} else {
*column += 1;
}
}
*cursor = target;
}
#[cfg(test)]
mod tests {
use super::*;
fn spans(source: &str) -> Vec<&str> {
detect_sensitive_candidates(source)
.into_iter()
.map(|candidate| {
let range = candidate.location().byte_range();
&source[range]
})
.collect()
}
#[test]
fn detects_isolated_grouped_recovery_like_code() {
let source = "ABCD-EFGH-IJKL-MNOP";
let candidates = detect_sensitive_candidates(source);
assert_eq!(candidates.len(), 1);
assert_eq!(
candidates[0].kind(),
SensitiveCandidateKind::RecoveryLikeCode
);
assert_eq!(candidates[0].evidence(), CandidateEvidence::Structural);
assert_eq!(candidates[0].location().byte_range(), 0..19);
assert_eq!(candidates[0].location().line(), 1);
assert_eq!(candidates[0].location().column(), 1);
}
#[test]
fn detects_candidate_inside_multiline_unicode_source() {
let source = "header 😀\nvalue: ABCD-EFGH-IJKL-MNOP\nfooter";
let candidates = detect_sensitive_candidates(source);
assert_eq!(candidates.len(), 1);
assert_eq!(
&source[candidates[0].location().byte_range()],
"ABCD-EFGH-IJKL-MNOP"
);
assert_eq!(candidates[0].location().line(), 2);
assert_eq!(candidates[0].location().column(), 8);
}
#[test]
fn detects_multiple_candidates_in_source_order() {
let source = "ABCD-EFGH-IJKL-MNOP\nQRST-UVWX-YZ12-3456";
assert_eq!(
spans(source),
["ABCD-EFGH-IJKL-MNOP", "QRST-UVWX-YZ12-3456"]
);
}
#[test]
fn rejects_numeric_only_grouped_values() {
assert!(detect_sensitive_candidates("1234-5678-9012-3456").is_empty());
}
#[test]
fn rejects_hexadecimal_only_grouped_values() {
assert!(detect_sensitive_candidates("ABCD-EF12-3456-7890").is_empty());
}
#[test]
fn rejects_lowercase_and_mixed_case_shapes() {
assert!(detect_sensitive_candidates("abcd-EFGH-IJKL-MNOP").is_empty());
assert!(detect_sensitive_candidates("AbCD-EFGH-IJKL-MNOP").is_empty());
}
#[test]
fn rejects_obvious_placeholder_values() {
assert!(detect_sensitive_candidates("XXXX-XXXX-XXXX-XXXX").is_empty());
}
#[test]
fn rejects_partial_or_longer_tokens() {
assert!(detect_sensitive_candidates("ABCD-EFGH-IJKL").is_empty());
assert!(detect_sensitive_candidates("XABCD-EFGH-IJKL-MNOP").is_empty());
assert!(detect_sensitive_candidates("ABCD-EFGH-IJKL-MNOPQ").is_empty());
assert!(detect_sensitive_candidates("ABCD-EFGH-IJKL-MNOP-QRST").is_empty());
}
#[test]
fn accepts_safe_surrounding_punctuation() {
let source = "(ABCD-EFGH-IJKL-MNOP),";
let candidates = detect_sensitive_candidates(source);
assert_eq!(candidates.len(), 1);
assert_eq!(candidates[0].location().byte_range(), 1..20);
}
#[test]
fn preserves_positions_for_multiple_candidates_after_unicode_and_newlines() {
let source = "😀 header\nABCD-EFGH-IJKL-MNOP\nαβγ QRST-UVWX-YZ12-3456\n";
let candidates = detect_sensitive_candidates(source);
assert_eq!(candidates.len(), 2);
assert_eq!(candidates[0].location().line(), 2);
assert_eq!(candidates[0].location().column(), 1);
assert_eq!(candidates[1].location().line(), 3);
assert_eq!(candidates[1].location().column(), 5);
}
#[test]
fn hyphen_heavy_non_candidates_remain_rejected() {
let source = "ordinary-value-with-many-hyphens---still-not-a-recovery-code\n".repeat(256);
assert!(detect_sensitive_candidates(&source).is_empty());
}
#[test]
fn candidate_model_remains_distinct_from_finding_semantics() {
let candidate = detect_sensitive_candidates("ABCD-EFGH-IJKL-MNOP")
.into_iter()
.next()
.expect("candidate should be detected");
assert_eq!(candidate.evidence(), CandidateEvidence::Structural);
let _ = candidate.kind();
let _ = candidate.location();
}
}