1use regex::Regex;
50use serde::{Deserialize, Serialize};
51use sha2::{Digest, Sha256};
52use std::borrow::Cow;
53use std::collections::HashMap;
54use std::ops::Range;
55use std::path::Path;
56use std::sync::OnceLock;
57
58const TRUNCATION_MARKER: &str = "… [truncated]";
60
61const REDACTED: &str = "[REDACTED]";
63
64pub const SECRET_ALLOWLIST_PATH: &str = ".kranz/secret-allowlist";
66
67#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
69#[serde(rename_all = "camelCase")]
70pub struct SecretFinding {
71 pub rule_id: String,
72 pub fingerprint: String,
73 pub location: String,
74 pub start: usize,
75 pub end: usize,
76}
77
78#[derive(Debug, Clone, PartialEq, Eq)]
80pub struct SecretScan {
81 pub redacted: String,
82 pub findings: Vec<SecretFinding>,
83}
84
85struct Rule {
89 id: &'static str,
90 re: Regex,
91 replacement: &'static str,
92 secret_group: Option<usize>,
93}
94
95fn rule(id: &'static str, pattern: &str, replacement: &'static str) -> Rule {
96 Rule {
97 id,
98 re: Regex::new(pattern).expect("static scrub regex must compile"),
99 replacement,
100 secret_group: None,
101 }
102}
103
104fn grouped_rule(
105 id: &'static str,
106 pattern: &str,
107 replacement: &'static str,
108 secret_group: usize,
109) -> Rule {
110 Rule {
111 id,
112 re: Regex::new(pattern).expect("static scrub regex must compile"),
113 replacement,
114 secret_group: Some(secret_group),
115 }
116}
117
118fn rules() -> &'static [Rule] {
124 static RULES: OnceLock<Vec<Rule>> = OnceLock::new();
125 RULES.get_or_init(|| {
126 vec![
127 rule(
130 "pem-private-key",
131 r"-----BEGIN [A-Z ]*PRIVATE KEY-----[\s\S]*?-----END [A-Z ]*PRIVATE KEY-----|-----BEGIN [A-Z ]*PRIVATE KEY-----[^\r\n]*",
132 REDACTED,
133 ),
134 grouped_rule(
138 "gcp-private-key-json",
139 r#"(?i)("private_key"\s*:\s*")(-----BEGIN[^"]*)"#,
140 "${1}[REDACTED]",
141 2,
142 ),
143 rule("anthropic-api-key", r"\bsk-ant-[A-Za-z0-9_-]{8,}", REDACTED),
145 rule(
149 "openai-project-key",
150 r"\bsk-proj-[A-Za-z0-9_-]{20,}",
151 REDACTED,
152 ),
153 rule("openai-api-key", r"\bsk-[A-Za-z0-9]{20,}", REDACTED),
155 rule("google-api-key", r"\bAIza[0-9A-Za-z_-]{35}\b", REDACTED),
157 rule(
159 "stripe-live-key",
160 r"\b(?:sk|rk|pk)_live_[0-9A-Za-z]{16,}",
161 REDACTED,
162 ),
163 rule("npm-token", r"\bnpm_[0-9A-Za-z]{36}\b", REDACTED),
165 rule("github-token", r"\bgh[pos]_[A-Za-z0-9]{20,}", REDACTED),
167 rule(
169 "github-fine-grained-token",
170 r"\bgithub_pat_[A-Za-z0-9_]{20,}",
171 REDACTED,
172 ),
173 rule("aws-access-key-id", r"\bAKIA[0-9A-Z]{16}\b", REDACTED),
175 grouped_rule(
177 "aws-secret-access-key",
178 r"(?i)\b(aws_secret_access_key\s*[=:]\s*)(\S+)",
179 "${1}[REDACTED]",
180 2,
181 ),
182 rule("slack-token", r"\bxox[baprs]-[A-Za-z0-9-]{10,}", REDACTED),
184 rule("sgian-client-token", r"\bsgc_[0-9a-fA-F]{64}\b", REDACTED),
186 rule(
188 "jwt",
189 r"\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{5,}",
190 REDACTED,
191 ),
192 grouped_rule(
194 "authorization-bearer",
195 r"(?i)\b(bearer\s+)([a-z0-9._~+/=-]{16,})",
196 "${1}[REDACTED]",
197 2,
198 ),
199 grouped_rule(
201 "authorization-basic",
202 r"(?i)\b(basic\s+)([a-z0-9+/]{16,}={0,2})",
203 "${1}[REDACTED]",
204 2,
205 ),
206 grouped_rule(
210 "connection-string-password",
211 r"([a-zA-Z][a-zA-Z0-9+.-]*://[^\s:/@]+:)([^\s:/@]+)(@)",
212 "${1}[REDACTED]${3}",
213 2,
214 ),
215 ]
216 })
217}
218
219fn generic_assignment_re() -> &'static Regex {
226 static RE: OnceLock<Regex> = OnceLock::new();
227 RE.get_or_init(|| {
228 Regex::new(
229 r#"(?i)((?:api[_-]?key|secret|token|password|passwd|credential)["']?\s*[:=]\s*["']?)([^\s"']{8,})"#,
230 )
231 .expect("generic assignment regex must compile")
232 })
233}
234
235fn entropy_assignment_re() -> &'static Regex {
241 static RE: OnceLock<Regex> = OnceLock::new();
242 RE.get_or_init(|| {
243 Regex::new(
244 r#"(?i)((?:access[_-]?token|auth[_-]?token|auth|client[_-]?secret|private[_-]?key)["']?\s*[:=]\s*["']?)([A-Za-z0-9+/_=-]{24,})"#,
245 )
246 .expect("entropy assignment regex must compile")
247 })
248}
249
250pub(crate) fn shannon_entropy(s: &str) -> f64 {
255 if s.is_empty() {
256 return 0.0;
257 }
258 let mut counts: HashMap<char, usize> = HashMap::new();
259 for c in s.chars() {
260 *counts.entry(c).or_insert(0) += 1;
261 }
262 let len = s.chars().count() as f64;
263 counts
264 .values()
265 .map(|&count| {
266 let p = count as f64 / len;
267 -p * p.log2()
268 })
269 .sum()
270}
271
272fn looks_like_secret_charset(s: &str) -> bool {
277 !s.is_empty()
278 && s.chars()
279 .all(|c| c.is_ascii_alphanumeric() || matches!(c, '+' | '/' | '_' | '-' | '='))
280}
281
282fn is_allowlisted(value: &str) -> bool {
287 let lower = value.to_ascii_lowercase();
288
289 const PLACEHOLDERS: &[&str] = &[
291 "xxxx",
292 "replace",
293 "example",
294 "changeme",
295 "your",
296 "dummy",
297 "placeholder",
298 "todo",
299 "none",
300 "redacted",
301 ];
302 if PLACEHOLDERS.iter().any(|p| lower.contains(p)) {
303 return true;
304 }
305
306 if let Some(first) = value.chars().next() {
308 if value.chars().all(|c| c == first) {
309 return true;
310 }
311 }
312
313 if is_uuid(value) {
315 return true;
316 }
317
318 if value.len() == 40 && value.chars().all(|c| c.is_ascii_hexdigit()) && lower == value {
321 return true;
322 }
323
324 false
325}
326
327fn is_uuid(s: &str) -> bool {
329 let groups: Vec<&str> = s.split('-').collect();
330 if groups.len() != 5 {
331 return false;
332 }
333 let widths = [8usize, 4, 4, 4, 12];
334 groups
335 .iter()
336 .zip(widths)
337 .all(|(g, w)| g.len() == w && g.chars().all(|c| c.is_ascii_hexdigit()))
338}
339
340fn is_high_entropy_secret(value: &str) -> bool {
344 value.len() >= 24
345 && looks_like_secret_charset(value)
346 && !is_allowlisted(value)
347 && shannon_entropy(value) >= 4.0
348}
349
350fn secret_fingerprint(rule_id: &str, value: &str) -> String {
351 let mut hasher = Sha256::new();
352 hasher.update(rule_id.as_bytes());
353 hasher.update([0]);
354 hasher.update(value.as_bytes());
355 let digest = hasher.finalize();
356 digest[..12].iter().map(|b| format!("{b:02x}")).collect()
357}
358
359fn push_finding(
360 out: &mut Vec<SecretFinding>,
361 occupied: &mut Vec<Range<usize>>,
362 rule_id: &str,
363 location: &str,
364 range: Range<usize>,
365 value: &str,
366) {
367 if is_allowlisted(value) {
368 return;
369 }
370 if occupied
371 .iter()
372 .any(|existing| existing.start < range.end && range.start < existing.end)
373 {
374 return;
375 }
376 occupied.push(range.clone());
377 out.push(SecretFinding {
378 rule_id: rule_id.to_string(),
379 fingerprint: secret_fingerprint(rule_id, value),
380 location: location.to_string(),
381 start: range.start,
382 end: range.end,
383 });
384}
385
386pub fn scan_text_at(text: &str, location: &str) -> Vec<SecretFinding> {
388 scan_text_with_assignments(text, text, location)
389}
390
391fn scan_text_with_assignments(
392 text: &str,
393 assignment_text: &str,
394 location: &str,
395) -> Vec<SecretFinding> {
396 let mut out = Vec::new();
397 let mut occupied: Vec<Range<usize>> = Vec::new();
398 for rule in rules() {
399 for caps in rule.re.captures_iter(text) {
400 let m = rule
401 .secret_group
402 .and_then(|idx| caps.get(idx))
403 .or_else(|| caps.get(0));
404 if let Some(m) = m {
405 push_finding(
406 &mut out,
407 &mut occupied,
408 rule.id,
409 location,
410 m.start()..m.end(),
411 m.as_str(),
412 );
413 }
414 }
415 }
416
417 for caps in generic_assignment_re().captures_iter(assignment_text) {
418 if let Some(value) = caps.get(2) {
419 push_finding(
420 &mut out,
421 &mut occupied,
422 "generic-secret-assignment",
423 location,
424 value.start()..value.end(),
425 value.as_str(),
426 );
427 }
428 }
429
430 for caps in entropy_assignment_re().captures_iter(assignment_text) {
431 if let Some(value) = caps.get(2) {
432 if is_high_entropy_secret(value.as_str()) {
433 push_finding(
434 &mut out,
435 &mut occupied,
436 "high-entropy-secret-assignment",
437 location,
438 value.start()..value.end(),
439 value.as_str(),
440 );
441 }
442 }
443 }
444 out
445}
446
447pub fn scan_text(text: &str) -> Vec<SecretFinding> {
449 scan_text_at(text, "text")
450}
451
452fn scrub_assignments(text: &str) -> Cow<'_, str> {
457 generic_assignment_re().replace_all(text, |caps: ®ex::Captures<'_>| {
458 let prefix = &caps[1];
459 let value = &caps[2];
460 if is_allowlisted(value) {
461 caps[0].to_owned()
462 } else {
463 format!("{prefix}{REDACTED}")
464 }
465 })
466}
467
468fn scrub_entropy(text: &str) -> Cow<'_, str> {
474 entropy_assignment_re().replace_all(text, |caps: ®ex::Captures<'_>| {
475 let prefix = &caps[1];
476 let value = &caps[2];
477 if is_high_entropy_secret(value) {
478 format!("{prefix}{REDACTED}")
479 } else {
480 caps[0].to_owned()
481 }
482 })
483}
484
485fn scrub_plain(text: &str) -> String {
494 let mut out = text.to_owned();
495 for rule in rules() {
496 if let Cow::Owned(replaced) = rule.re.replace_all(&out, rule.replacement) {
497 out = replaced;
498 }
499 }
500 if let Cow::Owned(replaced) = scrub_assignments(&out) {
503 out = replaced;
504 }
505 if let Cow::Owned(replaced) = scrub_entropy(&out) {
510 out = replaced;
511 }
512 out
513}
514
515fn scrub_json_text(text: &str) -> Option<String> {
519 serde_json::from_str::<serde_json::Value>(text).ok()?;
523 let bytes = text.as_bytes();
524 let mut cursor = 0;
525 let mut copied = 0;
526 let mut out = String::new();
527 let mut key: Option<String> = None;
528 while cursor < bytes.len() {
529 let start = cursor;
530 if bytes[cursor] == b'"' {
531 cursor += 1;
532 while cursor < bytes.len() {
533 match bytes[cursor] {
534 b'\\' => cursor += 2,
535 b'"' => {
536 cursor += 1;
537 break;
538 }
539 _ => cursor += 1,
540 }
541 }
542 let decoded: String = serde_json::from_str(&text[start..cursor]).ok()?;
543 let is_key = text[cursor..].trim_start().starts_with(':');
544 let redacted = if is_key {
545 scrub_plain(&decoded)
546 } else {
547 scrub_json_assignment(scrub_impl(&decoded), key.as_deref())
548 };
549 if redacted != decoded {
550 out.push_str(&text[copied..start]);
551 out.push_str(
554 &serde_json::to_string(&redacted)
555 .ok()?
556 .replace('<', "\\u003c")
557 .replace('>', "\\u003e"),
558 );
559 copied = cursor;
560 }
561 key = is_key.then_some(decoded);
562 } else if bytes[cursor].is_ascii_whitespace() || bytes[cursor] == b':' {
563 cursor += 1;
564 } else {
565 if key.is_some() && matches!(bytes[cursor], b'-' | b'0'..=b'9') {
568 while cursor < bytes.len()
569 && matches!(
570 bytes[cursor],
571 b'-' | b'+' | b'.' | b'e' | b'E' | b'0'..=b'9'
572 )
573 {
574 cursor += 1;
575 }
576 let value = &text[start..cursor];
577 let redacted = scrub_json_assignment(value.to_owned(), key.as_deref());
578 if redacted != value {
579 out.push_str(&text[copied..start]);
580 out.push_str(&serde_json::to_string(&redacted).ok()?);
581 copied = cursor;
582 }
583 } else {
584 cursor += 1;
585 }
586 key = None;
587 }
588 }
589 out.push_str(&text[copied..]);
590 Some(out)
591}
592
593fn scrub_json_assignment(mut value: String, key: Option<&str>) -> String {
594 if let Some(key) = key {
595 let prefix = format!("{key}=\"");
598 let contextual = format!("{prefix}{value}");
599 for (regex, entropy_only) in [
600 (generic_assignment_re(), false),
601 (entropy_assignment_re(), true),
602 ] {
603 let Some(caps) = regex.captures(&contextual) else {
604 continue;
605 };
606 let candidate = caps.get(2).expect("assignment value capture");
607 if candidate.start() == prefix.len()
608 && if entropy_only {
609 is_high_entropy_secret(candidate.as_str())
610 } else {
611 !is_allowlisted(candidate.as_str())
612 }
613 {
614 value.replace_range(..candidate.len(), REDACTED);
615 break;
616 }
617 }
618 }
619 value
620}
621
622fn scrub_impl(text: &str) -> String {
623 if let Some(redacted) = scrub_json_text(text) {
624 return redacted;
625 }
626 let mut out = String::new();
629 let mut plain_start = 0;
630 let mut body_start = None;
631 let mut cursor = 0;
632 for line in text.split_inclusive('\n') {
633 let start = cursor;
634 cursor += line.len();
635 let trimmed = line.trim();
636 if body_start.is_none() && matches!(trimmed, "```" | "```json" | "```JSON") {
637 body_start = Some(cursor);
638 } else if trimmed == "```" {
639 if let Some(body) = body_start.take() {
640 if let Some(redacted) = scrub_json_text(&text[body..start]) {
641 out.push_str(&scrub_plain(&text[plain_start..body]));
642 out.push_str(&redacted);
643 if !redacted.ends_with('\n') {
645 out.push('\n');
646 }
647 plain_start = start;
648 }
649 }
650 }
651 }
652 out.push_str(&scrub_plain(&text[plain_start..]));
653 out
654}
655
656pub fn scrub_with_findings(text: &str, location: &str) -> SecretScan {
658 SecretScan {
659 redacted: scrub_impl(text),
660 findings: scan_text_at(text, location),
661 }
662}
663
664pub fn scrub(text: &str) -> String {
665 scrub_impl(text)
666}
667
668pub fn scrub_json_value(value: &mut serde_json::Value, location: &str) -> Vec<SecretFinding> {
671 fn walk(value: &mut serde_json::Value, path: String, findings: &mut Vec<SecretFinding>) {
672 match value {
673 serde_json::Value::String(s) => {
674 let scan = scrub_with_findings(s, &path);
675 *s = scan.redacted;
676 findings.extend(scan.findings);
677 }
678 serde_json::Value::Array(items) => {
679 for (idx, item) in items.iter_mut().enumerate() {
680 walk(item, format!("{path}/{idx}"), findings);
681 }
682 }
683 serde_json::Value::Object(map) => {
684 for (key, item) in map.iter_mut() {
685 walk(item, format!("{path}/{key}"), findings);
686 }
687 }
688 serde_json::Value::Null | serde_json::Value::Bool(_) | serde_json::Value::Number(_) => {
689 }
690 }
691 }
692
693 let mut findings = Vec::new();
694 walk(value, location.to_string(), &mut findings);
695 findings
696}
697
698const GENERATED_DIFF_PATH_PREFIXES: &[&str] =
702 &["apps/dashboard/dist/", "crates/cli/assets/dashboard/dist/"];
703
704pub fn scan_unified_diff(diff: &str) -> Vec<SecretFinding> {
706 let mut findings = Vec::new();
707 let mut path = "<diff>".to_string();
708 let mut generated_dashboard_bundle = false;
709 let mut new_line: Option<usize> = None;
710
711 for line in diff.lines() {
712 if let Some(rest) = line.strip_prefix("+++ b/") {
713 path = rest.to_string();
714 generated_dashboard_bundle = GENERATED_DIFF_PATH_PREFIXES
715 .iter()
716 .any(|prefix| path.starts_with(prefix));
717 continue;
718 }
719 if line.starts_with("@@ ") {
720 new_line = parse_new_hunk_start(line);
721 continue;
722 }
723 if line.starts_with("+++") {
724 continue;
725 }
726 if let Some(added) = line.strip_prefix('+') {
727 let line_no = new_line.unwrap_or(0);
728 let location = if line_no == 0 {
729 path.clone()
730 } else {
731 format!("{path}:{line_no}")
732 };
733 let mut line_findings = scan_text_at(added, &location);
734 if generated_dashboard_bundle {
735 line_findings.retain(|finding| finding.rule_id != "generic-secret-assignment");
736 }
737 findings.extend(line_findings);
738 if let Some(n) = &mut new_line {
739 *n += 1;
740 }
741 } else if !line.starts_with('-') {
742 if let Some(n) = &mut new_line {
743 *n += 1;
744 }
745 }
746 }
747
748 findings
749}
750
751fn parse_new_hunk_start(line: &str) -> Option<usize> {
752 let plus = line.split_whitespace().find(|part| part.starts_with('+'))?;
753 let number = plus
754 .trim_start_matches('+')
755 .split(',')
756 .next()
757 .filter(|s| !s.is_empty())?;
758 number.parse().ok()
759}
760
761pub fn read_allowlist_text(text: &str) -> std::collections::BTreeSet<String> {
762 text.lines()
763 .map(str::trim)
764 .filter(|line| !line.is_empty() && !line.starts_with('#'))
765 .filter_map(|line| line.split_whitespace().next())
766 .map(str::to_string)
767 .collect()
768}
769
770pub fn filter_allowed(
771 findings: Vec<SecretFinding>,
772 allowed: &std::collections::BTreeSet<String>,
773) -> Vec<SecretFinding> {
774 findings
775 .into_iter()
776 .filter(|f| !allowed.contains(&f.fingerprint))
777 .collect()
778}
779
780pub fn format_findings(findings: &[SecretFinding]) -> String {
781 findings
782 .iter()
783 .map(|finding| {
784 format!(
785 "{} [{}] {} bytes {}..{}",
786 finding.fingerprint, finding.rule_id, finding.location, finding.start, finding.end
787 )
788 })
789 .collect::<Vec<_>>()
790 .join("\n")
791}
792
793const SCAN_PATH_MAX_FILE_BYTES: u64 = 8 * 1024 * 1024;
800
801fn read_scan_candidate(path: &Path) -> Option<Vec<u8>> {
821 use std::io::Read as _;
822 let (parent, name) = crate::paths::open_parent_nofollow(path).ok()?;
823 let mut options = cap_std::fs::OpenOptions::new();
824 {
825 use cap_fs_ext::OpenOptionsFollowExt as _;
826 use cap_primitives::fs::FollowSymlinks;
827 options.read(true).follow(FollowSymlinks::No);
828 }
829 #[cfg(unix)]
830 {
831 use cap_fs_ext::OpenOptionsExt as _;
832 options.custom_flags(libc::O_NONBLOCK);
833 }
834 let file = parent.open_with(name, &options).ok()?.into_std();
835 let metadata = file.metadata().ok()?;
836 if !metadata.file_type().is_file() || metadata.len() > SCAN_PATH_MAX_FILE_BYTES {
837 return None;
838 }
839 let mut buf = Vec::new();
840 (&mut &file)
841 .take(SCAN_PATH_MAX_FILE_BYTES + 1)
842 .read_to_end(&mut buf)
843 .ok()?;
844 if buf.len() as u64 > SCAN_PATH_MAX_FILE_BYTES {
845 return None;
846 }
847 Some(buf)
848}
849
850pub fn scan_paths(repo_root: &Path, paths: &[&Path]) -> Vec<SecretFinding> {
863 let mut findings = Vec::new();
864 for path in paths {
865 let full = if path.is_absolute() {
866 path.to_path_buf()
867 } else {
868 repo_root.join(path)
869 };
870 let Some(bytes) = read_scan_candidate(&full) else {
871 continue;
872 };
873 let text = String::from_utf8_lossy(&bytes);
874 let location = full
875 .strip_prefix(repo_root)
876 .ok()
877 .and_then(|p| p.to_str())
878 .unwrap_or_else(|| full.to_str().unwrap_or("<path>"));
879 let assignments = if full.extension().is_some_and(|ext| ext == "py") {
880 python_assignment_text(&text)
881 } else {
882 Cow::Borrowed(text.as_ref())
883 };
884 findings.extend(scan_text_with_assignments(&text, &assignments, location));
885 }
886 findings
887}
888
889fn python_assignment_text(text: &str) -> Cow<'_, str> {
895 let mut out = Cow::Borrowed(text);
896 let mut offset = 0;
897 for line in text.split_inclusive('\n') {
898 let trimmed = line.trim_end();
899 let keyword = trimmed.split_whitespace().next().unwrap_or("");
900 if trimmed.ends_with(':')
901 && matches!(
902 keyword,
903 "if" | "elif" | "while" | "for" | "with" | "except" | "class" | "match" | "case"
904 )
905 {
906 let colon = offset + trimmed.len() - 1;
907 out.to_mut().replace_range(colon..colon + 1, " ");
908 }
909 offset += line.len();
910 }
911 out
912}
913
914pub fn truncate_chars(text: &str, max: usize) -> String {
918 match text.char_indices().nth(max) {
919 None => text.to_owned(),
921 Some((cut_at, _)) => {
922 let mut out = String::with_capacity(cut_at + TRUNCATION_MARKER.len());
923 out.push_str(&text[..cut_at]);
924 out.push_str(TRUNCATION_MARKER);
925 out
926 }
927 }
928}
929
930pub fn scrub_and_truncate(text: &str, max: usize) -> String {
933 truncate_chars(&scrub(text), max)
934}
935
936#[cfg(test)]
937mod tests {
938 use super::*;
939
940 #[test]
941 fn entropy_of_empty_is_zero() {
942 assert_eq!(shannon_entropy(""), 0.0);
943 }
944
945 #[test]
946 fn entropy_of_uniform_string_is_zero() {
947 assert_eq!(shannon_entropy("aaaaaaaa"), 0.0);
948 }
949
950 #[test]
951 fn entropy_of_random_base64_is_high() {
952 let e = shannon_entropy("aB3xQ9zK7mP2wR5tY8uV1nJ4kL6dF0sG");
954 assert!(e >= 4.0, "entropy too low: {e}");
955 }
956
957 #[test]
958 fn entropy_of_english_word_is_low() {
959 let e = shannon_entropy("bureaucracy");
960 assert!(e < 4.0, "prose entropy unexpectedly high: {e}");
961 }
962
963 #[test]
964 fn uuid_recognized() {
965 assert!(is_uuid("550e8400-e29b-41d4-a716-446655440000"));
966 assert!(!is_uuid("not-a-uuid"));
967 assert!(!is_uuid("550e8400e29b41d4a716446655440000"));
968 }
969
970 #[test]
971 fn allowlist_covers_placeholders_and_shas() {
972 assert!(is_allowlisted("REPLACE_ME_WITH_REAL_KEY_1234567890"));
973 assert!(is_allowlisted("xxxxxxxxxxxxxxxxxxxxxxxx"));
974 assert!(is_allowlisted("aaaaaaaaaaaaaaaaaaaaaaaa"));
975 assert!(is_allowlisted("550e8400-e29b-41d4-a716-446655440000"));
976 assert!(is_allowlisted("da39a3ee5e6b4b0d3255bfef95601890afd80709"));
978 }
979
980 const SCRUB_NOFOLLOW_SECRET: &str = "sk-ant-api03-ScrubNofollowTestValue1";
989
990 fn scan_with_timeout(root: &Path, paths: &[&Path], secs: u64) -> Vec<SecretFinding> {
996 let root = root.to_path_buf();
997 let paths: Vec<std::path::PathBuf> = paths.iter().map(|p| p.to_path_buf()).collect();
998 let (tx, rx) = std::sync::mpsc::channel();
999 std::thread::spawn(move || {
1000 let refs: Vec<&Path> = paths.iter().map(std::path::PathBuf::as_path).collect();
1001 let _ = tx.send(scan_paths(&root, &refs));
1002 });
1003 rx.recv_timeout(std::time::Duration::from_secs(secs))
1004 .expect("scan_paths must not block")
1005 }
1006
1007 #[cfg(unix)]
1011 #[test]
1012 fn scrub_nofollow_fifo_does_not_block_checkpoint_scan() {
1013 let dir = tempfile::tempdir().unwrap();
1014 let fifo = dir.path().join("planted.fifo");
1015 let c_path = std::ffi::CString::new(fifo.to_str().expect("utf-8 temp path")).unwrap();
1016 let rc = unsafe { libc::mkfifo(c_path.as_ptr(), 0o644) };
1017 assert_eq!(rc, 0, "mkfifo failed: {}", std::io::Error::last_os_error());
1018
1019 let findings = scan_with_timeout(dir.path(), &[Path::new("planted.fifo")], 10);
1020 assert!(
1021 findings.is_empty(),
1022 "a FIFO is skipped, never scanned: {findings:?}"
1023 );
1024 }
1025
1026 #[cfg(unix)]
1029 #[test]
1030 fn scrub_nofollow_symlink_to_dev_zero_is_skipped() {
1031 let dir = tempfile::tempdir().unwrap();
1032 std::os::unix::fs::symlink("/dev/zero", dir.path().join("zero")).unwrap();
1033
1034 let findings = scan_with_timeout(dir.path(), &[Path::new("zero")], 10);
1035 assert!(
1036 findings.is_empty(),
1037 "a symlink to an unbounded source is skipped, never read through: {findings:?}"
1038 );
1039 }
1040
1041 #[cfg(unix)]
1045 #[test]
1046 fn scrub_nofollow_symlinked_file_is_not_read_through() {
1047 let dir = tempfile::tempdir().unwrap();
1048 let outside = tempfile::tempdir().unwrap();
1049 let real = outside.path().join("real.txt");
1050 std::fs::write(&real, SCRUB_NOFOLLOW_SECRET).unwrap();
1051 std::os::unix::fs::symlink(&real, dir.path().join("linked.txt")).unwrap();
1052
1053 let findings = scan_paths(dir.path(), &[Path::new("linked.txt")]);
1054 assert!(
1055 findings.is_empty(),
1056 "a symlink is never read through: {findings:?}"
1057 );
1058 let findings = scan_paths(dir.path(), &[real.as_path()]);
1060 assert!(
1061 findings.iter().any(|f| f.rule_id == "anthropic-api-key"),
1062 "the direct scan must flag the secret: {findings:?}"
1063 );
1064 }
1065
1066 #[test]
1071 fn scrub_nofollow_oversized_file_is_skipped_and_under_cap_scans() {
1072 let dir = tempfile::tempdir().unwrap();
1073 let mut content = SCRUB_NOFOLLOW_SECRET.as_bytes().to_vec();
1074 content.resize(SCAN_PATH_MAX_FILE_BYTES as usize + 1, b'x');
1075 std::fs::write(dir.path().join("big.txt"), &content).unwrap();
1076
1077 let findings = scan_with_timeout(dir.path(), &[Path::new("big.txt")], 10);
1078 assert!(
1079 findings.is_empty(),
1080 "an oversized file is skipped whole, never partially scanned: {findings:?}"
1081 );
1082
1083 std::fs::write(dir.path().join("small.txt"), SCRUB_NOFOLLOW_SECRET).unwrap();
1085 let findings = scan_paths(dir.path(), &[Path::new("small.txt")]);
1086 assert!(
1087 findings.iter().any(|f| f.rule_id == "anthropic-api-key"),
1088 "under-cap content still scans: {findings:?}"
1089 );
1090 }
1091
1092 #[test]
1099 fn composition_audit_secret_allowlist_waives_one_fingerprint_never_a_rule() {
1100 let text_a = "sk-ant-api03-CompositionAuditValueA1";
1101 let text_b = "sk-ant-api03-CompositionAuditValueB2";
1102 let findings = scan_text(&format!("{text_a} {text_b}"));
1103 assert_eq!(findings.len(), 2, "both keys must be found: {findings:?}");
1104
1105 let waived: std::collections::BTreeSet<String> =
1108 [findings[0].fingerprint.clone()].into_iter().collect();
1109 let remaining = filter_allowed(findings, &waived);
1110 assert_eq!(remaining.len(), 1);
1111 assert_eq!(remaining[0].rule_id, "anthropic-api-key");
1112
1113 let findings = scan_text(text_a);
1115 assert_eq!(
1116 filter_allowed(findings.clone(), &Default::default()),
1117 findings
1118 );
1119 let garbage = read_allowlist_text("# reviewed\nnot-a-fingerprint\n");
1120 assert_eq!(filter_allowed(findings.clone(), &garbage), findings);
1121 }
1122}