use super::{extract_token, keep_leftmost};
pub(super) const PREFIX_DETECTORS: &[(&str, &str, usize)] = &[
("aws-access-key-id", "AKIA", 20),
("aws-access-key-id", "ASIA", 20),
("github-token", "ghp_", 36),
("github-token", "gho_", 36),
("github-token", "ghu_", 36),
("github-token", "ghs_", 36),
("github-token", "ghr_", 36),
("github-token", "github_pat_", 93),
("openai-api-key", "sk-proj-", 88),
("anthropic-api-key", "sk-ant-", 108),
("stripe-secret-key", "sk_live_", 30),
("stripe-restricted-key", "rk_live_", 30),
("fly-token", "fm2_", 20),
("vercel-token", "vercel_", 20),
("slack-token", "xoxb-", 40),
("slack-token", "xoxa-", 40),
("slack-token", "xoxp-", 40),
("slack-token", "xoxr-", 40),
("slack-token", "xoxs-", 40),
("age-secret-key", "AGE-SECRET-KEY-", 60),
];
const SK_SAFE_PREFIXES: &[&str] = &["sk-learn", "sk-image", "sk-lego", "sk-base", "sk-misc"];
pub(super) fn check_known_patterns(text: &str) -> Option<(&str, &'static str)> {
let base = text.as_ptr() as usize;
let mut best: Option<(&str, &'static str)> = None;
for &(name, needle, min_len) in PREFIX_DETECTORS {
keep_leftmost(
&mut best,
find_prefix_token(text, needle, min_len).map(|m| (m, name)),
base,
);
}
if let Some(token) = find_bare_sk_token(text) {
keep_leftmost(&mut best, Some((token, "openai-api-key")), base);
}
let mut from = 0;
while let Some(rel) = text[from..].find("FlyV1 ") {
let pos = from + rel;
let at_boundary = pos == 0 || {
text[..pos]
.chars()
.next_back()
.is_none_or(|c| !c.is_ascii_alphanumeric())
};
if at_boundary {
let payload_start = pos + 6; let payload = extract_token(&text[payload_start..]);
if payload.len() >= 4 {
let candidate = &text[pos..payload_start + payload.len()];
keep_leftmost(&mut best, Some((candidate, "fly-token")), base);
break;
}
}
from = pos + "FlyV1 ".len();
}
keep_leftmost(
&mut best,
find_pem_private_key_block(text).map(|m| (m, "pem-private-key")),
base,
);
keep_leftmost(&mut best, find_jwt(text).map(|m| (m, "jwt")), base);
keep_leftmost(
&mut best,
find_url_userinfo(text).map(|m| (m, "url-userinfo")),
base,
);
best
}
pub(super) fn find_prefix_token<'a>(
text: &'a str,
needle: &str,
min_len: usize,
) -> Option<&'a str> {
let mut start = 0;
while let Some(rel) = text[start..].find(needle) {
let abs = start + rel;
let at_boundary = abs == 0 || {
let prev = text[..abs].chars().next_back().unwrap_or(' ');
!prev.is_ascii_alphanumeric()
};
if at_boundary {
let token = extract_token(&text[abs..]);
if token.len() >= min_len && !is_filename_shaped_prefix_match(token, needle) {
return Some(token);
}
}
start = abs + needle.len().max(1);
}
None
}
fn find_bare_sk_token(text: &str) -> Option<&str> {
let base = text.as_ptr() as usize;
let mut from = 0;
while from < text.len() {
let token = find_prefix_token(&text[from..], "sk-", 30)?;
let belongs_to_specific_detector = PREFIX_DETECTORS
.iter()
.any(|&(_, needle, _)| needle.starts_with("sk-") && token.starts_with(needle));
let is_safe_compound = SK_SAFE_PREFIXES.iter().any(|safe| token.starts_with(safe));
if !belongs_to_specific_detector && !is_safe_compound {
return Some(token);
}
let token_start = token.as_ptr() as usize - base;
from = token_start + "sk-".len();
}
None
}
const SOURCE_FILE_EXTENSIONS: &[&str] =
&[".py", ".rs", ".ts", ".js", ".sh", ".md", ".toml", ".json"];
pub(super) fn is_filename_shaped_prefix_match(token: &str, needle: &str) -> bool {
let token = token.trim_end_matches(|c: char| {
matches!(
c,
'`' | '"' | '\'' | ')' | ']' | '}' | '>' | ',' | ';' | ':' | '!' | '?'
)
});
let token = token.strip_suffix('.').unwrap_or(token);
let Some(payload) = token.strip_prefix(needle) else {
return false;
};
let payload = strip_source_citation_line_reference(payload);
let Some(stem) = SOURCE_FILE_EXTENSIONS
.iter()
.find_map(|extension| payload.strip_suffix(extension))
else {
return false;
};
stem.bytes().any(|byte| byte.is_ascii_lowercase())
&& stem
.bytes()
.any(|byte| matches!(byte, b'_' | b'-' | b'/' | b'.'))
&& stem
.bytes()
.all(|byte| byte.is_ascii_lowercase() || matches!(byte, b'_' | b'-' | b'/' | b'.'))
}
fn strip_source_citation_line_reference(payload: &str) -> &str {
let Some(colon) = payload.rfind(':') else {
return payload;
};
let (head, reference) = (&payload[..colon], &payload[colon + 1..]);
let is_digit_run = |s: &str| !s.is_empty() && s.bytes().all(|b| b.is_ascii_digit());
let is_line_reference = match reference.split_once('-') {
None => is_digit_run(reference),
Some((start, end)) => is_digit_run(start) && is_digit_run(end),
};
if is_line_reference {
head
} else {
payload
}
}
const PEM_BODY_LINE_MIN: usize = 40;
fn is_pem_body_line(line: &str) -> bool {
line.len() >= PEM_BODY_LINE_MIN
&& line
.bytes()
.all(|b| b.is_ascii_alphanumeric() || b == b'+' || b == b'/' || b == b'=')
}
fn next_line_break(text: &str, from: usize) -> Option<(usize, usize)> {
let rest = &text[from..];
let real = rest.find('\n').map(|i| (from + i, from + i + 1));
let escaped = rest.find("\\n").map(|i| (from + i, from + i + 2));
match (real, escaped) {
(Some(r), Some(e)) => Some(if r.0 <= e.0 { r } else { e }),
(r, e) => r.or(e),
}
}
fn line_bounds(text: &str, from: usize) -> (usize, usize) {
match next_line_break(text, from) {
Some((end, next)) => (end, next),
None => (text.len(), text.len()),
}
}
fn find_pem_private_key_block(text: &str) -> Option<&str> {
let mut search = 0;
while let Some(rel) = text[search..].find("-----BEGIN") {
let pos = search + rel;
let (header_end, body_start) = line_bounds(text, pos);
let header = &text[pos..header_end];
search = header_end;
if !header.contains("PRIVATE KEY-----") {
continue;
}
let next_begin = text[body_start..]
.find("-----BEGIN")
.map(|r| body_start + r)
.unwrap_or(text.len());
if let Some(end_rel) = text[body_start..next_begin].find("-----END") {
let end_pos = body_start + end_rel;
let (end_line, end_next) = line_bounds(text, end_pos);
if text[end_pos..end_line].contains("PRIVATE KEY-----") {
return Some(&text[pos..end_next]);
}
}
let mut body_end = None;
let mut cursor = body_start;
while cursor < text.len() {
let (line_end, next) = line_bounds(text, cursor);
let line = text[cursor..line_end]
.trim_end_matches('\r')
.trim_end_matches("\\r");
if is_pem_body_line(line) {
body_end = Some(next);
cursor = next;
continue;
}
let run = line
.bytes()
.take_while(|b| b.is_ascii_alphanumeric() || *b == b'+' || *b == b'/' || *b == b'=')
.count();
let whole_line = run == line.len();
let json_end = line[run..].starts_with('"');
if run > 0 && (whole_line || json_end) {
if body_end.is_some() {
body_end = Some(if whole_line { next } else { cursor + run });
} else if run >= PEM_BODY_LINE_MIN && json_end {
body_end = Some(cursor + run);
}
}
break;
}
if let Some(end) = body_end {
return Some(&text[pos..end]);
}
}
None
}
fn find_jwt(text: &str) -> Option<&str> {
let bytes = text.as_bytes();
let mut i = 0;
while i + 4 < bytes.len() {
if bytes[i..].starts_with(b"eyJ") {
let end = bytes[i..]
.iter()
.position(|&b| b == b' ' || b == b'\n' || b == b'\r' || b == b'\t')
.map(|p| i + p)
.unwrap_or(bytes.len());
let candidate = &text[i..end];
let dots = candidate.as_bytes().iter().filter(|&&b| b == b'.').count();
if dots >= 2 {
let parts: Vec<&str> = candidate.splitn(3, '.').collect();
if parts.len() == 3
&& parts[0].starts_with("eyJ")
&& parts[1].starts_with("eyJ")
&& parts[0].len() >= 10
&& parts[1].len() >= 10
{
return Some(candidate);
}
}
i = end + 1;
} else {
i += 1;
}
}
None
}
pub(super) fn find_url_userinfo(text: &str) -> Option<&str> {
let mut search = text;
let mut base = 0usize;
while let Some(at_rel) = search.find("://") {
let at_abs = base + at_rel;
let rest_start = at_abs + 3;
let rest = &text[rest_start..];
let authority_end = rest
.find(['/', '?', '#', ' ', '\n', '\r'])
.unwrap_or(rest.len());
if let Some(at_pos) = rest[..authority_end].rfind('@') {
let userinfo = &rest[..at_pos];
if let Some(colon) = userinfo.find(':') {
let pass = &userinfo[colon + 1..];
if !pass.is_empty() {
let scheme_start = text[..at_abs]
.char_indices()
.rev()
.find(|(_, c)| {
!c.is_ascii_alphanumeric() && *c != '+' && *c != '-' && *c != '.'
})
.map(|(idx, c)| idx + c.len_utf8())
.unwrap_or(0);
if !userinfo.contains(' ') && !userinfo.contains('\n') {
let end = rest_start
+ at_pos
+ 1
+ rest[at_pos + 1..]
.find([' ', '\n', '\r'])
.unwrap_or(rest[at_pos + 1..].len());
return Some(&text[scheme_start..end.min(text.len())]);
}
}
}
}
base = at_abs + 3;
search = &text[base..];
}
None
}