use std::fs;
use std::io::{self, Read, Write};
use std::path::{Path, PathBuf};
use std::process::ExitCode;
use std::sync::atomic::{AtomicBool, Ordering};
use anyhow::{Context, Result};
use clap::Args;
use serde_json::json;
use veloci::FormatHint;
use crate::util::{
CONFIG_ENV, ConfigArg, DetectorArg, Fatal, RulesCache, SKIPPED_DIRS, WalkOptions,
for_each_ordered, is_broken_pipe, is_file, walk_builder,
};
#[derive(Debug, Args)]
pub struct ScanArgs {
#[arg(value_name = "PATH")]
paths: Vec<PathBuf>,
#[arg(long)]
unprotected: bool,
#[arg(short = 'l', long)]
files_with_matches: bool,
#[arg(long, conflicts_with = "files_with_matches")]
json: bool,
#[arg(long, conflicts_with = "files_with_matches")]
show_value: bool,
#[arg(short = 'g', long, value_name = "GLOB")]
glob: Vec<String>,
#[arg(long)]
skip_hidden: bool,
#[arg(long)]
skip_ignored: bool,
#[arg(long)]
all_dirs: bool,
#[arg(short = 'L', long)]
follow: bool,
#[arg(short = 'd', long, value_name = "NUM")]
max_depth: Option<usize>,
#[arg(long, value_name = "SIZE", default_value = "10M", value_parser = parse_size)]
max_filesize: u64,
#[arg(long, value_name = "FILE", env = CONFIG_ENV)]
config: Option<PathBuf>,
#[command(flatten)]
enable: DetectorArg,
}
pub struct Scan<'a> {
pub walk: WalkOptions<'a>,
pub max_filesize: u64,
pub max_files: Option<usize>,
pub unprotected: bool,
pub show_value: bool,
pub relative_to: Option<&'a Path>,
}
impl Scan<'_> {
pub fn project(root: &Path) -> Scan<'_> {
Scan {
walk: WalkOptions {
globs: &[],
hidden: true,
ignored: true,
follow: false,
max_depth: Some(8),
skipped_dirs: SKIPPED_DIRS,
},
max_filesize: 1 << 20,
max_files: Some(50_000),
unprotected: true,
show_value: false,
relative_to: Some(root),
}
}
}
pub struct Hit {
pub path: PathBuf,
pub findings: Vec<Finding>,
pub protected: bool,
}
pub struct Finding {
pub detector: String,
pub line: Option<usize>,
pub value: Option<String>,
}
impl Hit {
pub fn detectors(&self) -> Vec<&str> {
let mut detectors: Vec<&str> = Vec::new();
for finding in &self.findings {
if !detectors.contains(&finding.detector.as_str()) {
detectors.push(&finding.detector);
}
}
detectors
}
}
#[derive(Default)]
pub struct Report {
pub hits: Vec<Hit>,
pub scanned: usize,
pub skipped: usize,
pub protected: usize,
pub errors: Vec<String>,
}
enum Outcome {
Clean,
Skipped,
Protected,
Hit(Hit),
Failed(String),
Fatal(String),
}
pub fn scan(roots: &[PathBuf], options: &Scan<'_>, config: ConfigArg) -> Result<Report> {
let rules = RulesCache::new(config);
let mut walks = Vec::new();
for root in roots {
walks.push(walk_builder(root, &options.walk)?.build());
}
let entries = walks
.into_iter()
.flatten()
.filter(|entry| entry.as_ref().map_or(true, is_file))
.take(options.max_files.unwrap_or(usize::MAX));
let mut report = Report::default();
let mut fatal = None;
let stop = AtomicBool::new(false);
let unreported = for_each_ordered(
entries,
&stop,
|| (),
|(), entry| {
let outcome = match entry {
Ok(entry) => scan_file(&rules, options, entry.path()),
Err(err) => Outcome::Failed(err.to_string()),
};
if matches!(outcome, Outcome::Fatal(_)) {
stop.store(true, Ordering::Relaxed);
}
outcome
},
|outcome| {
match outcome {
Outcome::Clean => report.scanned += 1,
Outcome::Skipped => report.skipped += 1,
Outcome::Protected => report.protected += 1,
Outcome::Hit(hit) => {
report.scanned += 1;
report.hits.push(hit);
}
Outcome::Failed(err) => report.errors.push(err),
Outcome::Fatal(err) => {
fatal = Some(err);
return Ok(false);
}
}
Ok(true)
},
)?;
let fatal = fatal.or_else(|| {
unreported.into_iter().find_map(|outcome| match outcome {
Outcome::Fatal(err) => Some(err),
_ => None,
})
});
if let Some(err) = fatal {
return Err(Fatal(err).into());
}
for hit in &mut report.hits {
if let Some(shown) = options
.relative_to
.and_then(|base| hit.path.strip_prefix(base).ok())
{
hit.path = shown.to_owned();
}
}
Ok(report)
}
fn scan_file(rules: &RulesCache, options: &Scan<'_>, path: &Path) -> Outcome {
let result = (|| -> Result<Outcome> {
let rules = match rules.for_file(path) {
Ok(rules) => rules,
Err(err) if err.is::<Fatal>() => return Ok(Outcome::Fatal(format!("{err:#}"))),
Err(err) => return Err(err),
};
let protected = rules
.agent
.as_ref()
.is_some_and(|agent| agent.is_protected(path));
if protected && options.unprotected {
return Ok(Outcome::Protected);
}
if rules.allowed_files.matches(path) {
return Ok(Outcome::Clean);
}
let mut file =
fs::File::open(path).with_context(|| format!("reading {}", path.display()))?;
let size = file
.metadata()
.with_context(|| format!("reading {}", path.display()))?
.len();
if size > options.max_filesize {
return Ok(Outcome::Skipped);
}
let mut data = Vec::with_capacity(size as usize);
file.read_to_end(&mut data)
.with_context(|| format!("reading {}", path.display()))?;
if data[..data.len().min(8192)].contains(&0) {
return Ok(Outcome::Skipped);
}
let redaction = rules
.redactor_for(Some(path))
.redact(&data, FormatHint::Path(path))
.with_context(|| format!("redacting {}", path.display()))?;
let findings: Vec<_> = redaction
.findings()
.iter()
.filter(|finding| !rules.allow.allows(finding))
.map(|finding| Finding {
detector: finding.detector.clone(),
line: finding.offset().map(|o| redaction.line_col(o).0),
value: options.show_value.then(|| finding.secret.clone()),
})
.collect();
if findings.is_empty() {
return Ok(Outcome::Clean);
}
Ok(Outcome::Hit(Hit {
path: path.to_owned(),
findings,
protected,
}))
})();
result.unwrap_or_else(|err| Outcome::Failed(format!("{err:#}")))
}
pub fn run(args: ScanArgs) -> Result<ExitCode> {
let implicit_root = args.paths.is_empty();
let paths = if implicit_root {
vec![PathBuf::from(".")]
} else {
args.paths.clone()
};
let options = Scan {
walk: WalkOptions {
globs: &args.glob,
hidden: !args.skip_hidden,
ignored: !args.skip_ignored,
follow: args.follow,
max_depth: args.max_depth,
skipped_dirs: if args.all_dirs { &[] } else { SKIPPED_DIRS },
},
max_filesize: args.max_filesize,
max_files: None,
unprotected: args.unprotected,
show_value: args.show_value,
relative_to: implicit_root.then_some(Path::new(".")),
};
let report = scan(
&paths,
&options,
ConfigArg {
config: args.config.clone(),
enable: args.enable.clone(),
},
)?;
for error in &report.errors {
eprintln!("error: {error}");
}
match print(&args, &report) {
Err(err) if is_broken_pipe(&err) => {}
result => result?,
}
Ok(if !report.hits.is_empty() {
ExitCode::from(1)
} else if !report.errors.is_empty() {
ExitCode::from(2)
} else {
ExitCode::SUCCESS
})
}
fn print(args: &ScanArgs, report: &Report) -> Result<()> {
let mut out = io::stdout().lock();
if args.json {
let files: Vec<_> = report
.hits
.iter()
.map(|hit| {
let findings: Vec<_> = hit
.findings
.iter()
.map(|finding| {
let mut entry = json!({
"detector": finding.detector,
"line": finding.line,
});
if let Some(value) = &finding.value {
entry["value"] = json!(value);
}
entry
})
.collect();
json!({
"path": hit.path.display().to_string(),
"count": hit.findings.len(),
"detectors": hit.detectors(),
"protected": hit.protected,
"findings": findings,
})
})
.collect();
let document = json!({
"files": files,
"scanned": report.scanned,
"skipped": report.skipped,
"protected_unread": report.protected,
"errors": report.errors,
});
serde_json::to_writer_pretty(&mut out, &document)?;
writeln!(out)?;
return Ok(());
}
if args.files_with_matches {
for hit in &report.hits {
writeln!(out, "{}", hit.path.display())?;
}
return Ok(());
}
if !report.hits.is_empty() {
let mut header = vec!["FILE", "FINDINGS", "DETECTORS"];
if args.show_value {
header.push("VALUE");
}
header.push("");
let header: Vec<String> = header.into_iter().map(String::from).collect();
let rows: Vec<Vec<String>> = report
.hits
.iter()
.map(|hit| {
let mut row = vec![
hit.path.display().to_string(),
hit.findings.len().to_string(),
hit.detectors().join(","),
];
if args.show_value {
row.push(value_cell(&hit.findings));
}
row.push(if hit.protected { "protected" } else { "" }.into());
row
})
.collect();
write_table(&mut out, &header, &rows)?;
}
out.flush()?;
let with = if args.unprotected {
"unprotected with secrets"
} else {
"with secrets"
};
let unread = if args.unprotected {
format!(", {} protected and not read", report.protected)
} else {
String::new()
};
eprintln!(
"scanned {} files: {} {with}, {} skipped as binary or over --max-filesize{unread}",
report.scanned,
report.hits.len(),
report.skipped,
);
Ok(())
}
const TABLE_VALUES: usize = 3;
const TABLE_VALUE_CHARS: usize = 60;
fn value_cell(findings: &[Finding]) -> String {
let values: Vec<&str> = findings
.iter()
.filter_map(|finding| finding.value.as_deref())
.collect();
let mut shown: Vec<String> = values
.iter()
.take(TABLE_VALUES)
.map(|value| {
if value.chars().count() > TABLE_VALUE_CHARS {
let head: String = value.chars().take(TABLE_VALUE_CHARS).collect();
format!("{head:?}…")
} else {
format!("{value:?}")
}
})
.collect();
if values.len() > TABLE_VALUES {
shown.push(format!("+{} more", values.len() - TABLE_VALUES));
}
shown.join(", ")
}
fn write_table(out: &mut impl Write, header: &[String], rows: &[Vec<String>]) -> Result<()> {
let mut widths = vec![0; header.len()];
for row in std::iter::once(header).chain(rows.iter().map(Vec::as_slice)) {
for (width, cell) in widths.iter_mut().zip(row) {
*width = (*width).max(cell.chars().count());
}
}
for row in std::iter::once(header).chain(rows.iter().map(Vec::as_slice)) {
let line: Vec<String> = row
.iter()
.zip(&widths)
.map(|(cell, w)| {
let pad = w.saturating_sub(cell.chars().count());
format!("{cell}{}", " ".repeat(pad))
})
.collect();
writeln!(out, "{}", line.join(" ").trim_end())?;
}
Ok(())
}
fn parse_size(text: &str) -> Result<u64, String> {
let text = text.trim();
let (number, shift) = match text.chars().last().map(|c| c.to_ascii_uppercase()) {
Some('K') => (&text[..text.len() - 1], 10),
Some('M') => (&text[..text.len() - 1], 20),
Some('G') => (&text[..text.len() - 1], 30),
_ => (text, 0),
};
number
.parse::<u64>()
.ok()
.and_then(|n| n.checked_mul(1 << shift))
.ok_or_else(|| format!("{text:?} is not a size such as 4096, 500K, 10M or 1G"))
}
#[cfg(test)]
mod tests {
use super::{Finding, parse_size, value_cell, write_table};
#[test]
fn parses_sizes() {
assert_eq!(parse_size("4096"), Ok(4096));
assert_eq!(parse_size("500K"), Ok(500 << 10));
assert_eq!(parse_size("10m"), Ok(10 << 20));
assert_eq!(parse_size("1G"), Ok(1 << 30));
assert!(parse_size("ten").is_err());
assert!(parse_size("").is_err());
}
#[test]
fn writes_cells_wider_than_u16() {
let wide = "x".repeat(70_000);
let header = vec!["A".to_string(), "B".to_string()];
let rows = vec![vec![wide.clone(), "y".to_string()]];
let mut out = Vec::new();
write_table(&mut out, &header, &rows).unwrap();
let text = String::from_utf8(out).unwrap();
assert!(text.contains(&format!("{wide} y")));
}
#[test]
fn cuts_values_short_in_the_table() {
let finding = |value: String| Finding {
detector: "test".into(),
line: None,
value: Some(value),
};
let long = "a".repeat(100);
let findings: Vec<Finding> = [long, "b".into(), "c".into(), "d".into(), "e".into()]
.into_iter()
.map(finding)
.collect();
let cell = value_cell(&findings);
assert_eq!(
cell,
format!("{:?}…, \"b\", \"c\", +2 more", "a".repeat(60))
);
}
}