use anyhow::{Context, bail};
use clap::{Args, Subcommand, ValueEnum, ValueHint};
use path_dedot::ParseDot;
use serde::{Deserialize, Serialize};
use std::{
fs,
net::IpAddr,
path::{Path, PathBuf},
str::FromStr,
};
use strum::Display;
use tracing::debug;
use url::Url;
use crate::{
cli::{
commands::{
azure::AzureRepoSpecifiers,
bitbucket::BitbucketRepoSpecifiers,
gitea::GiteaRepoSpecifiers,
github::GitHubRepoSpecifiers,
gitlab::GitLabRepoSpecifiers,
huggingface::HuggingFaceRepoSpecifiers,
inputs::{ContentFilteringArgs, InputSpecifierArgs},
output::{OutputArgs, ReportOutputFormat},
rules::{RuleCacheArgs, RuleSpecifierArgs},
view,
},
global::RAM_GB,
},
git_url::GitUrl,
rules::rule::Confidence,
util::expand_tilde,
};
pub const UNLIMITED_RESULTS: usize = usize::MAX;
fn default_scan_jobs() -> usize {
let cpu_count = std::thread::available_parallelism().map(usize::from).unwrap_or(1);
let desired = cpu_count;
match *RAM_GB {
Some(ram_gb) => {
let max_by_ram = ram_gb.ceil() as usize; let jobs = desired.min(max_by_ram).max(1);
debug!(
"Using {jobs} parallel scan jobs \
(cpus = {cpu_count}, desired = {desired}, \
ram = {ram_gb:.1} GiB, cap_by_ram = {max_by_ram})"
);
jobs
}
None => {
debug!("Using {desired} parallel scan jobs (cpus = {cpu_count}, ram unknown)");
desired
}
}
}
#[derive(Args, Debug, Clone)]
pub struct ScanArgs {
#[arg(global = true, long = "jobs", short = 'j', default_value_t = default_scan_jobs())]
pub num_jobs: usize,
#[command(flatten)]
pub rules: RuleSpecifierArgs,
#[command(flatten)]
pub rule_cache: RuleCacheArgs,
#[command(flatten)]
pub input_specifier_args: InputSpecifierArgs,
#[command(flatten)]
pub content_filtering_args: ContentFilteringArgs,
#[arg(global = true, long, short = 'c', default_value = "medium")]
pub confidence: ConfidenceLevel,
#[arg(global = true, long, default_value_t = false)]
pub disk_offload: bool,
#[arg(global = true, long, short = 'n', default_value_t = false)]
pub no_validate: bool,
#[arg(
global = true,
long = "validation-timeout",
default_value_t = 10,
value_name = "SECONDS",
value_parser = clap::value_parser!(u64).range(1..=60)
)]
pub validation_timeout: u64,
#[arg(
global = true,
long = "validation-retries",
default_value_t = 1,
value_name = "N",
value_parser = clap::value_parser!(u32).range(0..=5)
)]
pub validation_retries: u32,
#[arg(global = true, long = "validation-rps", value_name = "RPS")]
pub validation_rps: Option<f64>,
#[arg(global = true, long = "validation-rps-rule", value_name = "RULE_SELECTOR=RPS")]
pub validation_rps_rule: Vec<String>,
#[arg(global = true, long, default_value_t = false)]
pub full_validation_response: bool,
#[arg(
global = true,
long = "max-validation-response-length",
default_value_t = 2048,
value_name = "BYTES"
)]
pub max_validation_response_length: usize,
#[arg(global = true, long = "blast-radius", alias = "access-map", default_value_t = false)]
pub access_map: bool,
#[arg(global = true, long, default_value_t = false, conflicts_with = "validation_filter")]
pub only_valid: bool,
#[arg(global = true, long, value_enum, conflicts_with = "only_valid")]
pub validation_filter: Option<ValidationFilter>,
#[arg(global = true, long, default_value_t = false)]
pub include_hidden_findings: bool,
#[arg(global = true, long, short = 'e')]
pub min_entropy: Option<f32>,
#[arg(global = true, long, default_value_t = false)]
pub rule_stats: bool,
#[arg(global = true, long, default_value_t = false)]
pub no_dedup: bool,
#[arg(skip)]
pub view_report: bool,
#[arg(global = true, long, short = 'r', default_value_t = false)]
pub redact: bool,
#[arg(global = true, long, default_value_t = false)]
pub no_base64: bool,
#[arg(global = true, long = "turbo", default_value_t = false)]
pub turbo: bool,
#[arg(global = true, long, default_value_t = 1800, value_name = "SECONDS")]
pub git_repo_timeout: u64,
#[arg(global = true, long = "audit-log", value_name = "FILE", value_hint = ValueHint::FilePath)]
pub audit_log: Option<PathBuf>,
#[command(flatten)]
pub output_args: OutputArgs<ReportOutputFormat>,
#[arg(global = true, long, value_name = "FILE")]
pub baseline_file: Option<std::path::PathBuf>,
#[arg(global = true, long, default_value_t = false)]
pub manage_baseline: bool,
#[arg(global = true, long = "skip-regex", value_name = "PATTERN")]
pub skip_regex: Vec<String>,
#[arg(global = true, long = "skip-word", value_name = "WORD")]
pub skip_word: Vec<String>,
#[arg(
global = true,
long = "skip-aws-account",
value_name = "ACCOUNT_ID",
value_delimiter = ','
)]
pub skip_aws_account: Vec<String>,
#[arg(global = true, long = "skip-aws-account-file", value_name = "FILE")]
pub skip_aws_account_file: Option<PathBuf>,
#[arg(global = true, long = "ignore-comment", value_name = "DIRECTIVE")]
pub extra_ignore_comments: Vec<String>,
#[arg(global = true, long = "no-ignore", default_value_t = false)]
pub no_inline_ignore: bool,
#[arg(global = true, long = "no-ignore-if-contains", default_value_t = false)]
pub no_ignore_if_contains: bool,
#[arg(skip)]
pub view_report_port: u16,
#[arg(skip)]
pub view_report_address: String,
#[arg(global = true, long = "alert-webhook", value_name = "URL")]
pub alert_webhook: Vec<String>,
#[arg(global = true, long = "alert-format", value_name = "FORMAT")]
pub alert_format: Option<crate::alerts::AlertFormat>,
#[arg(global = true, long = "alert-on", value_name = "MODE", default_value = "findings")]
pub alert_on: crate::alerts::AlertOn,
#[arg(
global = true,
long = "alert-min-confidence",
value_name = "LEVEL",
default_value = "medium"
)]
pub alert_min_confidence: ConfidenceLevel,
#[arg(global = true, long = "alert-include-secret", default_value_t = false)]
pub alert_include_secret: bool,
#[arg(
global = true,
long = "alert-report-url",
value_name = "URL",
env = "KINGFISHER_ALERT_REPORT_URL"
)]
pub alert_report_url: Option<String>,
#[arg(global = true, long = "alert-detail", value_name = "MODE", default_value = "auto")]
pub alert_detail: crate::alerts::AlertDetail,
#[arg(
global = true,
long = "alert-finding-filter",
value_name = "FILTER",
default_value = "all"
)]
pub alert_finding_filter: crate::alerts::AlertFindingFilter,
#[arg(global = true, long = "alert-prevent-empty", default_value_t = false)]
pub alert_prevent_empty: bool,
#[arg(global = true, long = "alert-dry-run", default_value_t = false)]
pub alert_dry_run: bool,
#[arg(skip)]
pub config_webhook_overrides: Vec<ConfigWebhookOverride>,
}
#[derive(Debug, Clone, Default)]
pub struct ConfigWebhookOverride {
pub format: Option<crate::alerts::AlertFormat>,
pub on: Option<crate::alerts::AlertOn>,
pub min_confidence: Option<ConfidenceLevel>,
pub include_secret: Option<bool>,
pub report_url: Option<String>,
pub detail: Option<crate::alerts::AlertDetail>,
pub finding_filter: Option<crate::alerts::AlertFindingFilter>,
pub prevent_empty: Option<bool>,
}
#[derive(Copy, Clone, Debug, Display, PartialEq, Eq, PartialOrd, Ord, ValueEnum)]
#[strum(serialize_all = "kebab-case")]
pub enum ConfidenceLevel {
Low,
Medium,
High,
}
#[derive(
Copy, Clone, Debug, Display, PartialEq, Eq, PartialOrd, Ord, ValueEnum, Serialize, Deserialize,
)]
#[strum(serialize_all = "kebab-case")]
#[serde(rename_all = "kebab-case")]
pub enum ValidationFilter {
All,
Active,
Actionable,
}
impl ScanArgs {
pub fn validate_audit_log_collisions(
&self,
endpoint_config: Option<&Path>,
project_config: Option<&Path>,
) -> anyhow::Result<()> {
let Some(audit_log) = &self.audit_log else { return Ok(()) };
let audit_log = &expand_tilde(audit_log);
if let Some(output) = &self.output_args.output
&& paths_refer_to_same_file(audit_log, output)
{
bail!("--audit-log and --output must use different paths");
}
if self.baseline_file.is_some() || self.manage_baseline {
let baseline = self
.baseline_file
.clone()
.unwrap_or_else(|| std::path::PathBuf::from("baseline-file.yaml"));
if paths_refer_to_same_file(audit_log, &baseline) {
bail!("--audit-log and the baseline file must use different paths");
}
}
for rules_path in &self.rules.rules_path {
let rules_path = expand_tilde(rules_path);
if paths_refer_to_same_file(audit_log, &rules_path)
|| path_is_beneath(audit_log, &rules_path)
{
bail!("--audit-log must not overwrite a custom rules file");
}
}
for input in &self.input_specifier_args.path_inputs {
if input == Path::new("-") {
continue;
}
if paths_refer_to_same_file(audit_log, input) || path_is_beneath(audit_log, input) {
bail!(
"--audit-log must not alias or reside inside the scanned input {}",
input.display()
);
}
}
let file_inputs = self
.input_specifier_args
.gcs_service_account
.iter()
.chain(self.input_specifier_args.docker_archive.iter())
.chain(self.skip_aws_account_file.iter());
for input in file_inputs {
if paths_refer_to_same_file(audit_log, input) {
bail!("--audit-log must not overwrite input file {}", input.display());
}
}
if let Some(endpoint_config) = endpoint_config
&& paths_refer_to_same_file(audit_log, endpoint_config)
{
bail!("--audit-log must not overwrite the endpoint configuration file");
}
if let Some(project_config) = project_config
&& paths_refer_to_same_file(audit_log, project_config)
{
bail!("--audit-log must not overwrite the project configuration file");
}
Ok(())
}
pub fn effective_validation_filter(&self) -> ValidationFilter {
if self.only_valid {
ValidationFilter::Active
} else {
self.validation_filter.unwrap_or(ValidationFilter::All)
}
}
}
impl From<ConfidenceLevel> for Confidence {
fn from(level: ConfidenceLevel) -> Self {
match level {
ConfidenceLevel::Low => Confidence::Low,
ConfidenceLevel::Medium => Confidence::Medium,
ConfidenceLevel::High => Confidence::High,
}
}
}
#[derive(Args, Debug, Clone)]
pub struct ScanCommandArgs {
#[command(flatten)]
pub scan_args: ScanArgs,
#[arg(global = true, long = "view-report", default_value_t = false)]
pub view_report: bool,
#[arg(
global = true,
long = "view-report-port",
default_value_t = view::DEFAULT_PORT,
value_name = "PORT"
)]
pub view_report_port: u16,
#[arg(
global = true,
long = "view-report-address",
default_value = view::DEFAULT_ADDRESS,
value_name = "ADDRESS"
)]
pub view_report_address: String,
#[command(subcommand)]
pub provider: Option<ScanInputCommand>,
}
#[allow(clippy::large_enum_variant)]
#[derive(Debug)]
pub enum ScanOperation {
Scan(ScanArgs),
ListRepositories(ListRepositoriesCommand),
}
#[derive(Debug)]
pub enum ListRepositoriesCommand {
Github { api_url: Url, specifiers: GitHubRepoSpecifiers },
Gitlab { api_url: Url, specifiers: GitLabRepoSpecifiers },
Gitea { api_url: Url, specifiers: GiteaRepoSpecifiers },
Bitbucket { api_url: Url, specifiers: BitbucketRepoSpecifiers },
Azure { base_url: Url, specifiers: AzureRepoSpecifiers },
Huggingface { specifiers: HuggingFaceRepoSpecifiers },
}
fn load_github_event_users(
cli_users: Vec<String>,
user_file: Option<&Path>,
) -> anyhow::Result<Vec<String>> {
fn is_valid_github_username(user: &str) -> bool {
user.len() <= 39
&& !user.starts_with('-')
&& !user.ends_with('-')
&& user.chars().all(|c| c.is_ascii_alphanumeric() || c == '-')
}
fn push_user(users: &mut Vec<String>, raw: &str, source: &str) -> anyhow::Result<()> {
let user = raw.trim().trim_start_matches('@');
if user.is_empty() {
return Ok(());
}
if !is_valid_github_username(user) {
bail!("Invalid GitHub username in {source}: {user:?}");
}
if !users.iter().any(|existing| existing.eq_ignore_ascii_case(user)) {
users.push(user.to_string());
}
Ok(())
}
let mut users = Vec::new();
for user in &cli_users {
push_user(&mut users, user, "--user")?;
}
if let Some(path) = user_file {
let contents = fs::read_to_string(path).with_context(|| {
format!("Failed to read GitHub public event user file {}", path.display())
})?;
for (line_number, line) in contents.lines().enumerate() {
let trimmed = line.trim();
if trimmed.is_empty() || trimmed.starts_with('#') {
continue;
}
push_user(&mut users, trimmed, &format!("{}:{}", path.display(), line_number + 1))?;
}
}
Ok(users)
}
impl ScanCommandArgs {
fn infer_positional_git_urls(&mut self) {
let mut inferred_git_urls = Vec::new();
let mut retained_paths = Vec::new();
for path in self.scan_args.input_specifier_args.path_inputs.drain(..) {
if path.as_path() == Path::new("-") || path.exists() {
retained_paths.push(path);
continue;
}
if let Some(git_url) = parse_git_url_target(&path) {
inferred_git_urls.push(git_url);
} else {
retained_paths.push(path);
}
}
self.scan_args.input_specifier_args.path_inputs = retained_paths;
self.scan_args.input_specifier_args.git_url.extend(inferred_git_urls);
}
pub fn into_operation(mut self) -> anyhow::Result<ScanOperation> {
let mut used_provider_subcommand = false;
self.scan_args.view_report = self.view_report;
self.scan_args.view_report_port = self.view_report_port;
self.scan_args.view_report_address = self.view_report_address.clone();
if let Some(provider) = self.provider.take() {
used_provider_subcommand = true;
let scan_args = &mut self.scan_args;
let maybe_list = match provider {
ScanInputCommand::Filesystem(args) => {
if args.paths.is_empty() {
bail!("Provide at least one path when using the filesystem subcommand");
}
scan_args.input_specifier_args.path_inputs = args.paths;
scan_args.input_specifier_args.git_url = args.git_url;
None
}
ScanInputCommand::Github(args) => {
let mut specifiers = args.specifiers;
if args.event_user_file.is_some() && !args.public_events {
bail!("--user-file can only be used with --public-events");
}
if args.public_events {
if let (Some(audit_log), Some(user_file)) =
(scan_args.audit_log.as_deref(), args.event_user_file.as_deref())
&& paths_refer_to_same_file(
&expand_tilde(audit_log),
&expand_tilde(user_file),
)
{
bail!(
"--audit-log must not overwrite the GitHub public-event user file"
);
}
let event_users = load_github_event_users(
std::mem::take(&mut specifiers.user),
args.event_user_file.as_deref(),
)?;
if event_users.is_empty() {
bail!(
"You must specify at least one --user or --user-file when scanning GitHub public events"
);
}
if !specifiers.organization.is_empty() || specifiers.all_organizations {
bail!(
"GitHub public event scanning supports --user and --user-file only"
);
}
if args.include_contributors {
bail!("--include-contributors cannot be used with --public-events");
}
if args.list_only {
bail!("--list-only cannot be used with --public-events");
}
if specifiers.repo_type
!= crate::cli::commands::github::GitHubRepoType::Source
{
bail!("--repo-type cannot be used with --public-events");
}
scan_args.input_specifier_args.github_event_user = event_users;
scan_args.input_specifier_args.github_event_lookback_hours =
args.event_lookback_hours;
scan_args.input_specifier_args.github_exclude = specifiers.exclude_repos;
scan_args.input_specifier_args.github_api_url = args.api_url;
scan_args.input_specifier_args.repo_clone_limit = args.repo_clone_limit;
None
} else if specifiers.is_empty() {
bail!(
"You must specify at least one --user, --org, or use --all-orgs when scanning GitHub"
);
} else if args.list_only {
Some(ListRepositoriesCommand::Github { api_url: args.api_url, specifiers })
} else {
scan_args.input_specifier_args.github_include_gists =
specifiers.include_gists;
scan_args.input_specifier_args.github_user = specifiers.user;
scan_args.input_specifier_args.github_organization =
specifiers.organization;
scan_args.input_specifier_args.github_exclude = specifiers.exclude_repos;
scan_args.input_specifier_args.all_github_organizations =
specifiers.all_organizations;
scan_args.input_specifier_args.github_repo_type = specifiers.repo_type;
scan_args.input_specifier_args.github_api_url = args.api_url;
scan_args.input_specifier_args.repo_clone_limit = args.repo_clone_limit;
scan_args.input_specifier_args.include_contributors =
args.include_contributors;
None
}
}
ScanInputCommand::Gitlab(args) => {
if args.specifiers.is_empty() {
bail!(
"You must specify at least one --user, --group, or use --all-groups when scanning GitLab"
);
}
if args.list_only {
Some(ListRepositoriesCommand::Gitlab {
api_url: args.api_url,
specifiers: args.specifiers,
})
} else {
scan_args.input_specifier_args.gitlab_include_snippets =
args.specifiers.include_snippets;
scan_args.input_specifier_args.gitlab_user = args.specifiers.user;
scan_args.input_specifier_args.gitlab_group = args.specifiers.group;
scan_args.input_specifier_args.gitlab_exclude =
args.specifiers.exclude_repos;
scan_args.input_specifier_args.all_gitlab_groups =
args.specifiers.all_groups;
scan_args.input_specifier_args.gitlab_include_subgroups =
args.specifiers.include_subgroups;
scan_args.input_specifier_args.gitlab_repo_type = args.specifiers.repo_type;
scan_args.input_specifier_args.gitlab_api_url = args.api_url;
scan_args.input_specifier_args.repo_clone_limit = args.repo_clone_limit;
scan_args.input_specifier_args.include_contributors =
args.include_contributors;
None
}
}
ScanInputCommand::Gitea(args) => {
if args.specifiers.is_empty() {
bail!(
"Specify at least one --user, --org, or use --all-orgs when scanning Gitea"
);
}
if args.list_only {
Some(ListRepositoriesCommand::Gitea {
api_url: args.api_url,
specifiers: args.specifiers,
})
} else {
scan_args.input_specifier_args.gitea_user = args.specifiers.user;
scan_args.input_specifier_args.gitea_organization =
args.specifiers.organization;
scan_args.input_specifier_args.gitea_exclude =
args.specifiers.exclude_repos;
scan_args.input_specifier_args.all_gitea_organizations =
args.specifiers.all_organizations;
scan_args.input_specifier_args.gitea_repo_type = args.specifiers.repo_type;
scan_args.input_specifier_args.gitea_api_url = args.api_url;
None
}
}
ScanInputCommand::Bitbucket(args) => {
if args.specifiers.is_empty() {
bail!(
"You must specify at least one --user, --workspace, --project, or use --all-workspaces when scanning Bitbucket"
);
}
if args.list_only {
Some(ListRepositoriesCommand::Bitbucket {
api_url: args.api_url,
specifiers: args.specifiers,
})
} else {
scan_args.input_specifier_args.bitbucket_include_snippets =
args.specifiers.include_snippets;
scan_args.input_specifier_args.bitbucket_user = args.specifiers.user;
scan_args.input_specifier_args.bitbucket_workspace =
args.specifiers.workspace;
scan_args.input_specifier_args.bitbucket_project = args.specifiers.project;
scan_args.input_specifier_args.bitbucket_exclude =
args.specifiers.exclude_repos;
scan_args.input_specifier_args.all_bitbucket_workspaces =
args.specifiers.all_workspaces;
scan_args.input_specifier_args.bitbucket_repo_type =
args.specifiers.repo_type;
scan_args.input_specifier_args.bitbucket_api_url = args.api_url;
None
}
}
ScanInputCommand::Azure(args) => {
if args.specifiers.is_empty() {
bail!(
"You must specify at least one --organization, --project, or use --all-projects when scanning Azure DevOps"
);
}
if args.list_only {
Some(ListRepositoriesCommand::Azure {
base_url: args.base_url,
specifiers: args.specifiers,
})
} else {
scan_args.input_specifier_args.azure_organization =
args.specifiers.organization;
scan_args.input_specifier_args.azure_project = args.specifiers.project;
scan_args.input_specifier_args.azure_exclude =
args.specifiers.exclude_repos;
scan_args.input_specifier_args.all_azure_projects =
args.specifiers.all_projects;
scan_args.input_specifier_args.azure_repo_type = args.specifiers.repo_type;
scan_args.input_specifier_args.azure_base_url = args.base_url;
None
}
}
ScanInputCommand::Huggingface(args) => {
if args.specifiers.is_empty() {
bail!(
"You must specify at least one --user, --org, --model, --dataset, --space, or --bucket when scanning Hugging Face"
);
}
if args.list_only {
Some(ListRepositoriesCommand::Huggingface { specifiers: args.specifiers })
} else {
scan_args.input_specifier_args.huggingface_user = args.specifiers.user;
scan_args.input_specifier_args.huggingface_organization =
args.specifiers.organization;
scan_args.input_specifier_args.huggingface_model = args.specifiers.model;
scan_args.input_specifier_args.huggingface_dataset =
args.specifiers.dataset;
scan_args.input_specifier_args.huggingface_space = args.specifiers.space;
scan_args.input_specifier_args.huggingface_bucket = args.specifiers.bucket;
scan_args.input_specifier_args.huggingface_exclude =
args.specifiers.exclude;
None
}
}
ScanInputCommand::Slack(args) => {
scan_args.input_specifier_args.slack_query = Some(args.query);
scan_args.input_specifier_args.slack_api_url = args.api_url;
scan_args.input_specifier_args.max_results = args.max_results;
None
}
ScanInputCommand::Teams(args) => {
scan_args.input_specifier_args.teams_query = Some(args.query);
scan_args.input_specifier_args.teams_api_url = args.api_url;
scan_args.input_specifier_args.max_results = args.max_results;
None
}
ScanInputCommand::Jira(args) => {
scan_args.input_specifier_args.jira_url = Some(args.url);
scan_args.input_specifier_args.jql = Some(args.jql);
scan_args.input_specifier_args.max_results =
if args.all { UNLIMITED_RESULTS } else { args.max_results };
scan_args.input_specifier_args.jira_include_comments = args.include_comments;
scan_args.input_specifier_args.jira_include_changelog = args.include_changelog;
None
}
ScanInputCommand::Confluence(args) => {
scan_args.input_specifier_args.confluence_url = Some(args.url);
scan_args.input_specifier_args.cql = Some(args.cql);
scan_args.input_specifier_args.max_results =
if args.all { UNLIMITED_RESULTS } else { args.max_results };
None
}
ScanInputCommand::Postman(args) => {
if !args.all
&& args.workspaces.is_empty()
&& args.collections.is_empty()
&& args.environments.is_empty()
{
bail!(
"Specify --workspace, --collection, --environment, or --all when using the postman subcommand"
);
}
scan_args.input_specifier_args.postman_workspaces = args.workspaces;
scan_args.input_specifier_args.postman_collections = args.collections;
scan_args.input_specifier_args.postman_environments = args.environments;
scan_args.input_specifier_args.postman_all = args.all;
scan_args.input_specifier_args.postman_include_mocks_monitors =
args.include_mocks_monitors;
scan_args.input_specifier_args.postman_api_url = args.api_url;
scan_args.input_specifier_args.max_results = args.max_results;
None
}
ScanInputCommand::S3(args) => {
scan_args.input_specifier_args.s3_bucket = Some(args.bucket);
scan_args.input_specifier_args.s3_prefix = args.prefix;
scan_args.input_specifier_args.role_arn = args.role_arn;
scan_args.input_specifier_args.aws_local_profile = args.profile;
None
}
ScanInputCommand::Gcs(args) => {
scan_args.input_specifier_args.gcs_bucket = Some(args.bucket);
scan_args.input_specifier_args.gcs_prefix = args.prefix;
scan_args.input_specifier_args.gcs_service_account = args.service_account;
None
}
ScanInputCommand::Docker(args) => {
if args.images.is_empty() && args.archives.is_empty() {
bail!(
"Provide at least one image or --archive path when using the docker subcommand"
);
}
scan_args.input_specifier_args.docker_image = args.images;
scan_args.input_specifier_args.docker_archive = args.archives;
None
}
};
if let Some(list_command) = maybe_list {
return Ok(ScanOperation::ListRepositories(list_command));
}
}
let used_legacy_git_url_flag = !self.scan_args.input_specifier_args.git_url.is_empty();
self.infer_positional_git_urls();
if !self.scan_args.input_specifier_args.has_any_input() {
bail!(
"Specify a path or Git URL (for example: 'kingfisher scan github.com/org/repo'), or use a provider subcommand such as 'kingfisher scan github'"
);
}
for path in &self.scan_args.input_specifier_args.path_inputs {
if path.as_path() == Path::new("-") {
continue;
}
if !path.exists() {
bail!("Error: unrecognized scan target or path does not exist: {}", path.display());
}
}
if !used_provider_subcommand {
self.scan_args.input_specifier_args.emit_deprecated_warnings(used_legacy_git_url_flag);
}
if self.scan_args.manage_baseline {
self.scan_args.no_dedup = true;
}
if self.scan_args.turbo {
self.scan_args.no_base64 = true;
self.scan_args.input_specifier_args.commit_metadata = false;
}
if self.scan_args.access_map && self.scan_args.no_validate {
bail!("--blast-radius cannot be used with --no-validate");
}
self.scan_args.validate_audit_log_collisions(None, None)?;
Ok(ScanOperation::Scan(self.scan_args))
}
}
fn paths_refer_to_same_file(a: &Path, b: &Path) -> bool {
fn normalize(path: &Path) -> Option<PathBuf> {
std::path::absolute(path)
.ok()
.map(|abs| abs.parse_dot().map(|dedotted| dedotted.into_owned()).unwrap_or(abs))
}
let (Some(a), Some(b)) = (normalize(a), normalize(b)) else { return a == b };
if a == b {
return true;
}
if let Ok(same) = same_file::is_same_file(&a, &b) {
return same;
}
match (canonicalize_best_effort(&a), canonicalize_best_effort(&b)) {
(Some(resolved_a), Some(resolved_b)) => {
resolved_a.to_string_lossy().to_lowercase()
== resolved_b.to_string_lossy().to_lowercase()
}
_ => false,
}
}
fn canonicalize_best_effort(path: &Path) -> Option<PathBuf> {
if let Ok(canonical) = fs::canonicalize(path) {
return Some(canonical);
}
let mut tail = vec![path.file_name()?.to_os_string()];
for ancestor in path.ancestors().skip(1) {
if let Ok(canonical) = fs::canonicalize(ancestor) {
let mut resolved = canonical;
for component in tail.iter().rev() {
resolved.push(component);
}
return Some(resolved);
}
tail.push(ancestor.file_name()?.to_os_string());
}
None
}
fn path_is_beneath(candidate: &Path, root: &Path) -> bool {
let candidate_path = root_for_comparison(candidate);
let root_path = root_for_comparison(root);
if candidate_path.starts_with(&root_path) && candidate_path != root_path {
return true;
}
let Some(candidate) = canonicalize_best_effort(&candidate_path) else {
return false;
};
let Some(root) = canonicalize_best_effort(&root_path) else {
return false;
};
candidate.starts_with(&root) && candidate != root
}
fn root_for_comparison(path: &Path) -> PathBuf {
std::path::absolute(path)
.ok()
.and_then(|absolute| absolute.parse_dot().ok().map(|path| path.into_owned()))
.unwrap_or_else(|| path.to_path_buf())
}
fn parse_git_url_target(path: &Path) -> Option<GitUrl> {
let raw = path.to_str()?.trim();
if raw.is_empty() || raw == "-" || raw.contains('\\') {
return None;
}
if let Ok(url) = GitUrl::from_str(raw) {
return Some(url);
}
if raw.contains("://")
|| raw.starts_with('/')
|| raw.starts_with("./")
|| raw.starts_with("../")
|| raw.starts_with('~')
{
return None;
}
let (host, suffix) = raw.split_once('/')?;
if host.is_empty() || suffix.is_empty() {
return None;
}
let path_segments = suffix.split('/').filter(|segment| !segment.is_empty()).count();
if path_segments < 2 {
return None;
}
let host_looks_valid =
host.contains('.') || host == "localhost" || host.parse::<IpAddr>().is_ok();
if !host_looks_valid {
return None;
}
GitUrl::from_str(&format!("https://{raw}")).ok()
}
#[derive(Subcommand, Debug, Clone)]
pub enum ScanInputCommand {
#[command(hide = true)]
Filesystem(FilesystemScanArgs),
Github(GithubScanArgs),
Gitlab(GitLabScanArgs),
Gitea(GiteaScanArgs),
Bitbucket(BitbucketScanArgs),
Azure(AzureScanArgs),
Huggingface(HuggingfaceScanArgs),
Slack(SlackScanArgs),
Teams(TeamsScanArgs),
Jira(JiraScanArgs),
Confluence(ConfluenceScanArgs),
Postman(PostmanScanArgs),
S3(S3ScanArgs),
Gcs(GcsScanArgs),
Docker(DockerScanArgs),
}
#[derive(Args, Debug, Clone, Default)]
pub struct FilesystemScanArgs {
#[arg(value_name = "PATH", value_hint = ValueHint::AnyPath)]
pub paths: Vec<PathBuf>,
#[arg(long = "git-url", value_hint = ValueHint::Url)]
pub git_url: Vec<GitUrl>,
}
#[derive(Args, Debug, Clone)]
pub struct GithubScanArgs {
#[command(flatten)]
pub specifiers: GitHubRepoSpecifiers,
#[arg(
long = "public-events",
alias = "events",
default_value_t = false,
conflicts_with = "include_gists"
)]
pub public_events: bool,
#[arg(long = "event-lookback-hours", value_name = "HOURS", default_value_t = 24)]
pub event_lookback_hours: u64,
#[arg(long = "user-file", value_name = "FILE", value_hint = ValueHint::FilePath)]
pub event_user_file: Option<PathBuf>,
#[arg(long = "include-contributors", default_value_t = false)]
pub include_contributors: bool,
#[arg(long = "repo-clone-limit", value_name = "COUNT")]
pub repo_clone_limit: Option<usize>,
#[arg(long = "list-only")]
pub list_only: bool,
#[arg(
long = "api-url",
alias = "github-api-url",
default_value = "https://api.github.com/",
value_hint = ValueHint::Url
)]
pub api_url: Url,
}
#[derive(Args, Debug, Clone)]
pub struct GitLabScanArgs {
#[command(flatten)]
pub specifiers: GitLabRepoSpecifiers,
#[arg(long = "include-contributors", default_value_t = false)]
pub include_contributors: bool,
#[arg(long = "repo-clone-limit", value_name = "COUNT")]
pub repo_clone_limit: Option<usize>,
#[arg(long = "list-only")]
pub list_only: bool,
#[arg(
long = "api-url",
alias = "gitlab-api-url",
default_value = "https://gitlab.com/",
value_hint = ValueHint::Url
)]
pub api_url: Url,
}
#[derive(Args, Debug, Clone)]
pub struct GiteaScanArgs {
#[command(flatten)]
pub specifiers: GiteaRepoSpecifiers,
#[arg(long = "list-only")]
pub list_only: bool,
#[arg(
long = "api-url",
alias = "gitea-api-url",
default_value = "https://gitea.com/api/v1/",
value_hint = ValueHint::Url
)]
pub api_url: Url,
}
#[derive(Args, Debug, Clone)]
pub struct BitbucketScanArgs {
#[command(flatten)]
pub specifiers: BitbucketRepoSpecifiers,
#[arg(long = "list-only")]
pub list_only: bool,
#[arg(
long = "api-url",
alias = "bitbucket-api-url",
default_value = "https://api.bitbucket.org/2.0/",
value_hint = ValueHint::Url
)]
pub api_url: Url,
}
#[derive(Args, Debug, Clone)]
pub struct AzureScanArgs {
#[command(flatten)]
pub specifiers: AzureRepoSpecifiers,
#[arg(long = "list-only")]
pub list_only: bool,
#[arg(
long = "base-url",
alias = "azure-base-url",
default_value = "https://dev.azure.com/",
value_hint = ValueHint::Url
)]
pub base_url: Url,
}
#[derive(Args, Debug, Clone, Default)]
pub struct HuggingfaceScanArgs {
#[command(flatten)]
pub specifiers: HuggingFaceRepoSpecifiers,
#[arg(long = "list-only")]
pub list_only: bool,
}
#[derive(Args, Debug, Clone)]
pub struct SlackScanArgs {
#[arg(value_name = "QUERY")]
pub query: String,
#[arg(
long = "api-url",
alias = "slack-api-url",
default_value = "https://slack.com/api/",
value_hint = ValueHint::Url
)]
pub api_url: Url,
#[arg(long = "max-results", default_value_t = 100)]
pub max_results: usize,
}
#[derive(Args, Debug, Clone)]
pub struct TeamsScanArgs {
#[arg(value_name = "QUERY")]
pub query: String,
#[arg(
long = "api-url",
alias = "teams-api-url",
default_value = "https://graph.microsoft.com/",
value_hint = ValueHint::Url
)]
pub api_url: Url,
#[arg(long = "max-results", default_value_t = 100)]
pub max_results: usize,
}
#[derive(Args, Debug, Clone)]
pub struct JiraScanArgs {
#[arg(long = "url", alias = "jira-url", value_hint = ValueHint::Url)]
pub url: Url,
#[arg(long, alias = "jql")]
pub jql: String,
#[arg(long = "max-results", default_value_t = 100)]
pub max_results: usize,
#[arg(long = "all", conflicts_with = "max_results")]
pub all: bool,
#[arg(long = "include-comments", default_value_t = false)]
pub include_comments: bool,
#[arg(long = "include-changelog", default_value_t = false)]
pub include_changelog: bool,
}
#[derive(Args, Debug, Clone)]
pub struct ConfluenceScanArgs {
#[arg(long = "url", alias = "confluence-url", value_hint = ValueHint::Url)]
pub url: Url,
#[arg(long, alias = "cql")]
pub cql: String,
#[arg(long = "max-results", default_value_t = 100)]
pub max_results: usize,
#[arg(long = "all", conflicts_with = "max_results")]
pub all: bool,
}
#[derive(Args, Debug, Clone)]
pub struct PostmanScanArgs {
#[arg(long = "workspace", alias = "postman-workspace", value_name = "ID_OR_URL")]
pub workspaces: Vec<String>,
#[arg(long = "collection", alias = "postman-collection", value_name = "UID_OR_URL")]
pub collections: Vec<String>,
#[arg(long = "environment", alias = "postman-environment", value_name = "UID")]
pub environments: Vec<String>,
#[arg(
long = "all",
alias = "postman-all",
conflicts_with_all = ["workspaces", "collections", "environments"],
)]
pub all: bool,
#[arg(long = "include-mocks-monitors", alias = "postman-include-mocks-monitors")]
pub include_mocks_monitors: bool,
#[arg(
long = "api-url",
alias = "postman-api-url",
default_value = "https://api.getpostman.com/",
value_hint = ValueHint::Url,
)]
pub api_url: Url,
#[arg(long = "max-results", default_value_t = 100)]
pub max_results: usize,
}
#[derive(Args, Debug, Clone)]
pub struct S3ScanArgs {
#[arg(value_name = "BUCKET")]
pub bucket: String,
#[arg(long = "prefix", alias = "s3-prefix")]
pub prefix: Option<String>,
#[arg(long = "role-arn")]
pub role_arn: Option<String>,
#[arg(long = "profile", alias = "aws-local-profile")]
pub profile: Option<String>,
}
#[derive(Args, Debug, Clone)]
pub struct GcsScanArgs {
#[arg(value_name = "BUCKET")]
pub bucket: String,
#[arg(long = "prefix", alias = "gcs-prefix")]
pub prefix: Option<String>,
#[arg(long = "service-account", alias = "gcs-service-account", value_hint = ValueHint::FilePath)]
pub service_account: Option<PathBuf>,
}
#[derive(Args, Debug, Clone)]
pub struct DockerScanArgs {
#[arg(value_name = "IMAGE")]
pub images: Vec<String>,
#[arg(long = "archive", value_name = "PATH", value_hint = ValueHint::FilePath)]
pub archives: Vec<PathBuf>,
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn equivalent_output_paths_are_detected() {
assert!(paths_refer_to_same_file(Path::new("report.json"), Path::new("report.json")));
assert!(paths_refer_to_same_file(Path::new("report.json"), Path::new("./report.json")));
assert!(paths_refer_to_same_file(Path::new("a/../report.json"), Path::new("report.json")));
assert!(!paths_refer_to_same_file(Path::new("report.json"), Path::new("other.json")));
}
#[test]
fn case_variant_names_are_treated_as_the_same_file() {
assert!(paths_refer_to_same_file(Path::new("report.json"), Path::new("Report.json")));
assert!(paths_refer_to_same_file(
Path::new("sub/report.json"),
Path::new("sub/REPORT.JSON")
));
assert!(!paths_refer_to_same_file(Path::new("Report.json"), Path::new("Repo.json")));
}
#[cfg(unix)]
#[test]
fn symlink_aliases_are_detected_for_existing_files() {
let dir = tempfile::tempdir().unwrap();
let target = dir.path().join("report.json");
std::fs::write(&target, b"x").unwrap();
let link = dir.path().join("alias.json");
std::os::unix::fs::symlink(&target, &link).unwrap();
assert!(paths_refer_to_same_file(&target, &link));
assert!(paths_refer_to_same_file(Path::new("."), Path::new("./")));
}
#[cfg(unix)]
#[test]
fn symlinked_parent_directories_are_detected_for_new_files() {
let dir = tempfile::tempdir().unwrap();
let real = dir.path().join("real");
std::fs::create_dir(&real).unwrap();
let link = dir.path().join("link");
std::os::unix::fs::symlink(&real, &link).unwrap();
assert!(paths_refer_to_same_file(&real.join("report.json"), &link.join("report.json")));
assert!(!paths_refer_to_same_file(&real.join("report.json"), &link.join("other.json")));
}
#[cfg(unix)]
#[test]
fn hard_link_aliases_are_detected() {
let dir = tempfile::tempdir().unwrap();
let a = dir.path().join("a.json");
std::fs::write(&a, b"x").unwrap();
let b = dir.path().join("b.json");
std::fs::hard_link(&a, &b).unwrap();
assert!(paths_refer_to_same_file(&a, &b));
}
#[test]
fn audit_log_inside_scanned_directory_is_rejected() {
let dir = tempfile::tempdir().unwrap();
let input = dir.path().join("workspace");
std::fs::create_dir(&input).unwrap();
let audit_log = input.join("audit.jsonl");
assert!(path_is_beneath(&audit_log, &input));
}
}