1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
use crate::args::ScanArgs;
use keyhog_scanner::ScannerConfig;
#[derive(Debug, Clone)]
pub(super) struct ScannerConfigInput {
precision: bool,
fast: bool,
deep: bool,
decode_depth: Option<usize>,
no_decode: bool,
decode_size_limit: Option<usize>,
min_confidence: Option<f64>,
ml_threshold: Option<f64>,
no_suppress_test_fixtures: bool,
no_entropy: bool,
entropy_threshold: Option<f64>,
entropy_bpe_max_bytes_per_token: Option<f64>,
min_secret_len: Option<usize>,
per_chunk_timeout_ms: Option<u64>,
profile: bool,
perf_trace: bool,
entropy_source_files: bool,
no_entropy_ml_scoring: bool,
no_keyword_low_entropy: bool,
scan_comments: bool,
no_ml: bool,
ml_weight: Option<f64>,
no_unicode_norm: bool,
known_prefixes: Vec<String>,
secret_keywords: Vec<String>,
test_keywords: Vec<String>,
placeholder_keywords: Vec<String>,
}
impl ScannerConfigInput {
pub(super) fn from_scan_args(args: &ScanArgs) -> Self {
Self {
precision: args.precision,
fast: args.fast,
deep: args.deep,
decode_depth: args.decode_depth,
no_decode: args.no_decode,
decode_size_limit: args.decode_size_limit,
min_confidence: args.min_confidence,
ml_threshold: args.ml_threshold,
no_suppress_test_fixtures: args.no_suppress_test_fixtures,
no_entropy: args.no_entropy,
entropy_threshold: args.entropy_threshold,
entropy_bpe_max_bytes_per_token: args.entropy_bpe_max_bytes_per_token,
min_secret_len: args.min_secret_len,
per_chunk_timeout_ms: args.per_chunk_timeout_ms,
profile: args.profile,
perf_trace: args.perf_trace,
entropy_source_files: args.entropy_source_files,
no_entropy_ml_scoring: args.no_entropy_ml_scoring,
no_keyword_low_entropy: args.no_keyword_low_entropy,
scan_comments: args.scan_comments,
no_ml: args.no_ml,
ml_weight: args.ml_weight,
no_unicode_norm: args.no_unicode_norm,
known_prefixes: args.known_prefixes.clone(),
secret_keywords: args.secret_keywords.clone(),
test_keywords: args.test_keywords.clone(),
placeholder_keywords: args.placeholder_keywords.clone(),
}
}
}
pub(crate) fn build_scanner_config(args: &ScanArgs) -> ScannerConfig {
let input = ScannerConfigInput::from_scan_args(args);
build_scanner_config_from_input(&input)
}
pub(super) fn build_scanner_config_from_input(input: &ScannerConfigInput) -> ScannerConfig {
// The preset (`--fast` / `--deep`) is a BASE, not a terminal state. It
// seeds decode-depth / entropy / ml defaults; the per-flag overrides below
// then layer on top. Pre-fix this function early-returned at the preset, so
// `--deep --min-confidence 0.9` (or `--deep --entropy-threshold 5.0`, or any
// `--known-prefixes` / keyword list) silently dropped the explicit override
// - a coherence leak where "what the operator asked for" != "what ran". Only
// `--no-decode` / `--no-entropy` are clap-conflicting with the presets
// (`conflicts_with_all` on the `fast`/`deep` flags), so every other override
// is a legitimate refinement of the preset base and must take effect.
let mut config = if input.precision {
ScannerConfig::high_precision()
} else if input.fast {
ScannerConfig::fast()
} else if input.deep {
ScannerConfig::thorough()
} else {
ScannerConfig::default()
};
if let Some(depth) = input.decode_depth {
config.max_decode_depth = depth;
}
if input.no_decode {
config.max_decode_depth = 0;
}
if let Some(size) = input.decode_size_limit {
config.max_decode_bytes = size;
}
if let Some(conf) = input.min_confidence {
// Under `--precision` the 0.85 floor is a MINIMUM the operator may
// raise but not lower: `--precision --min-confidence 0.9` tightens to
// 0.9, while `--precision --min-confidence 0.3` stays at 0.85 (the
// documented "`--min-confidence` still overrides the floor on top"
// contract is one-directional - it cannot punch a hole in the precision
// bar). Every other mode lets the operator set the floor outright.
config.min_confidence = if input.precision {
conf.max(ScannerConfig::HIGH_PRECISION_MIN_CONFIDENCE)
} else {
conf
};
}
// `--ml-threshold` is the documented "minimum ML confidence score for
// generic entropy secrets" knob. Pre-fix it was parsed + range-validated
// but never read by any non-test path, so `--ml-threshold 0.9` silently did
// nothing (M21: a dead precision lever giving false confidence). Wire it as
// a confidence FLOOR composed with `.max()` - mirroring the precision-mode
// composition just above and the "minimum score" wording of the flag - so a
// raised threshold tightens the bar a generic/entropy finding must clear,
// while a lowered one can never punch below an operator's `--min-confidence`
// (or the precision floor). An unset flag leaves the canonical 0.40 floor
// untouched, so behaviour off the bug path is unchanged. Presence, not a
// float sentinel, is the intent signal: an explicit `--ml-threshold 0.5`
// is still an operator request and must not collapse into "unset".
if let Some(ml_threshold) = input.ml_threshold {
config.min_confidence = config.min_confidence.max(ml_threshold);
}
// Keep the fixture opt-out coherent: skip both value suppressions and the
// test/example path confidence penalty.
config.penalize_test_paths = !input.no_suppress_test_fixtures;
// `--no-entropy` is a one-way override. Clap rejects it with presets when
// both are typed on the CLI, but TOML defaults are merged after clap and can
// produce a preset + no_entropy combination. Honor the post-merge state
// instead of letting a config-file opt-out disappear under `fast`/`deep`.
if input.no_entropy {
config.entropy_enabled = false;
}
if let Some(threshold) = input.entropy_threshold {
config.entropy_threshold = threshold;
}
if let Some(bpe_bound) = input.entropy_bpe_max_bytes_per_token {
config.entropy_bpe_max_bytes_per_token = bpe_bound;
config.entropy_bpe_max_bytes_per_token_override = Some(bpe_bound);
}
if let Some(min_secret_len) = input.min_secret_len {
config.min_secret_len = min_secret_len;
}
config.per_chunk_timeout_ms = input.per_chunk_timeout_ms;
config.profile = input.profile;
config.perf_trace = input.perf_trace;
// Deep enables source-file entropy as part of its recovery contract. The
// positive flag can enable it in any other preset, but an absent flag must
// not erase the preset base.
config.entropy_in_source_files |= input.entropy_source_files;
// Detector TOMLs choose the entropy ML mode; this scan-wide negative flag
// can only disable those compiled modes. It cannot select model authority.
// No-op unless entropy and ML are both enabled.
config.entropy_ml_authoritative =
config.entropy_ml_authoritative && !input.no_entropy_ml_scoring;
// Keyword-anchored generic values use the relaxed entropy floor by default
// (the keyword key is the evidence; precision carried by the MoE);
// `--no-keyword-low-entropy` restores the high-entropy-only generic gate.
// No-op unless the generic keyword bridge fires (scan_generic_assignments).
// Composed with `&&` (not assigned) so the flag is one-directional: it can
// only DISABLE the relaxed floor, never re-enable it under a preset that
// turned it off (e.g. `--precision`, whose high_precision() base sets it
// false). Mirrors the one-directional precision min_confidence contract.
config.generic_keyword_low_entropy =
config.generic_keyword_low_entropy && !input.no_keyword_low_entropy;
// Deep treats comments as live leak surfaces. Preserve that preset base;
// `--scan-comments` remains the positive opt-in for other modes.
config.scan_comments |= input.scan_comments;
config.ml_enabled = !input.fast && !input.no_ml;
if let Some(weight) = input.ml_weight {
config.ml_weight = weight;
config.ml_weight_override = Some(weight);
}
config.unicode_normalization = !input.no_unicode_norm;
if !input.known_prefixes.is_empty() {
config.known_prefixes = input.known_prefixes.clone();
}
if !input.secret_keywords.is_empty() {
config.secret_keywords = input.secret_keywords.clone();
}
if !input.test_keywords.is_empty() {
config.test_keywords = input.test_keywords.clone();
}
if !input.placeholder_keywords.is_empty() {
config.placeholder_keywords = input.placeholder_keywords.clone();
}
// Re-run the NaN/range safety net AFTER every CLI flag and `.keyhog.toml`
// override has been merged in. `From<ScanConfig>` sanitises once at
// construction time, but the overrides above (e.g. `config.ml_weight =
// weight`, `config.entropy_threshold = threshold`) mutate the numeric
// fields directly afterwards and would otherwise smuggle out-of-range
// values straight to the engine: `--ml-weight 5.0` / `-1.0` (the ML blend
// `w*ml + (1-w)*heuristic` in scan_postprocess relies on `w in [0,1]`) and
// `--entropy-threshold 99` / `-5` (a threshold > 8.0 can never fire,
// disabling the entropy detector; a negative one makes `entropy >= thr`
// always true). Boundary parsers reject invalid CLI/TOML values; this
// defensive pass also covers programmatic construction. Idempotent.
config.sanitise();
config
}