khive_runtime/secret_gate/masking.rs
1use super::*;
2
3/// Upper bound on how many characters of a diagnostic-boundary input the
4/// canonical masker is ever asked to scan, regardless of a caller's own
5/// output cap. Chosen comfortably larger than the largest output cap among
6/// the callers of [`mask_bounded`] (1,024, `khive-mcp`'s
7/// `MAX_BACKEND_ERROR_MESSAGE_CHARS`) so a credential's terminating span
8/// remains inside the window for any message that is itself smaller than the
9/// window. Masking cost scales with this constant, never with a caller's raw
10/// input length. Shared by every diagnostic-boundary redaction site so the
11/// window cannot drift between them independently.
12pub const MASK_WINDOW_CHARS: usize = 4_096;
13
14/// Result of [`mask_bounded`]: masked, window- and output-bounded text plus
15/// the flags a caller needs to finish assembling a caller-visible
16/// diagnostic.
17#[derive(Debug, Clone, PartialEq, Eq)]
18pub struct BoundedMask {
19 /// Masked text, never longer than `output_cap_chars` plus one trailing
20 /// truncation-marker character. May be the bare truncation marker alone
21 /// when the retained window held a single token longer than the window.
22 pub text: String,
23 /// True when `text` is not the complete masked input: the raw input
24 /// exceeded `window_chars`, the masked window exceeded `output_cap_chars`,
25 /// or the window held one token longer than the window that had to be
26 /// dropped whole.
27 pub truncated: bool,
28 /// True when the canonical masker replaced a span inside the retained
29 /// window, or the window's only content was dropped whole for being an
30 /// oversized single token. A caller that decides whether raw content is
31 /// safe to echo back verbatim on this flag must also treat `truncated`
32 /// as reason enough on its own — content the window never retained is
33 /// exactly as unverified as content the masker redacted.
34 pub redacted: bool,
35}
36
37/// Bound a diagnostic-boundary masking call to at most `window_chars` of
38/// `text` before the canonical masker ever runs, then cap the masked result
39/// to `output_cap_chars`. Bounding happens BEFORE masking — not after, as a
40/// truncate-then-mask policy would — so scan cost is bounded by
41/// `window_chars` regardless of the caller's input length.
42///
43/// A window cut mid-token would let a masker that never saw the token's
44/// terminating shape (e.g. the `@` closing `scheme://user:pass@host`) emit
45/// the token's visible prefix unmasked — the exact split-secret hole a prior
46/// mask-the-full-input-first policy existed to close, at the cost of
47/// unbounded scan work. This function closes the hole without re-widening
48/// the window to the input's full length: any token straddling the window
49/// boundary is dropped in its entirety (back to the last whitespace inside
50/// the window) rather than masked, because a masker can only vouch for a
51/// token it saw whole. A single token that is itself longer than the window
52/// has no earlier whitespace to fall back to inside the window and is
53/// replaced by the truncation marker alone.
54///
55/// The same drop applies to a chain of bridged fragments straddling the
56/// boundary, not just the one partial token touching it: see
57/// `trailing_bridge_fragment_cut` (private to this module) for why a
58/// forward lookahead cannot bound this instead, and for the backward walk
59/// that closes it purely from data already inside the window.
60///
61/// `window_chars` must be at least `output_cap_chars`; debug builds assert
62/// this so a misconfigured call site fails loudly instead of silently
63/// capping tighter than it windows, and the cap is additionally clamped to
64/// `window_chars` in every build so a misconfigured call site cannot cap
65/// tighter than it windows even outside debug assertions.
66pub fn mask_bounded(
67 surface: RedactionSurface,
68 text: &str,
69 window_chars: usize,
70 output_cap_chars: usize,
71) -> BoundedMask {
72 debug_assert!(
73 window_chars >= output_cap_chars,
74 "the input window must stay at least as large as the output cap"
75 );
76 let output_cap_chars = output_cap_chars.min(window_chars);
77
78 let input_truncated = text.chars().nth(window_chars).is_some();
79 let mut window: String = text.chars().take(window_chars).collect();
80
81 if input_truncated {
82 match window
83 .char_indices()
84 .rev()
85 .find(|(_, ch)| ch.is_whitespace())
86 {
87 // Keep the whitespace itself; drop only the partial token after it.
88 Some((idx, ch)) => window.truncate(idx + ch.len_utf8()),
89 // A single token spans the whole window: nothing can be shown
90 // whole, so show nothing at all.
91 None => window.clear(),
92 }
93 }
94
95 if input_truncated && !window.is_empty() {
96 if let Some(cut) = trailing_bridge_fragment_cut(&window) {
97 window.truncate(cut);
98 }
99 }
100
101 if input_truncated && window.is_empty() {
102 return BoundedMask {
103 text: TRUNCATION_MARKER.to_string(),
104 truncated: true,
105 redacted: true,
106 };
107 }
108
109 debug_assert!(window.chars().count() <= window_chars);
110 let masked = mask_for_redaction_surface(surface, &window);
111 let redacted = masked.as_ref() != window.as_str();
112
113 let mut chars = masked.chars();
114 let mut bounded: String = chars.by_ref().take(output_cap_chars).collect();
115 let output_capped = chars.next().is_some();
116 if output_capped || input_truncated {
117 bounded.push_str(TRUNCATION_MARKER);
118 }
119
120 BoundedMask {
121 text: bounded,
122 truncated: input_truncated || output_capped,
123 redacted,
124 }
125}
126
127pub(super) const TRUNCATION_MARKER: &str = "…";
128
129/// Redact every detected secret span in `text`, replacing each with
130/// `***MASKED***`.
131///
132/// This is the masking counterpart to [`check`]: where `check` hard-blocks a
133/// write on the first match, `mask_secrets` is for content that must be emitted
134/// or stored with credentials stripped. Named Git/session/MCP callers enter it
135/// through [`mask_for_redaction_surface`]. It reuses the SAME canonical detector
136/// set as `check`/`scan`, so callers must never maintain a second, weaker masker.
137///
138/// Returns `Cow::Borrowed` when no secret is present (the common case), avoiding
139/// an allocation. Spans are discovered left to right against the ORIGINAL text,
140/// always evaluating trigger context over the full input — a high-entropy value
141/// whose only trigger word sits to the left of an earlier-redacted secret is
142/// still detected. Cumulative suffix-scan work is capped; when dense input
143/// exhausts the cap after a match, that match is extended through the remaining
144/// text so unscanned credentials cannot survive. See
145/// `docs/api/secret_gate.md#mask_secrets` for the scan-cursor mechanics.
146pub fn mask_secrets(text: &str) -> std::borrow::Cow<'_, str> {
147 let (spans, _scan_work_bytes) = collect_mask_spans(text);
148 if spans.is_empty() {
149 return std::borrow::Cow::Borrowed(text);
150 }
151 let mut out = String::with_capacity(text.len());
152 let mut cursor = 0;
153 for (start, end) in spans {
154 // Spans are non-overlapping and ascending (each starts at/after the prior
155 // `end`); `max(cursor)` is a defensive guard, never load-bearing.
156 let start = start.max(cursor);
157 out.push_str(&text[cursor..start]);
158 out.push_str(REDACTION_MARKER);
159 cursor = end.max(cursor);
160 }
161 out.push_str(&text[cursor..]);
162 std::borrow::Cow::Owned(out)
163}
164
165/// Collect absolute byte spans to redact and report cumulative scan bytes revisited.
166/// Exhausting the work budget extends the last confirmed secret span through the input tail.
167pub(super) fn collect_mask_spans(text: &str) -> (Vec<(usize, usize)>, usize) {
168 let base = text.as_ptr() as usize;
169 let context = EntropyScanContext::new(text);
170 let tokens = &context.tokens;
171 // Collect every secret span (absolute byte offsets into `text`) before
172 // writing any output, so trigger-context detection always sees the original
173 // string rather than the suffix after the previous redaction.
174 let mut spans: Vec<(usize, usize)> = Vec::new();
175 let mut from = 0;
176 let mut scan_work_bytes = 0usize;
177 while from < text.len() {
178 // Candidate enumeration may revisit the prefix of the original token
179 // containing `from`. Charge it as well as the remaining suffix; in a
180 // whitespace gap the scan still begins at `from`.
181 let token_index = tokens.partition_point(|&(offset, raw)| offset + raw.len() <= from);
182 let scan_start = tokens
183 .get(token_index)
184 .map_or(from, |&(offset, _)| offset.min(from));
185 let scan_len = text.len() - scan_start;
186 let next_scan_work = scan_work_bytes.saturating_add(scan_len);
187 if scan_work_bytes > 0 && next_scan_work > MAX_MASK_SCAN_WORK_BYTES {
188 // Every previous sweep ended at a confirmed match; extending that
189 // redaction through the remaining tail is fail-closed.
190 spans.last_mut().expect("a prior scan found a span").1 = text.len();
191 break;
192 }
193 scan_work_bytes = next_scan_work;
194 match scan_from(text, from, &context) {
195 Some((sub, _detector)) => {
196 let start = sub.as_ptr() as usize - base;
197 // The prefix detectors return whitespace-delimited tokens, so a
198 // credential glued to structural punctuation (JSON quotes/braces,
199 // sentence commas) carries that trailing punctuation into the
200 // match. Trim a conservative trailing set that can never be part
201 // of a credential, so redacting does not consume surrounding JSON
202 // or prose structure. `=` `/` `+` `.` `-` `_` are intentionally
203 // NOT trimmed — they are valid base64/JWT/key characters.
204 let core_len = sub
205 .trim_end_matches(['"', '\'', '`', '}', ']', ')', ',', ';'])
206 .len();
207 let end = extend_across_invisible_bridge(text, start + core_len.max(1));
208 push_mask_spans(text, start, end, &mut spans);
209 // `scan_from` only returns matches with start >= from, and `end`
210 // is strictly greater than `start`, so `from` strictly advances.
211 from = end;
212 }
213 None => break,
214 }
215 }
216 (spans, scan_work_bytes)
217}
218
219/// A character that splits a payload without showing anything: non-ASCII and not
220/// a letter or digit, so U+200B and its neighbours qualify while the letters of a
221/// non-ASCII password do not. The second half of that predicate is load-bearing:
222/// `redis://:密码@host` is ONE credential whose characters are non-ASCII, and a
223/// rule keyed on non-ASCII alone splits it and prints the password between two
224/// redaction markers.
225fn is_invisible_bridge_separator(c: char) -> bool {
226 !c.is_ascii() && !c.is_alphanumeric()
227}
228
229/// Byte offset a redaction must reach when the payload continues past `end`
230/// behind an INVISIBLE separator.
231///
232/// A gap made only of [`is_invisible_bridge_separator`] characters is not something
233/// a person types between a credential and the next word; it is how one payload is
234/// split so each half falls under a detector's length floor. Detection already
235/// reconstructs those chains ([`bridge_fragment_chain`]), but the masker redacted
236/// only the token the scan returned, so the rest of the same payload survived into
237/// stored text. Gaps holding any ASCII character — the ordinary spaces and newlines
238/// between a commit sha and the prose after it — are never walked, so this cannot
239/// eat surrounding text.
240pub(super) fn extend_across_invisible_bridge(text: &str, end: usize) -> usize {
241 let mut end = end;
242 for _ in 1..MAX_BRIDGE_FRAGMENTS {
243 let rest = &text[end..];
244 let Some(gap_len) = rest.find(|c: char| c.is_ascii_alphanumeric()) else {
245 break;
246 };
247 let gap = &rest[..gap_len];
248 if gap.is_empty() || !gap.chars().all(is_invisible_bridge_separator) {
249 break;
250 }
251 let fragment = &rest[gap_len..];
252 let fragment_len = fragment
253 .find(|c: char| !c.is_ascii_alphanumeric())
254 .unwrap_or(fragment.len());
255 if fragment_len < MIN_BRIDGE_FRAGMENT_LEN {
256 break;
257 }
258 end += gap_len + fragment_len;
259 }
260 end
261}
262
263/// Push the redaction spans for `text[start..end]`, breaking at
264/// [`is_invisible_bridge_separator`] characters so a separator that joined two
265/// fragments of one payload stays visible instead of being swallowed into a single
266/// marker.
267///
268/// A span holding no such character — every ordinary credential, base64 and JWT
269/// forms included, whose `.` `+` `/` `=` are ASCII, and non-ASCII passwords, whose
270/// letters are alphanumeric — is pushed whole, so this changes nothing for them. A
271/// span that yields no run at all is pushed whole as well: redacting more than
272/// necessary is the safe direction.
273fn push_mask_spans(text: &str, start: usize, end: usize, spans: &mut Vec<(usize, usize)>) {
274 let span = &text[start..end];
275 if !span.chars().any(is_invisible_bridge_separator) {
276 spans.push((start, end));
277 return;
278 }
279 let before = spans.len();
280 let mut run_start: Option<usize> = None;
281 for (offset, ch) in span.char_indices() {
282 if is_invisible_bridge_separator(ch) {
283 if let Some(run) = run_start.take() {
284 spans.push((start + run, start + offset));
285 }
286 } else {
287 run_start.get_or_insert(offset);
288 }
289 }
290 if let Some(run) = run_start {
291 spans.push((start + run, end));
292 }
293 if spans.len() == before {
294 spans.push((start, end));
295 }
296}
297
298/// Maximum characters of raw error text admitted to the masking pass.
299///
300/// This is NOT a tight bound like [`MAX_LOG_TEXT_OUTPUT_CHARS`] below — it exists only to
301/// stop a truly pathological input (gigabytes of attacker-controlled text funneled into one
302/// error/log line) from making the masking scan unbounded. It is a pure compute bound, not a
303/// safety bound: [`find_url_userinfo`] has no length limit on the password it recognizes, so
304/// no finite value of this constant can guarantee a credential never straddles it — a
305/// password longer than whatever this is set to always has a crossing case. The actual
306/// safety invariant lives in [`redact_crossing_boundary_url_userinfo`], the fallback
307/// [`bounded_masked_log_text`] runs after [`mask_secrets`]: it redacts any `scheme://user:`
308/// opening whose password run reaches this cut point without a terminating `@`, regardless
309/// of how long that password is. 1 MiB just keeps the scan itself cheap.
310pub(super) const MAX_LOG_TEXT_MASK_INPUT_CHARS: usize = 1_048_576;
311/// Maximum characters of masked error text emitted to a log record.
312pub(super) const MAX_LOG_TEXT_OUTPUT_CHARS: usize = 1_024;
313
314/// Bound and mask arbitrary error text for log emission.
315///
316/// Log records are a disclosure surface the same way wire errors are: they are
317/// shipped, aggregated, and read by consumers outside the process. Backend
318/// error text (gate backends included) can embed connection strings or
319/// credentials, so the FULL text (up to `MAX_LOG_TEXT_MASK_INPUT_CHARS`, a
320/// pure compute bound — see its doc comment) is masked with the canonical
321/// detector set before any truncation happens. Masking after truncation would
322/// let a secret whose tail sits past the bound lose the context (e.g. a URL's
323/// terminating `@`) a detector needs to recognize it, leaving its head
324/// unmasked in the log — truncate-then-mask must never replace
325/// mask-then-truncate here. Because `MAX_LOG_TEXT_MASK_INPUT_CHARS` is
326/// finite and `find_url_userinfo` has no bound on password length, a
327/// password long enough still crosses the cut before its terminating `@`
328/// ever appears; `redact_crossing_boundary_url_userinfo` closes that gap
329/// by redacting the unterminated opening directly, so no credential prefix
330/// survives regardless of secret length. Control (`Cc`), format (`Cf`), line
331/// separator (`Zl`), and paragraph separator (`Zp`) Unicode codepoints in the
332/// masked text are then escaped: a log line is plain text read by tooling
333/// outside this process, and an embedded line break or bidi/format override
334/// could forge or visually disguise part of the record. The result is bounded
335/// again for the emitted record. A truncation in either the masking pass or the
336/// output pass appends `…` so the record declares its own incompleteness.
337///
338/// This function, not [`mask_for_redaction_surface`], is the direct
339/// `mask_secrets` caller for general log-text bounding: it is not one of the
340/// three named redact-not-block surfaces (git ingest, session mirror, MCP
341/// diagnostics) that surface owns, and its own mask-then-truncate contract —
342/// with the additional crossing-boundary fallback above — is a strict
343/// superset of what the surface wrapper provides. It lives in this module
344/// specifically so it can stay a direct caller; the call-site census in
345/// `crates/khive-runtime/tests/adr115_redaction_call_site_census.rs` only
346/// requires callers *outside* this file to route through the wrapper.
347pub fn bounded_masked_log_text(text: &str) -> String {
348 let mask_input_truncated = text.chars().nth(MAX_LOG_TEXT_MASK_INPUT_CHARS).is_some();
349 let bounded_input: std::borrow::Cow<'_, str> = if mask_input_truncated {
350 std::borrow::Cow::Owned(text.chars().take(MAX_LOG_TEXT_MASK_INPUT_CHARS).collect())
351 } else {
352 std::borrow::Cow::Borrowed(text)
353 };
354 let masked = mask_secrets(&bounded_input);
355 let masked = if mask_input_truncated {
356 redact_crossing_boundary_url_userinfo(&masked)
357 } else {
358 masked
359 };
360 let neutralized = neutralize_log_unsafe_chars(&masked);
361
362 let mut chars = neutralized.chars();
363 let mut bounded: String = chars.by_ref().take(MAX_LOG_TEXT_OUTPUT_CHARS).collect();
364 if chars.next().is_some() || mask_input_truncated {
365 bounded.push('…');
366 }
367 bounded
368}
369
370/// Fallback for a `scheme://user:<password>` credential whose password run
371/// collided with [`bounded_masked_log_text`]'s truncation of the mask-scan
372/// input at [`MAX_LOG_TEXT_MASK_INPUT_CHARS`]. [`find_url_userinfo`] only
373/// recognizes a credential once it sees the terminating `@`; when that `@`
374/// sits past the truncation point, [`mask_secrets`] never sees the shape at
375/// all and the raw `scheme://user:<password prefix>` reaches the log. No
376/// finite value of [`MAX_LOG_TEXT_MASK_INPUT_CHARS`] can rule this out — the
377/// detector is unbounded, so any cap has a crossing case — so the invariant
378/// has to come from this fallback, not from the cap's size.
379///
380/// Only called when `bounded_masked_log_text` actually truncated the input.
381/// It scans `://` occurrences left to right and redacts at the EARLIEST
382/// unterminated `user:<password-run>` opening: a colon splits the tail into
383/// two non-empty pieces and no `@`, space, or newline appears anywhere from
384/// that occurrence to the end of the truncated text. An occurrence whose
385/// tail does contain one of those terminators is a complete URL or ordinary
386/// prose that ends inside the text (e.g. two URLs logged side by side) and
387/// is skipped, not redacted. The anchor must be the earliest such opening,
388/// never the last: a later `://` can sit INSIDE the crossing password
389/// itself (passwords may contain `://`), and anchoring there would leave
390/// the real `user:<password prefix>` before it in the emitted log. From the
391/// earliest unterminated opening's colon onward everything is redacted —
392/// zero password characters survive, no matter how long the password
393/// actually is.
394fn redact_crossing_boundary_url_userinfo(text: &str) -> std::borrow::Cow<'_, str> {
395 let mut search_from = 0usize;
396 while let Some(rel) = text[search_from..].find("://") {
397 let scheme_pos = search_from + rel;
398 let rest = &text[scheme_pos + 3..];
399 let terminated =
400 rest.contains('@') || rest.contains(' ') || rest.contains('\n') || rest.contains('\r');
401 if !terminated {
402 // Same rules as `find_url_userinfo`: the userinfo colon must sit
403 // in the authority component (before any `/`, `?`, or `#` — a
404 // later colon is path/query text), and only the password must be
405 // non-empty (an empty username, `redis://:pass`, is a standard
406 // connection-string form and no less a credential). The password
407 // run AFTER the colon is unrestricted — a crossing password may
408 // itself contain any of those delimiters.
409 let authority_end = rest.find(['/', '?', '#']).unwrap_or(rest.len());
410 if let Some(colon) = rest[..authority_end].find(':') {
411 let pass = &rest[colon + 1..];
412 if !pass.is_empty() {
413 let redact_from = scheme_pos + 3 + colon;
414 let mut out = String::with_capacity(redact_from + REDACTION_MARKER.len());
415 out.push_str(&text[..redact_from]);
416 out.push_str(REDACTION_MARKER);
417 return std::borrow::Cow::Owned(out);
418 }
419 }
420 }
421 search_from = scheme_pos + 3;
422 }
423 std::borrow::Cow::Borrowed(text)
424}
425
426/// `true` for a Unicode control (`Cc`), format (`Cf`), line separator (`Zl`), or
427/// paragraph separator (`Zp`) codepoint, tab excepted.
428///
429/// Classification is by Unicode general category rather than an ASCII byte range so that
430/// multi-byte control/format characters (bidi overrides, zero-width joiners, line/paragraph
431/// separators encoded as UTF-8) are caught the same way as single-byte C0 controls like
432/// CR/LF — a byte-range check would only ever see the latter. Tab is excepted: it is
433/// visually inert in a log line and common in legitimately reformatted prose.
434fn is_log_unsafe_char(c: char) -> bool {
435 if c == '\t' {
436 return false;
437 }
438 matches!(
439 unicode_general_category::get_general_category(c),
440 unicode_general_category::GeneralCategory::Control
441 | unicode_general_category::GeneralCategory::Format
442 | unicode_general_category::GeneralCategory::LineSeparator
443 | unicode_general_category::GeneralCategory::ParagraphSeparator
444 )
445}
446
447/// Escape every [`is_log_unsafe_char`] codepoint in `text` as `\u{XXXX}`.
448///
449/// Returns `Cow::Borrowed` when nothing needs escaping (the common case), avoiding an
450/// allocation. This runs on already-masked text: it must never be skipped for text that
451/// bypassed [`mask_secrets`], since a control character can sit inside a would-be secret
452/// span and is a distinct disclosure vector from the credential detectors (log injection /
453/// forgery, not credential leakage).
454fn neutralize_log_unsafe_chars(text: &str) -> std::borrow::Cow<'_, str> {
455 if !text.chars().any(is_log_unsafe_char) {
456 return std::borrow::Cow::Borrowed(text);
457 }
458 let mut out = String::with_capacity(text.len());
459 for c in text.chars() {
460 if is_log_unsafe_char(c) {
461 out.push_str(&format!("\\u{{{:04x}}}", c as u32));
462 } else {
463 out.push(c);
464 }
465 }
466 std::borrow::Cow::Owned(out)
467}