Skip to main content

khive_runtime/secret_gate/
masking.rs

1use super::*;
2
3/// Upper bound on how many characters of a diagnostic-boundary input the
4/// canonical masker is ever asked to scan, regardless of a caller's own
5/// output cap. Chosen comfortably larger than the largest output cap among
6/// the callers of [`mask_bounded`] (1,024, `khive-mcp`'s
7/// `MAX_BACKEND_ERROR_MESSAGE_CHARS`) so a credential's terminating span
8/// remains inside the window for any message that is itself smaller than the
9/// window. Masking cost scales with this constant, never with a caller's raw
10/// input length. Shared by every diagnostic-boundary redaction site so the
11/// window cannot drift between them independently.
12pub const MASK_WINDOW_CHARS: usize = 4_096;
13
14/// Result of [`mask_bounded`]: masked, window- and output-bounded text plus
15/// the flags a caller needs to finish assembling a caller-visible
16/// diagnostic.
17#[derive(Debug, Clone, PartialEq, Eq)]
18pub struct BoundedMask {
19    /// Masked text, never longer than `output_cap_chars` plus one trailing
20    /// truncation-marker character. May be the bare truncation marker alone
21    /// when the retained window held a single token longer than the window.
22    pub text: String,
23    /// True when `text` is not the complete masked input: the raw input
24    /// exceeded `window_chars`, the masked window exceeded `output_cap_chars`,
25    /// or the window held one token longer than the window that had to be
26    /// dropped whole.
27    pub truncated: bool,
28    /// True when the canonical masker replaced a span inside the retained
29    /// window, or the window's only content was dropped whole for being an
30    /// oversized single token. A caller that decides whether raw content is
31    /// safe to echo back verbatim on this flag must also treat `truncated`
32    /// as reason enough on its own — content the window never retained is
33    /// exactly as unverified as content the masker redacted.
34    pub redacted: bool,
35}
36
37/// Bound a diagnostic-boundary masking call to at most `window_chars` of
38/// `text` before the canonical masker ever runs, then cap the masked result
39/// to `output_cap_chars`. Bounding happens BEFORE masking — not after, as a
40/// truncate-then-mask policy would — so scan cost is bounded by
41/// `window_chars` regardless of the caller's input length.
42///
43/// A window cut mid-token would let a masker that never saw the token's
44/// terminating shape (e.g. the `@` closing `scheme://user:pass@host`) emit
45/// the token's visible prefix unmasked — the exact split-secret hole a prior
46/// mask-the-full-input-first policy existed to close, at the cost of
47/// unbounded scan work. This function closes the hole without re-widening
48/// the window to the input's full length: any token straddling the window
49/// boundary is dropped in its entirety (back to the last whitespace inside
50/// the window) rather than masked, because a masker can only vouch for a
51/// token it saw whole. A single token that is itself longer than the window
52/// has no earlier whitespace to fall back to inside the window and is
53/// replaced by the truncation marker alone.
54///
55/// The same drop applies to a chain of bridged fragments straddling the
56/// boundary, not just the one partial token touching it: see
57/// `trailing_bridge_fragment_cut` (private to this module) for why a
58/// forward lookahead cannot bound this instead, and for the backward walk
59/// that closes it purely from data already inside the window.
60///
61/// `window_chars` must be at least `output_cap_chars`; debug builds assert
62/// this so a misconfigured call site fails loudly instead of silently
63/// capping tighter than it windows, and the cap is additionally clamped to
64/// `window_chars` in every build so a misconfigured call site cannot cap
65/// tighter than it windows even outside debug assertions.
66pub fn mask_bounded(
67    surface: RedactionSurface,
68    text: &str,
69    window_chars: usize,
70    output_cap_chars: usize,
71) -> BoundedMask {
72    debug_assert!(
73        window_chars >= output_cap_chars,
74        "the input window must stay at least as large as the output cap"
75    );
76    let output_cap_chars = output_cap_chars.min(window_chars);
77
78    let input_truncated = text.chars().nth(window_chars).is_some();
79    let mut window: String = text.chars().take(window_chars).collect();
80
81    if input_truncated {
82        match window
83            .char_indices()
84            .rev()
85            .find(|(_, ch)| ch.is_whitespace())
86        {
87            // Keep the whitespace itself; drop only the partial token after it.
88            Some((idx, ch)) => window.truncate(idx + ch.len_utf8()),
89            // A single token spans the whole window: nothing can be shown
90            // whole, so show nothing at all.
91            None => window.clear(),
92        }
93    }
94
95    if input_truncated && !window.is_empty() {
96        if let Some(cut) = trailing_bridge_fragment_cut(&window) {
97            window.truncate(cut);
98        }
99    }
100
101    if input_truncated && window.is_empty() {
102        return BoundedMask {
103            text: TRUNCATION_MARKER.to_string(),
104            truncated: true,
105            redacted: true,
106        };
107    }
108
109    debug_assert!(window.chars().count() <= window_chars);
110    let masked = mask_for_redaction_surface(surface, &window);
111    let redacted = masked.as_ref() != window.as_str();
112
113    let mut chars = masked.chars();
114    let mut bounded: String = chars.by_ref().take(output_cap_chars).collect();
115    let output_capped = chars.next().is_some();
116    if output_capped || input_truncated {
117        bounded.push_str(TRUNCATION_MARKER);
118    }
119
120    BoundedMask {
121        text: bounded,
122        truncated: input_truncated || output_capped,
123        redacted,
124    }
125}
126
127pub(super) const TRUNCATION_MARKER: &str = "…";
128
129/// Redact every detected secret span in `text`, replacing each with
130/// `***MASKED***`.
131///
132/// This is the masking counterpart to [`check`]: where `check` hard-blocks a
133/// write on the first match, `mask_secrets` is for content that must be emitted
134/// or stored with credentials stripped. Named Git/session/MCP callers enter it
135/// through [`mask_for_redaction_surface`]. It reuses the SAME canonical detector
136/// set as `check`/`scan`, so callers must never maintain a second, weaker masker.
137///
138/// Returns `Cow::Borrowed` when no secret is present (the common case), avoiding
139/// an allocation. Spans are discovered left to right against the ORIGINAL text,
140/// always evaluating trigger context over the full input — a high-entropy value
141/// whose only trigger word sits to the left of an earlier-redacted secret is
142/// still detected. Cumulative suffix-scan work is capped; when dense input
143/// exhausts the cap after a match, that match is extended through the remaining
144/// text so unscanned credentials cannot survive. See
145/// `docs/api/secret_gate.md#mask_secrets` for the scan-cursor mechanics.
146pub fn mask_secrets(text: &str) -> std::borrow::Cow<'_, str> {
147    let (spans, _scan_work_bytes) = collect_mask_spans(text);
148    if spans.is_empty() {
149        return std::borrow::Cow::Borrowed(text);
150    }
151    let mut out = String::with_capacity(text.len());
152    let mut cursor = 0;
153    for (start, end) in spans {
154        // Spans are non-overlapping and ascending (each starts at/after the prior
155        // `end`); `max(cursor)` is a defensive guard, never load-bearing.
156        let start = start.max(cursor);
157        out.push_str(&text[cursor..start]);
158        out.push_str(REDACTION_MARKER);
159        cursor = end.max(cursor);
160    }
161    out.push_str(&text[cursor..]);
162    std::borrow::Cow::Owned(out)
163}
164
165/// Collect absolute byte spans to redact and report cumulative scan bytes revisited.
166/// Exhausting the work budget extends the last confirmed secret span through the input tail.
167pub(super) fn collect_mask_spans(text: &str) -> (Vec<(usize, usize)>, usize) {
168    let base = text.as_ptr() as usize;
169    let context = EntropyScanContext::new(text);
170    let tokens = &context.tokens;
171    // Collect every secret span (absolute byte offsets into `text`) before
172    // writing any output, so trigger-context detection always sees the original
173    // string rather than the suffix after the previous redaction.
174    let mut spans: Vec<(usize, usize)> = Vec::new();
175    let mut from = 0;
176    let mut scan_work_bytes = 0usize;
177    while from < text.len() {
178        // Candidate enumeration may revisit the prefix of the original token
179        // containing `from`. Charge it as well as the remaining suffix; in a
180        // whitespace gap the scan still begins at `from`.
181        let token_index = tokens.partition_point(|&(offset, raw)| offset + raw.len() <= from);
182        let scan_start = tokens
183            .get(token_index)
184            .map_or(from, |&(offset, _)| offset.min(from));
185        let scan_len = text.len() - scan_start;
186        let next_scan_work = scan_work_bytes.saturating_add(scan_len);
187        if scan_work_bytes > 0 && next_scan_work > MAX_MASK_SCAN_WORK_BYTES {
188            // Every previous sweep ended at a confirmed match; extending that
189            // redaction through the remaining tail is fail-closed.
190            spans.last_mut().expect("a prior scan found a span").1 = text.len();
191            break;
192        }
193        scan_work_bytes = next_scan_work;
194        match scan_from(text, from, &context) {
195            Some((sub, _detector)) => {
196                let start = sub.as_ptr() as usize - base;
197                // The prefix detectors return whitespace-delimited tokens, so a
198                // credential glued to structural punctuation (JSON quotes/braces,
199                // sentence commas) carries that trailing punctuation into the
200                // match. Trim a conservative trailing set that can never be part
201                // of a credential, so redacting does not consume surrounding JSON
202                // or prose structure. `=` `/` `+` `.` `-` `_` are intentionally
203                // NOT trimmed — they are valid base64/JWT/key characters.
204                let core_len = sub
205                    .trim_end_matches(['"', '\'', '`', '}', ']', ')', ',', ';'])
206                    .len();
207                let end = extend_across_invisible_bridge(text, start + core_len.max(1));
208                push_mask_spans(text, start, end, &mut spans);
209                // `scan_from` only returns matches with start >= from, and `end`
210                // is strictly greater than `start`, so `from` strictly advances.
211                from = end;
212            }
213            None => break,
214        }
215    }
216    (spans, scan_work_bytes)
217}
218
219/// A character that splits a payload without showing anything: non-ASCII and not
220/// a letter or digit, so U+200B and its neighbours qualify while the letters of a
221/// non-ASCII password do not. The second half of that predicate is load-bearing:
222/// `redis://:密码@host` is ONE credential whose characters are non-ASCII, and a
223/// rule keyed on non-ASCII alone splits it and prints the password between two
224/// redaction markers.
225fn is_invisible_bridge_separator(c: char) -> bool {
226    !c.is_ascii() && !c.is_alphanumeric()
227}
228
229/// Byte offset a redaction must reach when the payload continues past `end`
230/// behind an INVISIBLE separator.
231///
232/// A gap made only of [`is_invisible_bridge_separator`] characters is not something
233/// a person types between a credential and the next word; it is how one payload is
234/// split so each half falls under a detector's length floor. Detection already
235/// reconstructs those chains ([`bridge_fragment_chain`]), but the masker redacted
236/// only the token the scan returned, so the rest of the same payload survived into
237/// stored text. Gaps holding any ASCII character — the ordinary spaces and newlines
238/// between a commit sha and the prose after it — are never walked, so this cannot
239/// eat surrounding text.
240pub(super) fn extend_across_invisible_bridge(text: &str, end: usize) -> usize {
241    let mut end = end;
242    for _ in 1..MAX_BRIDGE_FRAGMENTS {
243        let rest = &text[end..];
244        let Some(gap_len) = rest.find(|c: char| c.is_ascii_alphanumeric()) else {
245            break;
246        };
247        let gap = &rest[..gap_len];
248        if gap.is_empty() || !gap.chars().all(is_invisible_bridge_separator) {
249            break;
250        }
251        let fragment = &rest[gap_len..];
252        let fragment_len = fragment
253            .find(|c: char| !c.is_ascii_alphanumeric())
254            .unwrap_or(fragment.len());
255        if fragment_len < MIN_BRIDGE_FRAGMENT_LEN {
256            break;
257        }
258        end += gap_len + fragment_len;
259    }
260    end
261}
262
263/// Push the redaction spans for `text[start..end]`, breaking at
264/// [`is_invisible_bridge_separator`] characters so a separator that joined two
265/// fragments of one payload stays visible instead of being swallowed into a single
266/// marker.
267///
268/// A span holding no such character — every ordinary credential, base64 and JWT
269/// forms included, whose `.` `+` `/` `=` are ASCII, and non-ASCII passwords, whose
270/// letters are alphanumeric — is pushed whole, so this changes nothing for them. A
271/// span that yields no run at all is pushed whole as well: redacting more than
272/// necessary is the safe direction.
273fn push_mask_spans(text: &str, start: usize, end: usize, spans: &mut Vec<(usize, usize)>) {
274    let span = &text[start..end];
275    if !span.chars().any(is_invisible_bridge_separator) {
276        spans.push((start, end));
277        return;
278    }
279    let before = spans.len();
280    let mut run_start: Option<usize> = None;
281    for (offset, ch) in span.char_indices() {
282        if is_invisible_bridge_separator(ch) {
283            if let Some(run) = run_start.take() {
284                spans.push((start + run, start + offset));
285            }
286        } else {
287            run_start.get_or_insert(offset);
288        }
289    }
290    if let Some(run) = run_start {
291        spans.push((start + run, end));
292    }
293    if spans.len() == before {
294        spans.push((start, end));
295    }
296}
297
298/// Maximum characters of raw error text admitted to the masking pass.
299///
300/// This is NOT a tight bound like [`MAX_LOG_TEXT_OUTPUT_CHARS`] below — it exists only to
301/// stop a truly pathological input (gigabytes of attacker-controlled text funneled into one
302/// error/log line) from making the masking scan unbounded. It is a pure compute bound, not a
303/// safety bound: [`find_url_userinfo`] has no length limit on the password it recognizes, so
304/// no finite value of this constant can guarantee a credential never straddles it — a
305/// password longer than whatever this is set to always has a crossing case. The actual
306/// safety invariant lives in [`redact_crossing_boundary_url_userinfo`], the fallback
307/// [`bounded_masked_log_text`] runs after [`mask_secrets`]: it redacts any `scheme://user:`
308/// opening whose password run reaches this cut point without a terminating `@`, regardless
309/// of how long that password is. 1 MiB just keeps the scan itself cheap.
310pub(super) const MAX_LOG_TEXT_MASK_INPUT_CHARS: usize = 1_048_576;
311/// Maximum characters of masked error text emitted to a log record.
312pub(super) const MAX_LOG_TEXT_OUTPUT_CHARS: usize = 1_024;
313
314/// Bound and mask arbitrary error text for log emission.
315///
316/// Log records are a disclosure surface the same way wire errors are: they are
317/// shipped, aggregated, and read by consumers outside the process. Backend
318/// error text (gate backends included) can embed connection strings or
319/// credentials, so the FULL text (up to `MAX_LOG_TEXT_MASK_INPUT_CHARS`, a
320/// pure compute bound — see its doc comment) is masked with the canonical
321/// detector set before any truncation happens. Masking after truncation would
322/// let a secret whose tail sits past the bound lose the context (e.g. a URL's
323/// terminating `@`) a detector needs to recognize it, leaving its head
324/// unmasked in the log — truncate-then-mask must never replace
325/// mask-then-truncate here. Because `MAX_LOG_TEXT_MASK_INPUT_CHARS` is
326/// finite and `find_url_userinfo` has no bound on password length, a
327/// password long enough still crosses the cut before its terminating `@`
328/// ever appears; `redact_crossing_boundary_url_userinfo` closes that gap
329/// by redacting the unterminated opening directly, so no credential prefix
330/// survives regardless of secret length. Control (`Cc`), format (`Cf`), line
331/// separator (`Zl`), and paragraph separator (`Zp`) Unicode codepoints in the
332/// masked text are then escaped: a log line is plain text read by tooling
333/// outside this process, and an embedded line break or bidi/format override
334/// could forge or visually disguise part of the record. The result is bounded
335/// again for the emitted record. A truncation in either the masking pass or the
336/// output pass appends `…` so the record declares its own incompleteness.
337///
338/// This function, not [`mask_for_redaction_surface`], is the direct
339/// `mask_secrets` caller for general log-text bounding: it is not one of the
340/// three named redact-not-block surfaces (git ingest, session mirror, MCP
341/// diagnostics) that surface owns, and its own mask-then-truncate contract —
342/// with the additional crossing-boundary fallback above — is a strict
343/// superset of what the surface wrapper provides. It lives in this module
344/// specifically so it can stay a direct caller; the call-site census in
345/// `crates/khive-runtime/tests/adr115_redaction_call_site_census.rs` only
346/// requires callers *outside* this file to route through the wrapper.
347pub fn bounded_masked_log_text(text: &str) -> String {
348    let mask_input_truncated = text.chars().nth(MAX_LOG_TEXT_MASK_INPUT_CHARS).is_some();
349    let bounded_input: std::borrow::Cow<'_, str> = if mask_input_truncated {
350        std::borrow::Cow::Owned(text.chars().take(MAX_LOG_TEXT_MASK_INPUT_CHARS).collect())
351    } else {
352        std::borrow::Cow::Borrowed(text)
353    };
354    let masked = mask_secrets(&bounded_input);
355    let masked = if mask_input_truncated {
356        redact_crossing_boundary_url_userinfo(&masked)
357    } else {
358        masked
359    };
360    let neutralized = neutralize_log_unsafe_chars(&masked);
361
362    let mut chars = neutralized.chars();
363    let mut bounded: String = chars.by_ref().take(MAX_LOG_TEXT_OUTPUT_CHARS).collect();
364    if chars.next().is_some() || mask_input_truncated {
365        bounded.push('…');
366    }
367    bounded
368}
369
370/// Fallback for a `scheme://user:<password>` credential whose password run
371/// collided with [`bounded_masked_log_text`]'s truncation of the mask-scan
372/// input at [`MAX_LOG_TEXT_MASK_INPUT_CHARS`]. [`find_url_userinfo`] only
373/// recognizes a credential once it sees the terminating `@`; when that `@`
374/// sits past the truncation point, [`mask_secrets`] never sees the shape at
375/// all and the raw `scheme://user:<password prefix>` reaches the log. No
376/// finite value of [`MAX_LOG_TEXT_MASK_INPUT_CHARS`] can rule this out — the
377/// detector is unbounded, so any cap has a crossing case — so the invariant
378/// has to come from this fallback, not from the cap's size.
379///
380/// Only called when `bounded_masked_log_text` actually truncated the input.
381/// It scans `://` occurrences left to right and redacts at the EARLIEST
382/// unterminated `user:<password-run>` opening: a colon splits the tail into
383/// two non-empty pieces and no `@`, space, or newline appears anywhere from
384/// that occurrence to the end of the truncated text. An occurrence whose
385/// tail does contain one of those terminators is a complete URL or ordinary
386/// prose that ends inside the text (e.g. two URLs logged side by side) and
387/// is skipped, not redacted. The anchor must be the earliest such opening,
388/// never the last: a later `://` can sit INSIDE the crossing password
389/// itself (passwords may contain `://`), and anchoring there would leave
390/// the real `user:<password prefix>` before it in the emitted log. From the
391/// earliest unterminated opening's colon onward everything is redacted —
392/// zero password characters survive, no matter how long the password
393/// actually is.
394fn redact_crossing_boundary_url_userinfo(text: &str) -> std::borrow::Cow<'_, str> {
395    let mut search_from = 0usize;
396    while let Some(rel) = text[search_from..].find("://") {
397        let scheme_pos = search_from + rel;
398        let rest = &text[scheme_pos + 3..];
399        let terminated =
400            rest.contains('@') || rest.contains(' ') || rest.contains('\n') || rest.contains('\r');
401        if !terminated {
402            // Same rules as `find_url_userinfo`: the userinfo colon must sit
403            // in the authority component (before any `/`, `?`, or `#` — a
404            // later colon is path/query text), and only the password must be
405            // non-empty (an empty username, `redis://:pass`, is a standard
406            // connection-string form and no less a credential). The password
407            // run AFTER the colon is unrestricted — a crossing password may
408            // itself contain any of those delimiters.
409            let authority_end = rest.find(['/', '?', '#']).unwrap_or(rest.len());
410            if let Some(colon) = rest[..authority_end].find(':') {
411                let pass = &rest[colon + 1..];
412                if !pass.is_empty() {
413                    let redact_from = scheme_pos + 3 + colon;
414                    let mut out = String::with_capacity(redact_from + REDACTION_MARKER.len());
415                    out.push_str(&text[..redact_from]);
416                    out.push_str(REDACTION_MARKER);
417                    return std::borrow::Cow::Owned(out);
418                }
419            }
420        }
421        search_from = scheme_pos + 3;
422    }
423    std::borrow::Cow::Borrowed(text)
424}
425
426/// `true` for a Unicode control (`Cc`), format (`Cf`), line separator (`Zl`), or
427/// paragraph separator (`Zp`) codepoint, tab excepted.
428///
429/// Classification is by Unicode general category rather than an ASCII byte range so that
430/// multi-byte control/format characters (bidi overrides, zero-width joiners, line/paragraph
431/// separators encoded as UTF-8) are caught the same way as single-byte C0 controls like
432/// CR/LF — a byte-range check would only ever see the latter. Tab is excepted: it is
433/// visually inert in a log line and common in legitimately reformatted prose.
434fn is_log_unsafe_char(c: char) -> bool {
435    if c == '\t' {
436        return false;
437    }
438    matches!(
439        unicode_general_category::get_general_category(c),
440        unicode_general_category::GeneralCategory::Control
441            | unicode_general_category::GeneralCategory::Format
442            | unicode_general_category::GeneralCategory::LineSeparator
443            | unicode_general_category::GeneralCategory::ParagraphSeparator
444    )
445}
446
447/// Escape every [`is_log_unsafe_char`] codepoint in `text` as `\u{XXXX}`.
448///
449/// Returns `Cow::Borrowed` when nothing needs escaping (the common case), avoiding an
450/// allocation. This runs on already-masked text: it must never be skipped for text that
451/// bypassed [`mask_secrets`], since a control character can sit inside a would-be secret
452/// span and is a distinct disclosure vector from the credential detectors (log injection /
453/// forgery, not credential leakage).
454fn neutralize_log_unsafe_chars(text: &str) -> std::borrow::Cow<'_, str> {
455    if !text.chars().any(is_log_unsafe_char) {
456        return std::borrow::Cow::Borrowed(text);
457    }
458    let mut out = String::with_capacity(text.len());
459    for c in text.chars() {
460        if is_log_unsafe_char(c) {
461            out.push_str(&format!("\\u{{{:04x}}}", c as u32));
462        } else {
463            out.push(c);
464        }
465    }
466    std::borrow::Cow::Owned(out)
467}