Skip to main content

omgbase_graph/
mask.rs

1//! Code masking with **byte**-preserving spans (§1).
2//!
3//! `spec/properties` §3.2 defines `mask_code`; `omgbase-properties`
4//! implements it one space per *character*, which the field scan cannot tell
5//! apart but which shifts byte offsets after any masked non-ASCII character.
6//! Node spans are bytes into the block's UTF-8 `raw` (§2.3), so the graph
7//! scanners run over this wrapper: the same masking decisions, with every
8//! masked character widened to as many spaces as its UTF-8 encoding is long.
9//! Offsets into the result are then offsets into `raw`.
10
11use omgbase_properties::mask_code;
12
13/// [`mask_code`] with `masked.len() == raw.len()` in bytes: a masked
14/// character of *n* UTF-8 bytes becomes *n* spaces; everything else is `raw`.
15#[must_use]
16pub fn mask_code_bytes(raw: &str) -> String {
17    let masked = mask_code(raw);
18    if masked == raw {
19        return masked;
20    }
21    let mut out = String::with_capacity(raw.len());
22    for (orig, m) in raw.chars().zip(masked.chars()) {
23        if m == orig {
24            out.push(orig);
25        } else {
26            debug_assert_eq!(m, ' ', "mask_code only blanks");
27            for _ in 0..orig.len_utf8() {
28                out.push(' ');
29            }
30        }
31    }
32    debug_assert_eq!(out.len(), raw.len());
33    out
34}
35
36#[cfg(test)]
37mod tests {
38    use super::*;
39
40    #[test]
41    fn preserves_byte_length() {
42        assert_eq!(mask_code_bytes("plain"), "plain");
43        assert_eq!(mask_code_bytes("a `é` b"), "a      b");
44        assert_eq!(mask_code_bytes("a `é` b").len(), "a `é` b".len());
45        assert_eq!(
46            mask_code_bytes("```\nk:: v\n```\nafter"),
47            "   \n     \n   \nafter"
48        );
49        let raw = "x `日本` [t](/p)";
50        let m = mask_code_bytes(raw);
51        assert_eq!(m.len(), raw.len());
52        assert_eq!(m.find("[t](/p)"), raw.find("[t](/p)"));
53    }
54}