Skip to main content

safe_chains/cst/
ansi_c.rs

1//! Bash's two `$`-quotes: ANSI-C `$'…'` and locale `$"…"`.
2//!
3//! Both must be read as bash reads them. Left to the generic `$` rule, `$'-delete'` became the
4//! literal `$` followed by a single-quoted `-delete`: a word starting with `$`, which no flag
5//! allowlist ever matches, while bash hands the command `-delete`.
6
7use super::budget;
8use super::{Word, WordPart};
9use winnow::ModalResult;
10use winnow::error::{ContextError, ErrMode};
11
12fn backtrack<T>() -> ModalResult<T> {
13    Err(ErrMode::Backtrack(ContextError::new()))
14}
15
16/// `$'…'`, kept RAW (the text between the quotes) so `--explain` echoes what was typed; the value
17/// is [`decode`]d on evaluation. A backslash escapes the next byte for scanning, so `$'a\'b'` is
18/// one quote holding `a\'b`. An unclosed quote consumes nothing.
19pub(super) fn ansi_c_quoted(input: &mut &str) -> ModalResult<WordPart> {
20    let Some(body) = input.strip_prefix("$'") else {
21        return backtrack();
22    };
23    let bytes = body.as_bytes();
24    let mut i = 0;
25    while i < bytes.len() && bytes[i] != b'\'' {
26        i += if bytes[i] == b'\\' { 2 } else { 1 };
27    }
28    if !budget::charge(0, i.min(bytes.len())) || i >= bytes.len() {
29        return backtrack();
30    }
31    let raw = body[..i].to_string();
32    *input = &body[i + 1..];
33    Ok(WordPart::AnsiC(raw))
34}
35
36/// `$"…"`: bash translates the string through the locale's message catalog, and with none
37/// installed (the case everywhere outside a localized script) the result is the plain
38/// double-quoted string, expansions included. So it parses as exactly that.
39pub(super) fn locale_quoted(input: &mut &str, double_quoted: fn(&mut &str) -> ModalResult<WordPart>) -> ModalResult<WordPart> {
40    if !input.starts_with("$\"") {
41        return backtrack();
42    }
43    let mut rest = &input[1..];
44    let part = double_quoted(&mut rest)?;
45    *input = rest;
46    Ok(part)
47}
48
49/// The value bash gives the body of `$'…'`, byte for byte (checked against bash 5.3):
50///
51/// - `\a \b \e \E \f \n \r \t \v \\ \' \" \?` are the usual characters;
52/// - `\NNN` is one to three octal digits, taken modulo 256;
53/// - `\xHH` is one or two hex digits, `\uHHHH` one to four, `\UHHHHHHHH` one to eight; with no
54///   digit the escape stays literal (`\x`);
55/// - `\cX` is the control character `X & 0x1f` (`\c?` is DEL, and `\c\\` reads one backslash);
56/// - any other `\X` stays as typed, backslash included;
57/// - a NUL from any escape ends the string there.
58///
59/// Bytes that are not valid UTF-8 become U+FFFD. Bash keeps the raw byte, but no such byte can
60/// spell anything the classifier matches, so the substitution changes no verdict.
61pub(crate) fn decode(raw: &str) -> String {
62    let b = raw.as_bytes();
63    let mut out: Vec<u8> = Vec::with_capacity(b.len());
64    let mut i = 0;
65    while i < b.len() {
66        if b[i] != b'\\' || i + 1 >= b.len() {
67            out.push(b[i]);
68            i += 1;
69            continue;
70        }
71        let c = b[i + 1];
72        i += 2;
73        match c {
74            b'a' => out.push(7),
75            b'b' => out.push(8),
76            b'e' | b'E' => out.push(27),
77            b'f' => out.push(12),
78            b'n' => out.push(b'\n'),
79            b'r' => out.push(b'\r'),
80            b't' => out.push(b'\t'),
81            b'v' => out.push(11),
82            b'\\' | b'\'' | b'"' | b'?' => out.push(c),
83            b'0'..=b'7' => {
84                let (v, used) = digits(&b[i..], 8, 2);
85                out.push(((u32::from(c - b'0') << (3 * used)) + v) as u8);
86                i += used;
87            }
88            b'x' | b'u' | b'U' => {
89                let max = match c {
90                    b'x' => 2,
91                    b'u' => 4,
92                    _ => 8,
93                };
94                let (v, used) = digits(&b[i..], 16, max);
95                i += used;
96                if used == 0 {
97                    out.extend_from_slice(&[b'\\', c]);
98                } else if c == b'x' {
99                    out.push(v as u8);
100                } else {
101                    let ch = char::from_u32(v).unwrap_or(char::REPLACEMENT_CHARACTER);
102                    out.extend_from_slice(ch.encode_utf8(&mut [0; 4]).as_bytes());
103                }
104            }
105            b'c' if i < b.len() => {
106                let x = b[i];
107                i += 1;
108                if x == b'\\' && b.get(i) == Some(&b'\\') {
109                    i += 1;
110                }
111                out.push(if x == b'?' { 0x7f } else { x.to_ascii_uppercase() & 0x1f });
112            }
113            _ => out.extend_from_slice(&[b'\\', c]),
114        }
115    }
116    if let Some(nul) = out.iter().position(|&x| x == 0) {
117        out.truncate(nul);
118    }
119    String::from_utf8_lossy(&out).into_owned()
120}
121
122/// Up to `max` leading digits of `radix`, and how many were read.
123fn digits(b: &[u8], radix: u32, max: usize) -> (u32, usize) {
124    let mut v = 0u32;
125    let mut used = 0;
126    while used < max {
127        let Some(d) = b.get(used).and_then(|&x| char::from(x).to_digit(radix)) else {
128            break;
129        };
130        v = v * radix + d;
131        used += 1;
132    }
133    (v, used)
134}
135
136impl Word {
137    /// Whether any part of the word is an ANSI-C quote, whose decoded value can differ from its
138    /// spelling in every way that matters (`$'\055delete'` is `-delete`).
139    pub fn has_ansi_c(&self) -> bool {
140        self.0.iter().any(|p| matches!(p, WordPart::AnsiC(_)))
141    }
142}
143
144#[cfg(test)]
145mod tests {
146    use super::decode;
147
148    #[test]
149    fn decodes_as_bash_does() {
150        let cases: &[(&str, &str)] = &[
151            ("-delete", "-delete"),
152            ("\\x2d\\x2Ddelete", "--delete"),
153            ("\\55x", "-x"),
154            ("\\0555", "-5"),
155            ("\\u2d", "-"),
156            ("\\u", "\\u"),
157            ("\\xZ", "\\xZ"),
158            ("\\x", "\\x"),
159            ("\\x41g", "Ag"),
160            ("\\101\\1012", "AA2"),
161            ("\\z", "\\z"),
162            ("\\q\\?\\\"", "\\q?\""),
163            ("a\\'b", "a'b"),
164            ("\\cA", "\u{1}"),
165            ("\\c?", "\u{7f}"),
166            ("\\c\\\\x", "\u{1c}x"),
167            ("\\c\\x", "\u{1c}x"),
168            ("\\c-", "\r"),
169            ("\\c", "\\c"),
170            ("\\e\\E", "\u{1b}\u{1b}"),
171            ("\\777", "\u{fffd}"),
172            ("\\u00410", "A0"),
173            ("\\U0001F600", "\u{1F600}"),
174            ("\\9", "\\9"),
175            ("\\t\\n", "\t\n"),
176        ];
177        for (raw, want) in cases {
178            assert_eq!(decode(raw), *want, "$'{raw}'");
179        }
180    }
181
182    #[test]
183    fn the_quote_renders_as_typed() {
184        for input in ["echo $'a\\'b'", "find / $'\\x2ddelete' x"] {
185            let script = crate::cst::parse(input).expect("parses");
186            assert_eq!(script.to_string().trim_end_matches(';'), input);
187            assert_eq!(crate::cst::parse(&script.to_string()), Some(script));
188        }
189    }
190
191    #[test]
192    fn a_nul_ends_the_string() {
193        for raw in ["a\\0b", "a\\x00b", "a\\u0000b", "a\\c@b", "a\\400b", "a\\x00"] {
194            assert_eq!(decode(raw), "a", "$'{raw}'");
195        }
196    }
197}