Skip to main content

fakecloud_core/
rfc2047.rs

1//! RFC 2047 encoded-words, the form S3 uses to carry non-ASCII user metadata
2//! (`x-amz-meta-*`) in HTTP headers.
3//!
4//! S3 decodes encoded-words in a metadata value before storing it and
5//! encodes a stored value that is not pure US-ASCII as `=?UTF-8?B?...?=` when
6//! returning it. A raw header byte outside US-ASCII is read as ISO-8859-1.
7
8use base64::engine::general_purpose::STANDARD as BASE64;
9use base64::Engine as _;
10
11/// Encode `value` for an HTTP header: unchanged when it is printable
12/// US-ASCII, otherwise a single `=?UTF-8?B?<base64>?=` encoded-word over its
13/// UTF-8 bytes. A control character (which a decoded encoded-word can carry)
14/// is encoded too, since a header value cannot hold it.
15pub fn encode(value: &str) -> String {
16    if value
17        .bytes()
18        .all(|b| b == b'\t' || (0x20..0x7f).contains(&b))
19    {
20        value.to_string()
21    } else {
22        format!("=?UTF-8?B?{}?=", BASE64.encode(value.as_bytes()))
23    }
24}
25
26/// Decode a raw header value. A value made only of encoded-words (separated
27/// by whitespace, which RFC 2047 says to drop between adjacent words) is
28/// decoded; anything else is read byte-for-byte as ISO-8859-1, which leaves
29/// a US-ASCII value unchanged.
30pub fn decode(raw: &[u8]) -> String {
31    let latin1 = || raw.iter().map(|&b| b as char).collect::<String>();
32    let Ok(text) = std::str::from_utf8(raw) else {
33        return latin1();
34    };
35    let mut words = text.split_ascii_whitespace().peekable();
36    if words.peek().is_none() {
37        return latin1();
38    }
39    let mut out = String::new();
40    for word in words {
41        match decode_word(word) {
42            Some(decoded) => out.push_str(&decoded),
43            None => return latin1(),
44        }
45    }
46    out
47}
48
49/// Decode one `=?charset?encoding?text?=` encoded-word. Supports the UTF-8,
50/// ISO-8859-1 and US-ASCII charsets and the `B` and `Q` encodings.
51fn decode_word(word: &str) -> Option<String> {
52    let inner = word.strip_prefix("=?")?.strip_suffix("?=")?;
53    let mut parts = inner.splitn(3, '?');
54    let charset = parts.next()?;
55    let encoding = parts.next()?;
56    let text = parts.next()?;
57    // RFC 2047 requires at least one encoded character.
58    if text.is_empty() {
59        return None;
60    }
61    // RFC 2231 allows a `*language` suffix on the charset.
62    let charset = charset.split('*').next()?.to_ascii_lowercase();
63    let bytes = match encoding {
64        "B" | "b" => BASE64.decode(text).ok()?,
65        "Q" | "q" => decode_q(text)?,
66        _ => return None,
67    };
68    match charset.as_str() {
69        "utf-8" | "utf8" => String::from_utf8(bytes).ok(),
70        "iso-8859-1" | "latin1" => Some(bytes.iter().map(|&b| b as char).collect()),
71        "us-ascii" => bytes
72            .is_ascii()
73            .then(|| bytes.iter().map(|&b| b as char).collect()),
74        _ => None,
75    }
76}
77
78/// The `Q` encoding: `_` is a space and `=XX` is a hex-escaped byte.
79fn decode_q(text: &str) -> Option<Vec<u8>> {
80    let bytes = text.as_bytes();
81    let mut out = Vec::with_capacity(bytes.len());
82    let mut i = 0;
83    while i < bytes.len() {
84        match bytes[i] {
85            b'_' => out.push(b' '),
86            b'=' => {
87                let hex = bytes.get(i + 1..i + 3)?;
88                if !hex.iter().all(u8::is_ascii_hexdigit) {
89                    return None;
90                }
91                out.push(u8::from_str_radix(std::str::from_utf8(hex).ok()?, 16).ok()?);
92                i += 2;
93            }
94            b => out.push(b),
95        }
96        i += 1;
97    }
98    Some(out)
99}
100
101#[cfg(test)]
102mod tests {
103    use super::*;
104
105    #[test]
106    fn ascii_round_trips_unchanged() {
107        assert_eq!(encode("AMAZONS3"), "AMAZONS3");
108        assert_eq!(decode(b"AMAZONS3"), "AMAZONS3");
109        assert_eq!(decode(b""), "");
110    }
111
112    #[test]
113    fn empty_encoded_word_is_kept_literally() {
114        assert_eq!(decode(b"=?UTF-8?B??="), "=?UTF-8?B??=");
115        assert_eq!(decode(b"=?UTF-8?Q??="), "=?UTF-8?Q??=");
116    }
117
118    #[test]
119    fn raw_non_ascii_header_bytes_match_the_s3_docs_example() {
120        // From the S3 user guide: `x-amz-meta-nonascii: ÄMÄZÕÑ S3` sent as
121        // raw UTF-8 bytes comes back as this encoded-word, because S3 reads
122        // the raw bytes as ISO-8859-1.
123        let stored = decode("ÄMÄZÕÑ S3".as_bytes());
124        assert_eq!(encode(&stored), "=?UTF-8?B?w4PChE3Dg8KEWsODwpXDg8KRIFMz?=");
125    }
126
127    #[test]
128    fn encoded_words_decode_before_storing() {
129        assert_eq!(decode(b"=?UTF-8?B?Y2Fmw6k=?="), "café");
130        assert_eq!(decode(b"=?utf-8?q?caf=C3=A9_au_lait?="), "café au lait");
131        assert_eq!(decode(b"=?ISO-8859-1?Q?caf=E9?="), "café");
132        assert_eq!(decode(b"=?UTF-8?B?Y2Fm?= =?UTF-8?B?w6k=?="), "café");
133        assert_eq!(encode("café"), "=?UTF-8?B?Y2Fmw6k=?=");
134    }
135
136    #[test]
137    fn control_characters_are_encoded_for_the_response() {
138        let stored = decode(b"=?UTF-8?B?YQpi?=");
139        assert_eq!(stored, "a\nb");
140        assert_eq!(encode(&stored), "=?UTF-8?B?YQpi?=");
141        assert_eq!(encode("a\tb c"), "a\tb c");
142    }
143
144    #[test]
145    fn malformed_or_mixed_words_stay_literal() {
146        assert_eq!(decode(b"=?UTF-8?B?***?="), "=?UTF-8?B?***?=");
147        assert_eq!(decode(b"=?UTF-8?X?abc?="), "=?UTF-8?X?abc?=");
148        assert_eq!(decode(b"=?KOI8-R?B?abc?="), "=?KOI8-R?B?abc?=");
149        assert_eq!(decode(b"=?UTF-8?Q?a=Z?="), "=?UTF-8?Q?a=Z?=");
150        assert_eq!(decode(b"=?UTF-8?Q?a=+F?="), "=?UTF-8?Q?a=+F?=");
151        assert_eq!(
152            decode(b"hello =?UTF-8?B?Y2Fmw6k=?="),
153            "hello =?UTF-8?B?Y2Fmw6k=?="
154        );
155    }
156}