Skip to main content

s3s_multipart/
content_disposition.rs

1// SPDX-License-Identifier: Apache-2.0
2// SPDX-FileCopyrightText: 2023-2026 The s3s Authors
3
4use crate::utils::trim_ows;
5
6/// A parsed `Content-Disposition` value.
7///
8/// The `name` and `file_name` fields borrow the original header value bytes
9/// and are not guaranteed to be UTF-8. Callers decide how to convert them.
10///
11/// Only the `form-data` disposition type is recognized; a value with any other
12/// type (for example `attachment`) is rejected by
13/// [`parse_content_disposition`].
14#[derive(Debug, Clone, Copy, PartialEq, Eq)]
15pub struct ContentDisposition<'a> {
16    /// The `name` parameter, if present.
17    pub name: Option<&'a [u8]>,
18    /// The `filename` parameter, if present.
19    pub file_name: Option<&'a [u8]>,
20}
21
22/// Parses a single `Content-Disposition` header value.
23///
24/// The parser is deliberately tolerant:
25///
26/// - the disposition type is matched case-insensitively and must be
27///   `form-data`;
28/// - parameter names are matched case-insensitively;
29/// - parameters may appear in any order;
30/// - unknown parameters (including `filename*`) are ignored;
31/// - parameter values may be quoted strings or bare tokens;
32/// - a missing `name` is represented as `None`.
33///
34/// Returned `name` and `file_name` values borrow the input and are returned
35/// verbatim: they are always sub-slices of `value`, so parsing never
36/// allocates. Quoted values keep their backslash escape sequences
37/// (`quoted-pair`), which RFC 2046 expects a parser to decode; decoding and
38/// UTF-8 conversion are the caller's responsibility. For example,
39/// `filename="a\"b.txt"` yields `a\"b.txt`, not `a"b.txt`.
40///
41/// # Best-effort semantics
42///
43/// Parsing stops at the first malformed parameter and the parameters parsed so
44/// far are returned. A malformed tail is therefore indistinguishable from a
45/// part without a `name`: both yield `name: None`. For example
46/// `form-data; name="unterminated` and `form-data; junk` both return
47/// `Some(ContentDisposition { name: None, file_name: None })`. Callers that
48/// must reject malformed values have to validate the raw header themselves; a
49/// `None` return only means that the value is not a `form-data`
50/// `Content-Disposition`.
51#[must_use]
52pub fn parse_content_disposition(value: &[u8]) -> Option<ContentDisposition<'_>> {
53    let input = trim_ows(value);
54    let semicolon = memchr::memchr(b';', input);
55
56    let disposition = match semicolon {
57        Some(idx) => trim_ows(&input[..idx]),
58        None => input,
59    };
60    if !disposition.eq_ignore_ascii_case(b"form-data") {
61        return None;
62    }
63
64    let Some(semicolon) = semicolon else {
65        return Some(ContentDisposition {
66            name: None,
67            file_name: None,
68        });
69    };
70
71    let mut name = None;
72    let mut file_name = None;
73    let mut rest = &input[semicolon.saturating_add(1)..];
74
75    while !trim_ows(rest).is_empty() {
76        let Some((key, parsed_value, next)) = parse_parameter(rest) else {
77            break;
78        };
79
80        if name.is_none() && key.eq_ignore_ascii_case(b"name") {
81            name = Some(parsed_value);
82        } else if file_name.is_none() && key.eq_ignore_ascii_case(b"filename") {
83            file_name = Some(parsed_value);
84        }
85
86        rest = next;
87    }
88
89    Some(ContentDisposition { name, file_name })
90}
91
92fn parse_parameter(input: &[u8]) -> Option<(&[u8], &[u8], &[u8])> {
93    let input = trim_ows(input);
94    let equals = memchr::memchr(b'=', input)?;
95    let key = trim_ows(&input[..equals]);
96    if key.is_empty() {
97        return None;
98    }
99
100    let value_input = trim_ows(&input[equals.saturating_add(1)..]);
101    let (value, next) = if value_input.first() == Some(&b'"') {
102        let value = parse_quoted(value_input)?;
103        let next = skip_until_semicolon(value.next);
104        (value.text, next)
105    } else {
106        let semicolon = memchr::memchr(b';', value_input);
107        match semicolon {
108            Some(idx) => (trim_ows(&value_input[..idx]), &value_input[idx.saturating_add(1)..]),
109            None => (trim_ows(value_input), &b""[..]),
110        }
111    };
112
113    Some((key, value, next))
114}
115
116fn parse_quoted(input: &[u8]) -> Option<ParsedQuoted<'_>> {
117    if input.first() != Some(&b'"') {
118        return None;
119    }
120
121    // Bulk-scan for the closing quote or the first backslash. A value that
122    // contains escapes falls back to the byte-wise scan below, so escape-heavy
123    // input is never slower than scanning every byte up front.
124    let idx = memchr::memchr2(b'"', b'\\', &input[1..])?;
125    let pos = idx.saturating_add(1);
126    if input[pos] == b'"' {
127        return Some(ParsedQuoted {
128            text: &input[1..pos],
129            next: &input[pos.saturating_add(1)..],
130        });
131    }
132
133    let mut escaped = false;
134    let mut end = None;
135    for (offset, byte) in input[pos..].iter().enumerate() {
136        if escaped {
137            escaped = false;
138        } else if *byte == b'\\' {
139            escaped = true;
140        } else if *byte == b'"' {
141            end = Some(pos.saturating_add(offset));
142            break;
143        }
144    }
145
146    let end = end?;
147    Some(ParsedQuoted {
148        text: &input[1..end],
149        next: &input[end.saturating_add(1)..],
150    })
151}
152
153fn skip_until_semicolon(input: &[u8]) -> &[u8] {
154    match memchr::memchr(b';', input) {
155        Some(idx) => &input[idx.saturating_add(1)..],
156        None => &b""[..],
157    }
158}
159
160struct ParsedQuoted<'a> {
161    text: &'a [u8],
162    next: &'a [u8],
163}
164
165#[cfg(test)]
166#[allow(clippy::expect_used, clippy::panic, clippy::unreachable, clippy::unwrap_used)]
167mod tests {
168    use super::*;
169
170    #[allow(clippy::type_complexity)]
171    fn parse(value: &[u8]) -> Option<(Option<&[u8]>, Option<&[u8]>)> {
172        let cd = parse_content_disposition(value)?;
173        Some((cd.name, cd.file_name))
174    }
175
176    #[allow(clippy::type_complexity, clippy::unnecessary_wraps)]
177    fn expected<'a>(name: Option<&'a [u8]>, file_name: Option<&'a [u8]>) -> Option<(Option<&'a [u8]>, Option<&'a [u8]>)> {
178        Some((name, file_name))
179    }
180
181    #[test]
182    fn parses_canonical_form() {
183        assert_eq!(
184            parse(b"form-data; name=\"file\"; filename=\"a.txt\""),
185            expected(Some(b"file"), Some(b"a.txt"))
186        );
187    }
188
189    #[test]
190    fn accepts_parameter_order_and_case() {
191        assert_eq!(
192            parse(b"FORM-DATA; FILENAME=\"a.txt\"; NAME=\"file\""),
193            expected(Some(b"file"), Some(b"a.txt"))
194        );
195    }
196
197    #[test]
198    fn accepts_token_values() {
199        assert_eq!(parse(b"form-data; name=file; filename=a.txt"), expected(Some(b"file"), Some(b"a.txt")));
200    }
201
202    #[test]
203    fn ignores_unknown_parameters() {
204        assert_eq!(
205            parse(b"form-data; size=10; name=\"file\"; filename*=UTF-8''a.txt; x=y"),
206            expected(Some(b"file"), None)
207        );
208    }
209
210    #[test]
211    fn handles_escaped_quote_without_early_termination() {
212        // Deliberate RFC 2046 difference: `quoted-pair` is not decoded, the
213        // value is returned verbatim (see the function documentation).
214        assert_eq!(
215            parse(b"form-data; name=\"fi\\\"le\"; filename=\"a.txt\""),
216            expected(Some(b"fi\\\"le"), Some(b"a.txt"))
217        );
218    }
219
220    #[test]
221    fn handles_missing_name() {
222        assert_eq!(parse(b"form-data; filename=\"a.txt\""), expected(None, Some(b"a.txt")));
223        assert_eq!(parse(b"form-data"), expected(None, None));
224    }
225
226    #[test]
227    fn rejects_non_form_data() {
228        assert_eq!(parse(b"attachment; filename=\"a.txt\""), None);
229        assert_eq!(parse(b""), None);
230    }
231
232    #[test]
233    fn tolerates_optional_whitespace() {
234        assert_eq!(
235            parse(b" form-data ; name = \"file\" ; filename = \"a.txt\" "),
236            expected(Some(b"file"), Some(b"a.txt"))
237        );
238        // OWS is trimmed around the quotes and kept inside them.
239        assert_eq!(parse(b"form-data; name=\" a \""), expected(Some(b" a "), None));
240        assert_eq!(parse(b"form-data; name=\"\ta\t\""), expected(Some(b"\ta\t"), None));
241    }
242
243    #[test]
244    fn quoted_value_is_returned_verbatim() {
245        assert_eq!(parse(b"form-data; name=\"a\\tb\""), expected(Some(b"a\\tb"), None));
246    }
247
248    #[test]
249    fn malformed_tail_parameter_stops_parsing() {
250        assert_eq!(parse(b"form-data; name=\"file\"; badparam"), expected(Some(b"file"), None));
251    }
252
253    #[test]
254    fn unterminated_quoted_value_stops_parsing() {
255        assert_eq!(parse(b"form-data; name=\"unterminated"), expected(None, None));
256        assert_eq!(parse(b"form-data; name=\"file\"; filename=\"a.txt"), expected(Some(b"file"), None));
257        // Trailing backslash: the escaped byte never arrives.
258        assert_eq!(parse(b"form-data; name=\"abc\\"), expected(None, None));
259    }
260
261    #[test]
262    fn first_duplicate_parameter_wins() {
263        assert_eq!(parse(b"form-data; name=\"one\"; name=\"two\""), expected(Some(b"one"), None));
264        assert_eq!(
265            parse(b"form-data; name=\"file\"; filename=\"a\"; filename=\"b\""),
266            expected(Some(b"file"), Some(b"a"))
267        );
268    }
269
270    #[test]
271    fn parameter_helpers_reject_invalid_shapes() {
272        assert_eq!(parse_parameter(b"=value"), None);
273        assert!(parse_quoted(b"not-quoted").is_none());
274    }
275}