Skip to main content

ruff_python_parser/
string.rs

1//! Parsing of string literals, bytes literals, and implicit string concatenation.
2
3use bstr::ByteSlice;
4use std::fmt;
5
6use ruff_python_ast::token::TokenKind;
7use ruff_python_ast::{self as ast, AnyStringFlags, AtomicNodeIndex, Expr, StringFlags};
8use ruff_text_size::{Ranged, TextRange, TextSize};
9
10use crate::error::{LexicalError, LexicalErrorType};
11
12#[derive(Debug)]
13pub(crate) enum StringType {
14    Str(ast::StringLiteral),
15    Bytes(ast::BytesLiteral),
16    FString(ast::FString),
17    TString(ast::TString),
18}
19
20impl Ranged for StringType {
21    fn range(&self) -> TextRange {
22        match self {
23            Self::Str(node) => node.range(),
24            Self::Bytes(node) => node.range(),
25            Self::FString(node) => node.range(),
26            Self::TString(node) => node.range(),
27        }
28    }
29}
30
31impl From<StringType> for Expr {
32    fn from(string: StringType) -> Self {
33        match string {
34            StringType::Str(node) => Expr::from(node),
35            StringType::Bytes(node) => Expr::from(node),
36            StringType::FString(node) => Expr::from(node),
37            StringType::TString(node) => Expr::from(node),
38        }
39    }
40}
41
42#[derive(Debug, Clone, Copy, PartialEq, Eq)]
43pub(crate) enum InterpolatedStringKind {
44    FString,
45    TString,
46}
47
48impl InterpolatedStringKind {
49    #[inline]
50    pub(crate) const fn start_token(self) -> TokenKind {
51        match self {
52            InterpolatedStringKind::FString => TokenKind::FStringStart,
53            InterpolatedStringKind::TString => TokenKind::TStringStart,
54        }
55    }
56
57    #[inline]
58    pub(crate) const fn middle_token(self) -> TokenKind {
59        match self {
60            InterpolatedStringKind::FString => TokenKind::FStringMiddle,
61            InterpolatedStringKind::TString => TokenKind::TStringMiddle,
62        }
63    }
64
65    #[inline]
66    pub(crate) const fn end_token(self) -> TokenKind {
67        match self {
68            InterpolatedStringKind::FString => TokenKind::FStringEnd,
69            InterpolatedStringKind::TString => TokenKind::TStringEnd,
70        }
71    }
72}
73
74impl fmt::Display for InterpolatedStringKind {
75    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
76        match self {
77            InterpolatedStringKind::FString => f.write_str("f-string"),
78            InterpolatedStringKind::TString => f.write_str("t-string"),
79        }
80    }
81}
82
83enum EscapedChar {
84    Literal(char),
85    Escape(char),
86}
87
88struct StringParser<'src> {
89    /// The raw content of the string e.g., the `foo` part in `"foo"`.
90    source: &'src str,
91    /// Current position of the parser in the source.
92    cursor: usize,
93    /// Flags that can be used to query information about the string.
94    flags: AnyStringFlags,
95    /// The location of the first character in the source from the start of the file.
96    offset: TextSize,
97    /// The range of the string literal.
98    range: TextRange,
99}
100
101impl<'src> StringParser<'src> {
102    fn new(source: &'src str, flags: AnyStringFlags, offset: TextSize, range: TextRange) -> Self {
103        Self {
104            source,
105            cursor: 0,
106            flags,
107            offset,
108            range,
109        }
110    }
111
112    #[inline]
113    fn skip_bytes(&mut self, bytes: usize) -> &str {
114        let skipped_str = &self.source[self.cursor..self.cursor + bytes];
115        self.cursor += bytes;
116        skipped_str
117    }
118
119    /// Returns the current position of the parser considering the offset.
120    #[inline]
121    fn position(&self) -> TextSize {
122        self.compute_position(self.cursor)
123    }
124
125    /// Computes the position of the cursor considering the offset.
126    #[inline]
127    fn compute_position(&self, cursor: usize) -> TextSize {
128        self.offset + TextSize::try_from(cursor).unwrap()
129    }
130
131    /// Returns the next byte in the string, if there is one.
132    ///
133    /// # Panics
134    ///
135    /// When the next byte is a part of a multi-byte character.
136    #[inline]
137    fn next_byte(&mut self) -> Option<u8> {
138        self.source.as_bytes()[self.cursor..].first().map(|&byte| {
139            self.cursor += 1;
140            byte
141        })
142    }
143
144    #[inline]
145    fn next_char(&mut self) -> Option<char> {
146        self.source[self.cursor..].chars().next().inspect(|c| {
147            self.cursor += c.len_utf8();
148        })
149    }
150
151    #[inline]
152    fn peek_byte(&self) -> Option<u8> {
153        self.source.as_bytes()[self.cursor..].first().copied()
154    }
155
156    fn parse_unicode_literal(&mut self, literal_number: usize) -> Result<char, LexicalError> {
157        let mut p: u32 = 0u32;
158        for i in 1..=literal_number {
159            let start = self.position();
160            match self.next_char() {
161                Some(c) => match c.to_digit(16) {
162                    Some(d) => p += d << ((literal_number - i) * 4),
163                    None => {
164                        return Err(LexicalError::new(
165                            LexicalErrorType::UnicodeError,
166                            TextRange::at(start, TextSize::try_from(c.len_utf8()).unwrap()),
167                        ));
168                    }
169                },
170                None => {
171                    return Err(LexicalError::new(
172                        LexicalErrorType::UnicodeError,
173                        TextRange::empty(self.position()),
174                    ));
175                }
176            }
177        }
178        match p {
179            0xD800..=0xDFFF => Ok(std::char::REPLACEMENT_CHARACTER),
180            _ => std::char::from_u32(p).ok_or(LexicalError::new(
181                LexicalErrorType::UnicodeError,
182                TextRange::empty(self.position()),
183            )),
184        }
185    }
186
187    fn parse_octet(&mut self, o: u8) -> char {
188        let mut radix_bytes = [o, 0, 0];
189        let mut len = 1;
190
191        while len < 3 {
192            let Some(b'0'..=b'7') = self.peek_byte() else {
193                break;
194            };
195
196            radix_bytes[len] = self.next_byte().unwrap();
197            len += 1;
198        }
199
200        // OK because radix_bytes is always going to be in the ASCII range.
201        let radix_str = std::str::from_utf8(&radix_bytes[..len]).expect("ASCII bytes");
202        let value = u32::from_str_radix(radix_str, 8).unwrap();
203        char::from_u32(value).unwrap()
204    }
205
206    fn parse_unicode_name(&mut self) -> Result<char, LexicalError> {
207        let start_pos = self.position();
208        let Some('{') = self.next_char() else {
209            return Err(LexicalError::new(
210                LexicalErrorType::MissingUnicodeLbrace,
211                TextRange::empty(start_pos),
212            ));
213        };
214
215        let start_pos = self.position();
216        let Some(close_idx) = self.source[self.cursor..].find('}') else {
217            return Err(LexicalError::new(
218                LexicalErrorType::MissingUnicodeRbrace,
219                TextRange::empty(self.compute_position(self.source.len())),
220            ));
221        };
222
223        let name_and_ending = self.skip_bytes(close_idx + 1);
224        let name = &name_and_ending[..name_and_ending.len() - 1];
225
226        unicode_names2::character(name).ok_or_else(|| {
227            LexicalError::new(
228                LexicalErrorType::UnicodeError,
229                // The cursor is right after the `}` character, so we subtract 1 to get the correct
230                // range of the unicode name.
231                TextRange::new(
232                    start_pos,
233                    self.compute_position(self.cursor - '}'.len_utf8()),
234                ),
235            )
236        })
237    }
238
239    /// Parse an escaped character, returning the new character.
240    fn parse_escaped_char(&mut self) -> Result<Option<EscapedChar>, LexicalError> {
241        let Some(first_char) = self.next_char() else {
242            // TODO: check when this error case happens
243            return Err(LexicalError::new(
244                LexicalErrorType::StringError,
245                TextRange::empty(self.position()),
246            ));
247        };
248
249        let new_char = match first_char {
250            '\\' => '\\',
251            '\'' => '\'',
252            '\"' => '"',
253            'a' => '\x07',
254            'b' => '\x08',
255            'f' => '\x0c',
256            'n' => '\n',
257            'r' => '\r',
258            't' => '\t',
259            'v' => '\x0b',
260            o @ '0'..='7' => self.parse_octet(o as u8),
261            'x' => self.parse_unicode_literal(2)?,
262            'u' if !self.flags.is_byte_string() => self.parse_unicode_literal(4)?,
263            'U' if !self.flags.is_byte_string() => self.parse_unicode_literal(8)?,
264            'N' if !self.flags.is_byte_string() => self.parse_unicode_name()?,
265            // Special cases where the escape sequence is not a single character
266            '\n' => return Ok(None),
267            '\r' => {
268                if self.peek_byte() == Some(b'\n') {
269                    self.next_byte();
270                }
271
272                return Ok(None);
273            }
274            _ => return Ok(Some(EscapedChar::Escape(first_char))),
275        };
276
277        Ok(Some(EscapedChar::Literal(new_char)))
278    }
279
280    fn parse_interpolated_string_middle(
281        mut self,
282    ) -> Result<ast::InterpolatedStringLiteralElement, LexicalError> {
283        // Fast-path: if the f-string or t-string doesn't contain any escape sequences, return the literal.
284        let Some(mut index) = memchr::memchr3(b'{', b'}', b'\\', self.source.as_bytes()) else {
285            return Ok(ast::InterpolatedStringLiteralElement {
286                value: self.source.into(),
287                range: self.range,
288                node_index: AtomicNodeIndex::NONE,
289            });
290        };
291
292        let mut value = String::with_capacity(self.source.len());
293        loop {
294            // Add the characters before the escape sequence (or curly brace) to the string.
295            let before_with_slash_or_brace = self.skip_bytes(index + 1);
296            let before = &before_with_slash_or_brace[..before_with_slash_or_brace.len() - 1];
297            value.push_str(before);
298
299            // Add the escaped character to the string.
300            match self.source.as_bytes()[self.cursor - 1] {
301                // If there are any curly braces inside a `F/TStringMiddle` token,
302                // then they were escaped (i.e. `{{` or `}}`). This means that
303                // the raw source contains a doubled brace, but the literal value only
304                // contains one brace.
305                brace @ (b'{' | b'}') => {
306                    if self.peek_byte() == Some(brace) {
307                        self.next_byte();
308                    }
309                    value.push(char::from(brace));
310                }
311                // We can encounter a `\` as the last character in a `F/TStringMiddle`
312                // token which is valid in this context. For example,
313                //
314                // ```python
315                // f"\{foo} \{bar:\}"
316                // # ^     ^^     ^
317                // ```
318                //
319                // Here, the `F/TStringMiddle` token content will be "\" and " \"
320                // which is invalid if we look at the content in isolation:
321                //
322                // ```python
323                // "\"
324                // ```
325                //
326                // However, the content is syntactically valid in the context of
327                // the f/t-string because it's a substring of the entire f/t-string.
328                // This is still an invalid escape sequence, but we don't want to
329                // raise a syntax error as is done by the CPython parser. It might
330                // be supported in the future, refer to point 3: https://peps.python.org/pep-0701/#rejected-ideas
331                b'\\' => {
332                    if !self.flags.is_raw_string() && self.peek_byte().is_some() {
333                        if let Some(brace @ (b'{' | b'}')) = self.peek_byte()
334                            && self.source.as_bytes().get(self.cursor + 1).copied() == Some(brace)
335                        {
336                            // Leave the doubled brace for the next iteration to collapse.
337                            value.push('\\');
338                        } else {
339                            match self.parse_escaped_char()? {
340                                None => {}
341                                Some(EscapedChar::Literal(c)) => value.push(c),
342                                Some(EscapedChar::Escape(c)) => {
343                                    value.push('\\');
344                                    value.push(c);
345                                }
346                            }
347                        }
348                    } else {
349                        value.push('\\');
350                    }
351                }
352                ch => {
353                    unreachable!("Expected '{{', '}}', or '\\' but got {:?}", ch);
354                }
355            }
356
357            let Some(next_index) =
358                memchr::memchr3(b'{', b'}', b'\\', &self.source.as_bytes()[self.cursor..])
359            else {
360                // Add the rest of the string to the value.
361                let rest = &self.source[self.cursor..];
362                value.push_str(rest);
363                break;
364            };
365
366            index = next_index;
367        }
368
369        Ok(ast::InterpolatedStringLiteralElement {
370            value: value.into_boxed_str(),
371            range: self.range,
372            node_index: AtomicNodeIndex::NONE,
373        })
374    }
375
376    fn parse_bytes(mut self) -> Result<StringType, LexicalError> {
377        if let Some(index) = self.source.as_bytes().find_non_ascii_byte() {
378            let ch = self.source.chars().nth(index).unwrap();
379            return Err(LexicalError::new(
380                LexicalErrorType::InvalidByteLiteral,
381                TextRange::at(
382                    self.compute_position(index),
383                    TextSize::try_from(ch.len_utf8()).unwrap(),
384                ),
385            ));
386        }
387
388        if self.flags.is_raw_string() {
389            // For raw strings, no escaping is necessary.
390            return Ok(StringType::Bytes(ast::BytesLiteral {
391                value: self.source.as_bytes().into(),
392                range: self.range,
393                flags: self.flags.into(),
394                node_index: AtomicNodeIndex::NONE,
395            }));
396        }
397
398        let Some(mut escape) = memchr::memchr(b'\\', self.source.as_bytes()) else {
399            // If the string doesn't contain any escape sequences, return the owned string.
400            return Ok(StringType::Bytes(ast::BytesLiteral {
401                value: self.source.as_bytes().into(),
402                range: self.range,
403                flags: self.flags.into(),
404                node_index: AtomicNodeIndex::NONE,
405            }));
406        };
407
408        // If the string contains escape sequences, we need to parse them.
409        let mut value = Vec::with_capacity(self.source.len());
410        loop {
411            // Add the characters before the escape sequence to the string.
412            let before_with_slash = self.skip_bytes(escape + 1);
413            let before = &before_with_slash[..before_with_slash.len() - 1];
414            value.extend_from_slice(before.as_bytes());
415
416            // Add the escaped character to the string.
417            match self.parse_escaped_char()? {
418                None => {}
419                Some(EscapedChar::Literal(c)) => value.push(c as u8),
420                Some(EscapedChar::Escape(c)) => {
421                    value.push(b'\\');
422                    value.push(c as u8);
423                }
424            }
425
426            let Some(next_escape) = memchr::memchr(b'\\', &self.source.as_bytes()[self.cursor..])
427            else {
428                // Add the rest of the string to the value.
429                let rest = &self.source[self.cursor..];
430                value.extend_from_slice(rest.as_bytes());
431                break;
432            };
433
434            // Update the position of the next escape sequence.
435            escape = next_escape;
436        }
437
438        Ok(StringType::Bytes(ast::BytesLiteral {
439            value: value.into_boxed_slice(),
440            range: self.range,
441            flags: self.flags.into(),
442            node_index: AtomicNodeIndex::NONE,
443        }))
444    }
445
446    fn parse_string(mut self) -> Result<StringType, LexicalError> {
447        if self.flags.is_raw_string() {
448            // For raw strings, no escaping is necessary.
449            return Ok(StringType::Str(ast::StringLiteral {
450                value: self.source.into(),
451                range: self.range,
452                flags: self.flags.into(),
453                node_index: AtomicNodeIndex::NONE,
454            }));
455        }
456
457        let Some(mut escape) = memchr::memchr(b'\\', self.source.as_bytes()) else {
458            // If the string doesn't contain any escape sequences, return the owned string.
459            return Ok(StringType::Str(ast::StringLiteral {
460                value: self.source.into(),
461                range: self.range,
462                flags: self.flags.into(),
463                node_index: AtomicNodeIndex::NONE,
464            }));
465        };
466
467        // If the string contains escape sequences, we need to parse them.
468        let mut value = String::with_capacity(self.source.len());
469
470        loop {
471            // Add the characters before the escape sequence to the string.
472            let before_with_slash = self.skip_bytes(escape + 1);
473            let before = &before_with_slash[..before_with_slash.len() - 1];
474            value.push_str(before);
475
476            // Add the escaped character to the string.
477            match self.parse_escaped_char()? {
478                None => {}
479                Some(EscapedChar::Literal(c)) => value.push(c),
480                Some(EscapedChar::Escape(c)) => {
481                    value.push('\\');
482                    value.push(c);
483                }
484            }
485
486            let Some(next_escape) = self.source[self.cursor..].find('\\') else {
487                // Add the rest of the string to the value.
488                let rest = &self.source[self.cursor..];
489                value.push_str(rest);
490                break;
491            };
492
493            // Update the position of the next escape sequence.
494            escape = next_escape;
495        }
496
497        Ok(StringType::Str(ast::StringLiteral {
498            value: value.into_boxed_str(),
499            range: self.range,
500            flags: self.flags.into(),
501            node_index: AtomicNodeIndex::NONE,
502        }))
503    }
504
505    fn parse(self) -> Result<StringType, LexicalError> {
506        if self.flags.is_byte_string() {
507            self.parse_bytes()
508        } else {
509            self.parse_string()
510        }
511    }
512}
513
514pub(crate) fn parse_string_literal(
515    source: &str,
516    flags: AnyStringFlags,
517    range: TextRange,
518) -> Result<StringType, LexicalError> {
519    StringParser::new(source, flags, range.start() + flags.opener_len(), range).parse()
520}
521
522pub(crate) fn parse_interpolated_string_literal_element(
523    source: &str,
524    flags: AnyStringFlags,
525    range: TextRange,
526) -> Result<ast::InterpolatedStringLiteralElement, LexicalError> {
527    StringParser::new(source, flags, range.start(), range).parse_interpolated_string_middle()
528}
529
530#[cfg(test)]
531mod tests {
532    use ruff_python_ast::Suite;
533
534    use crate::error::LexicalErrorType;
535    use crate::{InterpolatedStringErrorType, ParseError, ParseErrorType, Parsed, parse_module};
536
537    const WINDOWS_EOL: &str = "\r\n";
538    const MAC_EOL: &str = "\r";
539    const UNIX_EOL: &str = "\n";
540
541    fn parse_suite(source: &str) -> Result<Suite, ParseError> {
542        parse_module(source).map(Parsed::into_suite)
543    }
544
545    fn nested_format_spec(prefix: char, depth: usize) -> String {
546        let mut replacement_field = String::from("{spec}");
547        for _ in 0..depth {
548            replacement_field = format!("{{foo:{replacement_field}}}");
549        }
550        format!(r#"{prefix}"{replacement_field}""#)
551    }
552
553    fn string_parser_escaped_eol(eol: &str) -> Suite {
554        let source = format!(r"'text \{eol}more text'");
555        parse_suite(&source).unwrap()
556    }
557
558    #[test]
559    fn test_string_parser_escaped_unix_eol() {
560        let suite = string_parser_escaped_eol(UNIX_EOL);
561        insta::assert_debug_snapshot!(suite);
562    }
563
564    #[test]
565    fn test_string_parser_escaped_mac_eol() {
566        let suite = string_parser_escaped_eol(MAC_EOL);
567        insta::assert_debug_snapshot!(suite);
568    }
569
570    #[test]
571    fn test_string_parser_escaped_windows_eol() {
572        let suite = string_parser_escaped_eol(WINDOWS_EOL);
573        insta::assert_debug_snapshot!(suite);
574    }
575
576    #[test]
577    fn test_parse_fstring() {
578        let source = r#"f"{a}{ b }{{foo}}""#;
579        let suite = parse_suite(source).unwrap();
580        insta::assert_debug_snapshot!(suite);
581    }
582
583    #[test]
584    fn test_parse_fstring_nested_spec() {
585        let source = r#"f"{foo:{spec}}""#;
586        let suite = parse_suite(source).unwrap();
587        insta::assert_debug_snapshot!(suite);
588    }
589
590    #[test]
591    fn parse_fstring_nested_spec_grows_stack() {
592        assert!(parse_suite(&nested_format_spec('f', 200)).is_ok());
593    }
594
595    #[test]
596    fn test_parse_fstring_not_nested_spec() {
597        let source = r#"f"{foo:spec}""#;
598        let suite = parse_suite(source).unwrap();
599        insta::assert_debug_snapshot!(suite);
600    }
601
602    #[test]
603    fn test_parse_empty_fstring() {
604        let source = r#"f"""#;
605        let suite = parse_suite(source).unwrap();
606        insta::assert_debug_snapshot!(suite);
607    }
608
609    #[test]
610    fn test_fstring_parse_self_documenting_base() {
611        let source = r#"f"{user=}""#;
612        let suite = parse_suite(source).unwrap();
613        insta::assert_debug_snapshot!(suite);
614    }
615
616    #[test]
617    fn test_fstring_parse_self_documenting_base_more() {
618        let source = r#"f"mix {user=} with text and {second=}""#;
619        let suite = parse_suite(source).unwrap();
620        insta::assert_debug_snapshot!(suite);
621    }
622
623    #[test]
624    fn test_fstring_parse_self_documenting_format() {
625        let source = r#"f"{user=:>10}""#;
626        let suite = parse_suite(source).unwrap();
627        insta::assert_debug_snapshot!(suite);
628    }
629
630    fn parse_fstring_error(source: &str) -> InterpolatedStringErrorType {
631        parse_suite(source)
632            .map_err(|e| match e.error {
633                ParseErrorType::Lexical(LexicalErrorType::FStringError(e)) => e,
634                ParseErrorType::FStringError(e) => e,
635                e => unreachable!("Expected FStringError: {:?}", e),
636            })
637            .expect_err("Expected error")
638    }
639
640    #[test]
641    fn test_parse_invalid_fstring() {
642        use InterpolatedStringErrorType::{InvalidConversionFlag, LambdaWithoutParentheses};
643
644        assert_eq!(parse_fstring_error(r#"f"{5!x}""#), InvalidConversionFlag);
645        assert_eq!(
646            parse_fstring_error("f'{lambda x:{x}}'"),
647            LambdaWithoutParentheses
648        );
649        // NOTE: The parser produces the `LambdaWithoutParentheses` for this case, but
650        // since the parser only return the first error to maintain compatibility with
651        // the rest of the codebase, this test case fails. The `LambdaWithoutParentheses`
652        // error appears after the unexpected `FStringMiddle` token, which is between the
653        // `:` and the `{`.
654        // assert_eq!(parse_fstring_error("f'{lambda x: {x}}'"), LambdaWithoutParentheses);
655        assert!(parse_suite(r#"f"{class}""#).is_err());
656    }
657
658    #[test]
659    fn test_parse_fstring_not_equals() {
660        let source = r#"f"{1 != 2}""#;
661        let suite = parse_suite(source).unwrap();
662        insta::assert_debug_snapshot!(suite);
663    }
664
665    #[test]
666    fn test_parse_fstring_equals() {
667        let source = r#"f"{42 == 42}""#;
668        let suite = parse_suite(source).unwrap();
669        insta::assert_debug_snapshot!(suite);
670    }
671
672    #[test]
673    fn test_parse_fstring_self_doc_prec_space() {
674        let source = r#"f"{x   =}""#;
675        let suite = parse_suite(source).unwrap();
676        insta::assert_debug_snapshot!(suite);
677    }
678
679    #[test]
680    fn test_parse_fstring_self_doc_trailing_space() {
681        let source = r#"f"{x=   }""#;
682        let suite = parse_suite(source).unwrap();
683        insta::assert_debug_snapshot!(suite);
684    }
685
686    #[test]
687    fn test_parse_fstring_yield_expr() {
688        let source = r#"f"{yield}""#;
689        let suite = parse_suite(source).unwrap();
690        insta::assert_debug_snapshot!(suite);
691    }
692
693    #[test]
694    fn test_parse_tstring() {
695        let source = r#"t"{a}{ b }{{foo}}""#;
696        let suite = parse_suite(source).unwrap();
697        insta::assert_debug_snapshot!(suite);
698    }
699
700    #[test]
701    fn test_parse_tstring_nested_spec() {
702        let source = r#"t"{foo:{spec}}""#;
703        let suite = parse_suite(source).unwrap();
704        insta::assert_debug_snapshot!(suite);
705    }
706
707    #[test]
708    fn parse_tstring_nested_spec_grows_stack() {
709        assert!(parse_suite(&nested_format_spec('t', 200)).is_ok());
710    }
711
712    #[test]
713    fn test_parse_tstring_not_nested_spec() {
714        let source = r#"t"{foo:spec}""#;
715        let suite = parse_suite(source).unwrap();
716        insta::assert_debug_snapshot!(suite);
717    }
718
719    #[test]
720    fn test_parse_empty_tstring() {
721        let source = r#"t"""#;
722        let suite = parse_suite(source).unwrap();
723        insta::assert_debug_snapshot!(suite);
724    }
725
726    #[test]
727    fn test_tstring_parse_self_documenting_base() {
728        let source = r#"t"{user=}""#;
729        let suite = parse_suite(source).unwrap();
730        insta::assert_debug_snapshot!(suite);
731    }
732
733    #[test]
734    fn test_tstring_parse_self_documenting_base_more() {
735        let source = r#"t"mix {user=} with text and {second=}""#;
736        let suite = parse_suite(source).unwrap();
737        insta::assert_debug_snapshot!(suite);
738    }
739
740    #[test]
741    fn test_tstring_parse_self_documenting_format() {
742        let source = r#"t"{user=:>10}""#;
743        let suite = parse_suite(source).unwrap();
744        insta::assert_debug_snapshot!(suite);
745    }
746
747    fn parse_tstring_error(source: &str) -> InterpolatedStringErrorType {
748        parse_suite(source)
749            .map_err(|e| match e.error {
750                ParseErrorType::Lexical(LexicalErrorType::TStringError(e)) => e,
751                ParseErrorType::TStringError(e) => e,
752                e => unreachable!("Expected TStringError: {:?}", e),
753            })
754            .expect_err("Expected error")
755    }
756
757    #[test]
758    fn test_parse_invalid_tstring() {
759        use InterpolatedStringErrorType::{InvalidConversionFlag, LambdaWithoutParentheses};
760
761        assert_eq!(parse_tstring_error(r#"t"{5!x}""#), InvalidConversionFlag);
762        assert_eq!(
763            parse_tstring_error("t'{lambda x:{x}}'"),
764            LambdaWithoutParentheses
765        );
766        // NOTE: The parser produces the `LambdaWithoutParentheses` for this case, but
767        // since the parser only return the first error to maintain compatibility with
768        // the rest of the codebase, this test case fails. The `LambdaWithoutParentheses`
769        // error appears after the unexpected `tStringMiddle` token, which is between the
770        // `:` and the `{`.
771        // assert_eq!(parse_tstring_error("f'{lambda x: {x}}'"), LambdaWithoutParentheses);
772        assert!(parse_suite(r#"t"{class}""#).is_err());
773    }
774
775    #[test]
776    fn test_parse_tstring_not_equals() {
777        let source = r#"t"{1 != 2}""#;
778        let suite = parse_suite(source).unwrap();
779        insta::assert_debug_snapshot!(suite);
780    }
781
782    #[test]
783    fn test_parse_tstring_equals() {
784        let source = r#"t"{42 == 42}""#;
785        let suite = parse_suite(source).unwrap();
786        insta::assert_debug_snapshot!(suite);
787    }
788
789    #[test]
790    fn test_parse_tstring_self_doc_prec_space() {
791        let source = r#"t"{x   =}""#;
792        let suite = parse_suite(source).unwrap();
793        insta::assert_debug_snapshot!(suite);
794    }
795
796    #[test]
797    fn test_parse_tstring_self_doc_trailing_space() {
798        let source = r#"t"{x=   }""#;
799        let suite = parse_suite(source).unwrap();
800        insta::assert_debug_snapshot!(suite);
801    }
802
803    #[test]
804    fn test_parse_tstring_yield_expr() {
805        let source = r#"t"{yield}""#;
806        let suite = parse_suite(source).unwrap();
807        insta::assert_debug_snapshot!(suite);
808    }
809
810    #[test]
811    fn test_parse_string_concat() {
812        let source = "'Hello ' 'world'";
813        let suite = parse_suite(source).unwrap();
814        insta::assert_debug_snapshot!(suite);
815    }
816
817    #[test]
818    fn test_parse_u_string_concat_1() {
819        let source = "'Hello ' u'world'";
820        let suite = parse_suite(source).unwrap();
821        insta::assert_debug_snapshot!(suite);
822    }
823
824    #[test]
825    fn test_parse_u_string_concat_2() {
826        let source = "u'Hello ' 'world'";
827        let suite = parse_suite(source).unwrap();
828        insta::assert_debug_snapshot!(suite);
829    }
830
831    #[test]
832    fn test_parse_f_string_concat_1() {
833        let source = "'Hello ' f'world'";
834        let suite = parse_suite(source).unwrap();
835        insta::assert_debug_snapshot!(suite);
836    }
837
838    #[test]
839    fn test_parse_f_string_concat_2() {
840        let source = "'Hello ' f'world'";
841        let suite = parse_suite(source).unwrap();
842        insta::assert_debug_snapshot!(suite);
843    }
844
845    #[test]
846    fn test_parse_f_string_concat_3() {
847        let source = "'Hello ' f'world{\"!\"}'";
848        let suite = parse_suite(source).unwrap();
849        insta::assert_debug_snapshot!(suite);
850    }
851
852    #[test]
853    fn test_parse_f_string_concat_4() {
854        let source = "'Hello ' f'world{\"!\"}' 'again!'";
855        let suite = parse_suite(source).unwrap();
856        insta::assert_debug_snapshot!(suite);
857    }
858
859    #[test]
860    fn test_parse_u_f_string_concat_1() {
861        let source = "u'Hello ' f'world'";
862        let suite = parse_suite(source).unwrap();
863        insta::assert_debug_snapshot!(suite);
864    }
865
866    #[test]
867    fn test_parse_u_f_string_concat_2() {
868        let source = "u'Hello ' f'world' '!'";
869        let suite = parse_suite(source).unwrap();
870        insta::assert_debug_snapshot!(suite);
871    }
872
873    #[test]
874    fn test_parse_t_string_concat_1_error() {
875        let source = "'Hello ' t'world'";
876        let suite = parse_suite(source).unwrap_err();
877        insta::assert_debug_snapshot!(suite);
878    }
879
880    #[test]
881    fn test_parse_t_string_concat_2_error() {
882        let source = "'Hello ' t'world'";
883        let suite = parse_suite(source).unwrap_err();
884        insta::assert_debug_snapshot!(suite);
885    }
886
887    #[test]
888    fn test_parse_t_string_concat_3_error() {
889        let source = "'Hello ' t'world{\"!\"}'";
890        let suite = parse_suite(source).unwrap_err();
891        insta::assert_debug_snapshot!(suite);
892    }
893
894    #[test]
895    fn test_parse_t_string_concat_4_error() {
896        let source = "'Hello ' t'world{\"!\"}' 'again!'";
897        let suite = parse_suite(source).unwrap_err();
898        insta::assert_debug_snapshot!(suite);
899    }
900
901    #[test]
902    fn test_parse_u_t_string_concat_1_error() {
903        let source = "u'Hello ' t'world'";
904        let suite = parse_suite(source).unwrap_err();
905        insta::assert_debug_snapshot!(suite);
906    }
907
908    #[test]
909    fn test_parse_u_t_string_concat_2_error() {
910        let source = "u'Hello ' t'world' '!'";
911        let suite = parse_suite(source).unwrap_err();
912        insta::assert_debug_snapshot!(suite);
913    }
914
915    #[test]
916    fn test_parse_f_t_string_concat_1_error() {
917        let source = "f'Hello ' t'world'";
918        let suite = parse_suite(source).unwrap_err();
919        insta::assert_debug_snapshot!(suite);
920    }
921
922    #[test]
923    fn test_parse_f_t_string_concat_2_error() {
924        let source = "f'Hello ' t'world' '!'";
925        let suite = parse_suite(source).unwrap_err();
926        insta::assert_debug_snapshot!(suite);
927    }
928
929    #[test]
930    fn test_parse_string_triple_quotes_with_kind() {
931        let source = "u'''Hello, world!'''";
932        let suite = parse_suite(source).unwrap();
933        insta::assert_debug_snapshot!(suite);
934    }
935
936    #[test]
937    fn test_single_quoted_byte() {
938        // single quote
939        let source = r##"b'\x00\x01\x02\x03\x04\x05\x06\x07\x08\t\n\x0b\x0c\r\x0e\x0f\x10\x11\x12\x13\x14\x15\x16\x17\x18\x19\x1a\x1b\x1c\x1d\x1e\x1f !"#$%&\'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\\]^_`abcdefghijklmnopqrstuvwxyz{|}~\x7f\x80\x81\x82\x83\x84\x85\x86\x87\x88\x89\x8a\x8b\x8c\x8d\x8e\x8f\x90\x91\x92\x93\x94\x95\x96\x97\x98\x99\x9a\x9b\x9c\x9d\x9e\x9f\xa0\xa1\xa2\xa3\xa4\xa5\xa6\xa7\xa8\xa9\xaa\xab\xac\xad\xae\xaf\xb0\xb1\xb2\xb3\xb4\xb5\xb6\xb7\xb8\xb9\xba\xbb\xbc\xbd\xbe\xbf\xc0\xc1\xc2\xc3\xc4\xc5\xc6\xc7\xc8\xc9\xca\xcb\xcc\xcd\xce\xcf\xd0\xd1\xd2\xd3\xd4\xd5\xd6\xd7\xd8\xd9\xda\xdb\xdc\xdd\xde\xdf\xe0\xe1\xe2\xe3\xe4\xe5\xe6\xe7\xe8\xe9\xea\xeb\xec\xed\xee\xef\xf0\xf1\xf2\xf3\xf4\xf5\xf6\xf7\xf8\xf9\xfa\xfb\xfc\xfd\xfe\xff'"##;
940        let suite = parse_suite(source).unwrap();
941        insta::assert_debug_snapshot!(suite);
942    }
943
944    #[test]
945    fn test_double_quoted_byte() {
946        // double quote
947        let source = r##"b"\x00\x01\x02\x03\x04\x05\x06\x07\x08\t\n\x0b\x0c\r\x0e\x0f\x10\x11\x12\x13\x14\x15\x16\x17\x18\x19\x1a\x1b\x1c\x1d\x1e\x1f !\"#$%&'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\\]^_`abcdefghijklmnopqrstuvwxyz{|}~\x7f\x80\x81\x82\x83\x84\x85\x86\x87\x88\x89\x8a\x8b\x8c\x8d\x8e\x8f\x90\x91\x92\x93\x94\x95\x96\x97\x98\x99\x9a\x9b\x9c\x9d\x9e\x9f\xa0\xa1\xa2\xa3\xa4\xa5\xa6\xa7\xa8\xa9\xaa\xab\xac\xad\xae\xaf\xb0\xb1\xb2\xb3\xb4\xb5\xb6\xb7\xb8\xb9\xba\xbb\xbc\xbd\xbe\xbf\xc0\xc1\xc2\xc3\xc4\xc5\xc6\xc7\xc8\xc9\xca\xcb\xcc\xcd\xce\xcf\xd0\xd1\xd2\xd3\xd4\xd5\xd6\xd7\xd8\xd9\xda\xdb\xdc\xdd\xde\xdf\xe0\xe1\xe2\xe3\xe4\xe5\xe6\xe7\xe8\xe9\xea\xeb\xec\xed\xee\xef\xf0\xf1\xf2\xf3\xf4\xf5\xf6\xf7\xf8\xf9\xfa\xfb\xfc\xfd\xfe\xff""##;
948        let suite = parse_suite(source).unwrap();
949        insta::assert_debug_snapshot!(suite);
950    }
951
952    #[test]
953    fn test_escape_char_in_byte_literal() {
954        // backslash does not escape
955        let source = r#"b"omkmok\Xaa""#; // spell-checker:ignore omkmok
956        let suite = parse_suite(source).unwrap();
957        insta::assert_debug_snapshot!(suite);
958    }
959
960    #[test]
961    fn test_raw_byte_literal_1() {
962        let source = r"rb'\x1z'";
963        let suite = parse_suite(source).unwrap();
964        insta::assert_debug_snapshot!(suite);
965    }
966
967    #[test]
968    fn test_raw_byte_literal_2() {
969        let source = r"rb'\\'";
970        let suite = parse_suite(source).unwrap();
971        insta::assert_debug_snapshot!(suite);
972    }
973
974    #[test]
975    fn test_escape_octet() {
976        let source = r"b'\43a\4\1234'";
977        let suite = parse_suite(source).unwrap();
978        insta::assert_debug_snapshot!(suite);
979    }
980
981    #[test]
982    fn test_fstring_escaped_newline() {
983        let source = r#"f"\n{x}""#;
984        let suite = parse_suite(source).unwrap();
985        insta::assert_debug_snapshot!(suite);
986    }
987
988    #[test]
989    fn test_fstring_constant_range() {
990        let source = r#"f"aaa{bbb}ccc{ddd}eee""#;
991        let suite = parse_suite(source).unwrap();
992        insta::assert_debug_snapshot!(suite);
993    }
994
995    #[test]
996    fn test_fstring_unescaped_newline() {
997        let source = r#"f"""
998{x}""""#;
999        let suite = parse_suite(source).unwrap();
1000        insta::assert_debug_snapshot!(suite);
1001    }
1002
1003    #[test]
1004    fn test_fstring_escaped_character() {
1005        let source = r#"f"\\{x}""#;
1006        let suite = parse_suite(source).unwrap();
1007        insta::assert_debug_snapshot!(suite);
1008    }
1009
1010    #[test]
1011    fn test_raw_fstring() {
1012        let source = r#"rf"{x}""#;
1013        let suite = parse_suite(source).unwrap();
1014        insta::assert_debug_snapshot!(suite);
1015    }
1016
1017    #[test]
1018    fn test_triple_quoted_raw_fstring() {
1019        let source = r#"rf"""{x}""""#;
1020        let suite = parse_suite(source).unwrap();
1021        insta::assert_debug_snapshot!(suite);
1022    }
1023
1024    #[test]
1025    fn test_fstring_line_continuation() {
1026        let source = r#"rf"\
1027{x}""#;
1028        let suite = parse_suite(source).unwrap();
1029        insta::assert_debug_snapshot!(suite);
1030    }
1031
1032    #[test]
1033    fn test_parse_fstring_nested_string_spec() {
1034        let source = r#"f"{foo:{''}}""#;
1035        let suite = parse_suite(source).unwrap();
1036        insta::assert_debug_snapshot!(suite);
1037    }
1038
1039    #[test]
1040    fn test_parse_fstring_nested_concatenation_string_spec() {
1041        let source = r#"f"{foo:{'' ''}}""#;
1042        let suite = parse_suite(source).unwrap();
1043        insta::assert_debug_snapshot!(suite);
1044    }
1045
1046    #[test]
1047    fn test_tstring_escaped_newline() {
1048        let source = r#"t"\n{x}""#;
1049        let suite = parse_suite(source).unwrap();
1050        insta::assert_debug_snapshot!(suite);
1051    }
1052
1053    #[test]
1054    fn test_tstring_constant_range() {
1055        let source = r#"t"aaa{bbb}ccc{ddd}eee""#;
1056        let suite = parse_suite(source).unwrap();
1057        insta::assert_debug_snapshot!(suite);
1058    }
1059
1060    #[test]
1061    fn test_tstring_unescaped_newline() {
1062        let source = r#"t"""
1063{x}""""#;
1064        let suite = parse_suite(source).unwrap();
1065        insta::assert_debug_snapshot!(suite);
1066    }
1067
1068    #[test]
1069    fn test_tstring_escaped_character() {
1070        let source = r#"t"\\{x}""#;
1071        let suite = parse_suite(source).unwrap();
1072        insta::assert_debug_snapshot!(suite);
1073    }
1074
1075    #[test]
1076    fn test_raw_tstring() {
1077        let source = r#"rt"{x}""#;
1078        let suite = parse_suite(source).unwrap();
1079        insta::assert_debug_snapshot!(suite);
1080    }
1081
1082    #[test]
1083    fn test_triple_quoted_raw_tstring() {
1084        let source = r#"rt"""{x}""""#;
1085        let suite = parse_suite(source).unwrap();
1086        insta::assert_debug_snapshot!(suite);
1087    }
1088
1089    #[test]
1090    fn test_tstring_line_continuation() {
1091        let source = r#"rt"\
1092{x}""#;
1093        let suite = parse_suite(source).unwrap();
1094        insta::assert_debug_snapshot!(suite);
1095    }
1096
1097    #[test]
1098    fn test_parse_tstring_nested_string_spec() {
1099        let source = r#"t"{foo:{''}}""#;
1100        let suite = parse_suite(source).unwrap();
1101        insta::assert_debug_snapshot!(suite);
1102    }
1103
1104    #[test]
1105    fn test_parse_tstring_nested_concatenation_string_spec() {
1106        let source = r#"t"{foo:{'' ''}}""#;
1107        let suite = parse_suite(source).unwrap();
1108        insta::assert_debug_snapshot!(suite);
1109    }
1110
1111    /// <https://github.com/astral-sh/ruff/issues/8355>
1112    #[test]
1113    fn test_dont_panic_on_8_in_octal_escape() {
1114        let source = r"bold = '\038[1m'";
1115        let suite = parse_suite(source).unwrap();
1116        insta::assert_debug_snapshot!(suite);
1117    }
1118
1119    #[test]
1120    fn test_invalid_unicode_literal() {
1121        let source = r"'\x1ó34'";
1122        let error = parse_suite(source).unwrap_err();
1123        insta::assert_debug_snapshot!(error);
1124    }
1125
1126    #[test]
1127    fn test_missing_unicode_lbrace_error() {
1128        let source = r"'\N '";
1129        let error = parse_suite(source).unwrap_err();
1130        insta::assert_debug_snapshot!(error);
1131    }
1132
1133    #[test]
1134    fn test_missing_unicode_rbrace_error() {
1135        let source = r"'\N{SPACE'";
1136        let error = parse_suite(source).unwrap_err();
1137        insta::assert_debug_snapshot!(error);
1138    }
1139
1140    #[test]
1141    fn test_invalid_unicode_name_error() {
1142        let source = r"'\N{INVALID}'";
1143        let error = parse_suite(source).unwrap_err();
1144        insta::assert_debug_snapshot!(error);
1145    }
1146
1147    #[test]
1148    fn test_invalid_byte_literal_error() {
1149        let source = r"b'123a𝐁c'";
1150        let error = parse_suite(source).unwrap_err();
1151        insta::assert_debug_snapshot!(error);
1152    }
1153
1154    macro_rules! test_aliases_parse {
1155        ($($name:ident: $alias:expr,)*) => {
1156        $(
1157            #[test]
1158            fn $name() {
1159                let source = format!(r#""\N{{{0}}}""#, $alias);
1160                let suite = parse_suite(&source).unwrap();
1161                insta::assert_debug_snapshot!(suite);
1162            }
1163        )*
1164        }
1165    }
1166
1167    test_aliases_parse! {
1168        test_backspace_alias: "BACKSPACE",
1169        test_bell_alias: "BEL",
1170        test_carriage_return_alias: "CARRIAGE RETURN",
1171        test_delete_alias: "DELETE",
1172        test_escape_alias: "ESCAPE",
1173        test_form_feed_alias: "FORM FEED",
1174        test_hts_alias: "HTS",
1175        test_character_tabulation_with_justification_alias: "CHARACTER TABULATION WITH JUSTIFICATION",
1176    }
1177}