1use bstr::ByteSlice;
4use std::fmt;
5
6use ruff_python_ast::token::TokenKind;
7use ruff_python_ast::{self as ast, AnyStringFlags, AtomicNodeIndex, Expr, StringFlags};
8use ruff_text_size::{Ranged, TextRange, TextSize};
9
10use crate::error::{LexicalError, LexicalErrorType};
11
12#[derive(Debug)]
13pub(crate) enum StringType {
14 Str(ast::StringLiteral),
15 Bytes(ast::BytesLiteral),
16 FString(ast::FString),
17 TString(ast::TString),
18}
19
20impl Ranged for StringType {
21 fn range(&self) -> TextRange {
22 match self {
23 Self::Str(node) => node.range(),
24 Self::Bytes(node) => node.range(),
25 Self::FString(node) => node.range(),
26 Self::TString(node) => node.range(),
27 }
28 }
29}
30
31impl From<StringType> for Expr {
32 fn from(string: StringType) -> Self {
33 match string {
34 StringType::Str(node) => Expr::from(node),
35 StringType::Bytes(node) => Expr::from(node),
36 StringType::FString(node) => Expr::from(node),
37 StringType::TString(node) => Expr::from(node),
38 }
39 }
40}
41
42#[derive(Debug, Clone, Copy, PartialEq, Eq)]
43pub(crate) enum InterpolatedStringKind {
44 FString,
45 TString,
46}
47
48impl InterpolatedStringKind {
49 #[inline]
50 pub(crate) const fn start_token(self) -> TokenKind {
51 match self {
52 InterpolatedStringKind::FString => TokenKind::FStringStart,
53 InterpolatedStringKind::TString => TokenKind::TStringStart,
54 }
55 }
56
57 #[inline]
58 pub(crate) const fn middle_token(self) -> TokenKind {
59 match self {
60 InterpolatedStringKind::FString => TokenKind::FStringMiddle,
61 InterpolatedStringKind::TString => TokenKind::TStringMiddle,
62 }
63 }
64
65 #[inline]
66 pub(crate) const fn end_token(self) -> TokenKind {
67 match self {
68 InterpolatedStringKind::FString => TokenKind::FStringEnd,
69 InterpolatedStringKind::TString => TokenKind::TStringEnd,
70 }
71 }
72}
73
74impl fmt::Display for InterpolatedStringKind {
75 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
76 match self {
77 InterpolatedStringKind::FString => f.write_str("f-string"),
78 InterpolatedStringKind::TString => f.write_str("t-string"),
79 }
80 }
81}
82
83enum EscapedChar {
84 Literal(char),
85 Escape(char),
86}
87
88struct StringParser<'src> {
89 source: &'src str,
91 cursor: usize,
93 flags: AnyStringFlags,
95 offset: TextSize,
97 range: TextRange,
99}
100
101impl<'src> StringParser<'src> {
102 fn new(source: &'src str, flags: AnyStringFlags, offset: TextSize, range: TextRange) -> Self {
103 Self {
104 source,
105 cursor: 0,
106 flags,
107 offset,
108 range,
109 }
110 }
111
112 #[inline]
113 fn skip_bytes(&mut self, bytes: usize) -> &str {
114 let skipped_str = &self.source[self.cursor..self.cursor + bytes];
115 self.cursor += bytes;
116 skipped_str
117 }
118
119 #[inline]
121 fn position(&self) -> TextSize {
122 self.compute_position(self.cursor)
123 }
124
125 #[inline]
127 fn compute_position(&self, cursor: usize) -> TextSize {
128 self.offset + TextSize::try_from(cursor).unwrap()
129 }
130
131 #[inline]
137 fn next_byte(&mut self) -> Option<u8> {
138 self.source.as_bytes()[self.cursor..].first().map(|&byte| {
139 self.cursor += 1;
140 byte
141 })
142 }
143
144 #[inline]
145 fn next_char(&mut self) -> Option<char> {
146 self.source[self.cursor..].chars().next().inspect(|c| {
147 self.cursor += c.len_utf8();
148 })
149 }
150
151 #[inline]
152 fn peek_byte(&self) -> Option<u8> {
153 self.source.as_bytes()[self.cursor..].first().copied()
154 }
155
156 fn parse_unicode_literal(&mut self, literal_number: usize) -> Result<char, LexicalError> {
157 let mut p: u32 = 0u32;
158 for i in 1..=literal_number {
159 let start = self.position();
160 match self.next_char() {
161 Some(c) => match c.to_digit(16) {
162 Some(d) => p += d << ((literal_number - i) * 4),
163 None => {
164 return Err(LexicalError::new(
165 LexicalErrorType::UnicodeError,
166 TextRange::at(start, TextSize::try_from(c.len_utf8()).unwrap()),
167 ));
168 }
169 },
170 None => {
171 return Err(LexicalError::new(
172 LexicalErrorType::UnicodeError,
173 TextRange::empty(self.position()),
174 ));
175 }
176 }
177 }
178 match p {
179 0xD800..=0xDFFF => Ok(std::char::REPLACEMENT_CHARACTER),
180 _ => std::char::from_u32(p).ok_or(LexicalError::new(
181 LexicalErrorType::UnicodeError,
182 TextRange::empty(self.position()),
183 )),
184 }
185 }
186
187 fn parse_octet(&mut self, o: u8) -> char {
188 let mut radix_bytes = [o, 0, 0];
189 let mut len = 1;
190
191 while len < 3 {
192 let Some(b'0'..=b'7') = self.peek_byte() else {
193 break;
194 };
195
196 radix_bytes[len] = self.next_byte().unwrap();
197 len += 1;
198 }
199
200 let radix_str = std::str::from_utf8(&radix_bytes[..len]).expect("ASCII bytes");
202 let value = u32::from_str_radix(radix_str, 8).unwrap();
203 char::from_u32(value).unwrap()
204 }
205
206 fn parse_unicode_name(&mut self) -> Result<char, LexicalError> {
207 let start_pos = self.position();
208 let Some('{') = self.next_char() else {
209 return Err(LexicalError::new(
210 LexicalErrorType::MissingUnicodeLbrace,
211 TextRange::empty(start_pos),
212 ));
213 };
214
215 let start_pos = self.position();
216 let Some(close_idx) = self.source[self.cursor..].find('}') else {
217 return Err(LexicalError::new(
218 LexicalErrorType::MissingUnicodeRbrace,
219 TextRange::empty(self.compute_position(self.source.len())),
220 ));
221 };
222
223 let name_and_ending = self.skip_bytes(close_idx + 1);
224 let name = &name_and_ending[..name_and_ending.len() - 1];
225
226 unicode_names2::character(name).ok_or_else(|| {
227 LexicalError::new(
228 LexicalErrorType::UnicodeError,
229 TextRange::new(
232 start_pos,
233 self.compute_position(self.cursor - '}'.len_utf8()),
234 ),
235 )
236 })
237 }
238
239 fn parse_escaped_char(&mut self) -> Result<Option<EscapedChar>, LexicalError> {
241 let Some(first_char) = self.next_char() else {
242 return Err(LexicalError::new(
244 LexicalErrorType::StringError,
245 TextRange::empty(self.position()),
246 ));
247 };
248
249 let new_char = match first_char {
250 '\\' => '\\',
251 '\'' => '\'',
252 '\"' => '"',
253 'a' => '\x07',
254 'b' => '\x08',
255 'f' => '\x0c',
256 'n' => '\n',
257 'r' => '\r',
258 't' => '\t',
259 'v' => '\x0b',
260 o @ '0'..='7' => self.parse_octet(o as u8),
261 'x' => self.parse_unicode_literal(2)?,
262 'u' if !self.flags.is_byte_string() => self.parse_unicode_literal(4)?,
263 'U' if !self.flags.is_byte_string() => self.parse_unicode_literal(8)?,
264 'N' if !self.flags.is_byte_string() => self.parse_unicode_name()?,
265 '\n' => return Ok(None),
267 '\r' => {
268 if self.peek_byte() == Some(b'\n') {
269 self.next_byte();
270 }
271
272 return Ok(None);
273 }
274 _ => return Ok(Some(EscapedChar::Escape(first_char))),
275 };
276
277 Ok(Some(EscapedChar::Literal(new_char)))
278 }
279
280 fn parse_interpolated_string_middle(
281 mut self,
282 ) -> Result<ast::InterpolatedStringLiteralElement, LexicalError> {
283 let Some(mut index) = memchr::memchr3(b'{', b'}', b'\\', self.source.as_bytes()) else {
285 return Ok(ast::InterpolatedStringLiteralElement {
286 value: self.source.into(),
287 range: self.range,
288 node_index: AtomicNodeIndex::NONE,
289 });
290 };
291
292 let mut value = String::with_capacity(self.source.len());
293 loop {
294 let before_with_slash_or_brace = self.skip_bytes(index + 1);
296 let before = &before_with_slash_or_brace[..before_with_slash_or_brace.len() - 1];
297 value.push_str(before);
298
299 match self.source.as_bytes()[self.cursor - 1] {
301 brace @ (b'{' | b'}') => {
306 if self.peek_byte() == Some(brace) {
307 self.next_byte();
308 }
309 value.push(char::from(brace));
310 }
311 b'\\' => {
332 if !self.flags.is_raw_string() && self.peek_byte().is_some() {
333 if let Some(brace @ (b'{' | b'}')) = self.peek_byte()
334 && self.source.as_bytes().get(self.cursor + 1).copied() == Some(brace)
335 {
336 value.push('\\');
338 } else {
339 match self.parse_escaped_char()? {
340 None => {}
341 Some(EscapedChar::Literal(c)) => value.push(c),
342 Some(EscapedChar::Escape(c)) => {
343 value.push('\\');
344 value.push(c);
345 }
346 }
347 }
348 } else {
349 value.push('\\');
350 }
351 }
352 ch => {
353 unreachable!("Expected '{{', '}}', or '\\' but got {:?}", ch);
354 }
355 }
356
357 let Some(next_index) =
358 memchr::memchr3(b'{', b'}', b'\\', &self.source.as_bytes()[self.cursor..])
359 else {
360 let rest = &self.source[self.cursor..];
362 value.push_str(rest);
363 break;
364 };
365
366 index = next_index;
367 }
368
369 Ok(ast::InterpolatedStringLiteralElement {
370 value: value.into_boxed_str(),
371 range: self.range,
372 node_index: AtomicNodeIndex::NONE,
373 })
374 }
375
376 fn parse_bytes(mut self) -> Result<StringType, LexicalError> {
377 if let Some(index) = self.source.as_bytes().find_non_ascii_byte() {
378 let ch = self.source.chars().nth(index).unwrap();
379 return Err(LexicalError::new(
380 LexicalErrorType::InvalidByteLiteral,
381 TextRange::at(
382 self.compute_position(index),
383 TextSize::try_from(ch.len_utf8()).unwrap(),
384 ),
385 ));
386 }
387
388 if self.flags.is_raw_string() {
389 return Ok(StringType::Bytes(ast::BytesLiteral {
391 value: self.source.as_bytes().into(),
392 range: self.range,
393 flags: self.flags.into(),
394 node_index: AtomicNodeIndex::NONE,
395 }));
396 }
397
398 let Some(mut escape) = memchr::memchr(b'\\', self.source.as_bytes()) else {
399 return Ok(StringType::Bytes(ast::BytesLiteral {
401 value: self.source.as_bytes().into(),
402 range: self.range,
403 flags: self.flags.into(),
404 node_index: AtomicNodeIndex::NONE,
405 }));
406 };
407
408 let mut value = Vec::with_capacity(self.source.len());
410 loop {
411 let before_with_slash = self.skip_bytes(escape + 1);
413 let before = &before_with_slash[..before_with_slash.len() - 1];
414 value.extend_from_slice(before.as_bytes());
415
416 match self.parse_escaped_char()? {
418 None => {}
419 Some(EscapedChar::Literal(c)) => value.push(c as u8),
420 Some(EscapedChar::Escape(c)) => {
421 value.push(b'\\');
422 value.push(c as u8);
423 }
424 }
425
426 let Some(next_escape) = memchr::memchr(b'\\', &self.source.as_bytes()[self.cursor..])
427 else {
428 let rest = &self.source[self.cursor..];
430 value.extend_from_slice(rest.as_bytes());
431 break;
432 };
433
434 escape = next_escape;
436 }
437
438 Ok(StringType::Bytes(ast::BytesLiteral {
439 value: value.into_boxed_slice(),
440 range: self.range,
441 flags: self.flags.into(),
442 node_index: AtomicNodeIndex::NONE,
443 }))
444 }
445
446 fn parse_string(mut self) -> Result<StringType, LexicalError> {
447 if self.flags.is_raw_string() {
448 return Ok(StringType::Str(ast::StringLiteral {
450 value: self.source.into(),
451 range: self.range,
452 flags: self.flags.into(),
453 node_index: AtomicNodeIndex::NONE,
454 }));
455 }
456
457 let Some(mut escape) = memchr::memchr(b'\\', self.source.as_bytes()) else {
458 return Ok(StringType::Str(ast::StringLiteral {
460 value: self.source.into(),
461 range: self.range,
462 flags: self.flags.into(),
463 node_index: AtomicNodeIndex::NONE,
464 }));
465 };
466
467 let mut value = String::with_capacity(self.source.len());
469
470 loop {
471 let before_with_slash = self.skip_bytes(escape + 1);
473 let before = &before_with_slash[..before_with_slash.len() - 1];
474 value.push_str(before);
475
476 match self.parse_escaped_char()? {
478 None => {}
479 Some(EscapedChar::Literal(c)) => value.push(c),
480 Some(EscapedChar::Escape(c)) => {
481 value.push('\\');
482 value.push(c);
483 }
484 }
485
486 let Some(next_escape) = self.source[self.cursor..].find('\\') else {
487 let rest = &self.source[self.cursor..];
489 value.push_str(rest);
490 break;
491 };
492
493 escape = next_escape;
495 }
496
497 Ok(StringType::Str(ast::StringLiteral {
498 value: value.into_boxed_str(),
499 range: self.range,
500 flags: self.flags.into(),
501 node_index: AtomicNodeIndex::NONE,
502 }))
503 }
504
505 fn parse(self) -> Result<StringType, LexicalError> {
506 if self.flags.is_byte_string() {
507 self.parse_bytes()
508 } else {
509 self.parse_string()
510 }
511 }
512}
513
514pub(crate) fn parse_string_literal(
515 source: &str,
516 flags: AnyStringFlags,
517 range: TextRange,
518) -> Result<StringType, LexicalError> {
519 StringParser::new(source, flags, range.start() + flags.opener_len(), range).parse()
520}
521
522pub(crate) fn parse_interpolated_string_literal_element(
523 source: &str,
524 flags: AnyStringFlags,
525 range: TextRange,
526) -> Result<ast::InterpolatedStringLiteralElement, LexicalError> {
527 StringParser::new(source, flags, range.start(), range).parse_interpolated_string_middle()
528}
529
530#[cfg(test)]
531mod tests {
532 use ruff_python_ast::Suite;
533
534 use crate::error::LexicalErrorType;
535 use crate::{InterpolatedStringErrorType, ParseError, ParseErrorType, Parsed, parse_module};
536
537 const WINDOWS_EOL: &str = "\r\n";
538 const MAC_EOL: &str = "\r";
539 const UNIX_EOL: &str = "\n";
540
541 fn parse_suite(source: &str) -> Result<Suite, ParseError> {
542 parse_module(source).map(Parsed::into_suite)
543 }
544
545 fn nested_format_spec(prefix: char, depth: usize) -> String {
546 let mut replacement_field = String::from("{spec}");
547 for _ in 0..depth {
548 replacement_field = format!("{{foo:{replacement_field}}}");
549 }
550 format!(r#"{prefix}"{replacement_field}""#)
551 }
552
553 fn string_parser_escaped_eol(eol: &str) -> Suite {
554 let source = format!(r"'text \{eol}more text'");
555 parse_suite(&source).unwrap()
556 }
557
558 #[test]
559 fn test_string_parser_escaped_unix_eol() {
560 let suite = string_parser_escaped_eol(UNIX_EOL);
561 insta::assert_debug_snapshot!(suite);
562 }
563
564 #[test]
565 fn test_string_parser_escaped_mac_eol() {
566 let suite = string_parser_escaped_eol(MAC_EOL);
567 insta::assert_debug_snapshot!(suite);
568 }
569
570 #[test]
571 fn test_string_parser_escaped_windows_eol() {
572 let suite = string_parser_escaped_eol(WINDOWS_EOL);
573 insta::assert_debug_snapshot!(suite);
574 }
575
576 #[test]
577 fn test_parse_fstring() {
578 let source = r#"f"{a}{ b }{{foo}}""#;
579 let suite = parse_suite(source).unwrap();
580 insta::assert_debug_snapshot!(suite);
581 }
582
583 #[test]
584 fn test_parse_fstring_nested_spec() {
585 let source = r#"f"{foo:{spec}}""#;
586 let suite = parse_suite(source).unwrap();
587 insta::assert_debug_snapshot!(suite);
588 }
589
590 #[test]
591 fn parse_fstring_nested_spec_grows_stack() {
592 assert!(parse_suite(&nested_format_spec('f', 200)).is_ok());
593 }
594
595 #[test]
596 fn test_parse_fstring_not_nested_spec() {
597 let source = r#"f"{foo:spec}""#;
598 let suite = parse_suite(source).unwrap();
599 insta::assert_debug_snapshot!(suite);
600 }
601
602 #[test]
603 fn test_parse_empty_fstring() {
604 let source = r#"f"""#;
605 let suite = parse_suite(source).unwrap();
606 insta::assert_debug_snapshot!(suite);
607 }
608
609 #[test]
610 fn test_fstring_parse_self_documenting_base() {
611 let source = r#"f"{user=}""#;
612 let suite = parse_suite(source).unwrap();
613 insta::assert_debug_snapshot!(suite);
614 }
615
616 #[test]
617 fn test_fstring_parse_self_documenting_base_more() {
618 let source = r#"f"mix {user=} with text and {second=}""#;
619 let suite = parse_suite(source).unwrap();
620 insta::assert_debug_snapshot!(suite);
621 }
622
623 #[test]
624 fn test_fstring_parse_self_documenting_format() {
625 let source = r#"f"{user=:>10}""#;
626 let suite = parse_suite(source).unwrap();
627 insta::assert_debug_snapshot!(suite);
628 }
629
630 fn parse_fstring_error(source: &str) -> InterpolatedStringErrorType {
631 parse_suite(source)
632 .map_err(|e| match e.error {
633 ParseErrorType::Lexical(LexicalErrorType::FStringError(e)) => e,
634 ParseErrorType::FStringError(e) => e,
635 e => unreachable!("Expected FStringError: {:?}", e),
636 })
637 .expect_err("Expected error")
638 }
639
640 #[test]
641 fn test_parse_invalid_fstring() {
642 use InterpolatedStringErrorType::{InvalidConversionFlag, LambdaWithoutParentheses};
643
644 assert_eq!(parse_fstring_error(r#"f"{5!x}""#), InvalidConversionFlag);
645 assert_eq!(
646 parse_fstring_error("f'{lambda x:{x}}'"),
647 LambdaWithoutParentheses
648 );
649 assert!(parse_suite(r#"f"{class}""#).is_err());
656 }
657
658 #[test]
659 fn test_parse_fstring_not_equals() {
660 let source = r#"f"{1 != 2}""#;
661 let suite = parse_suite(source).unwrap();
662 insta::assert_debug_snapshot!(suite);
663 }
664
665 #[test]
666 fn test_parse_fstring_equals() {
667 let source = r#"f"{42 == 42}""#;
668 let suite = parse_suite(source).unwrap();
669 insta::assert_debug_snapshot!(suite);
670 }
671
672 #[test]
673 fn test_parse_fstring_self_doc_prec_space() {
674 let source = r#"f"{x =}""#;
675 let suite = parse_suite(source).unwrap();
676 insta::assert_debug_snapshot!(suite);
677 }
678
679 #[test]
680 fn test_parse_fstring_self_doc_trailing_space() {
681 let source = r#"f"{x= }""#;
682 let suite = parse_suite(source).unwrap();
683 insta::assert_debug_snapshot!(suite);
684 }
685
686 #[test]
687 fn test_parse_fstring_yield_expr() {
688 let source = r#"f"{yield}""#;
689 let suite = parse_suite(source).unwrap();
690 insta::assert_debug_snapshot!(suite);
691 }
692
693 #[test]
694 fn test_parse_tstring() {
695 let source = r#"t"{a}{ b }{{foo}}""#;
696 let suite = parse_suite(source).unwrap();
697 insta::assert_debug_snapshot!(suite);
698 }
699
700 #[test]
701 fn test_parse_tstring_nested_spec() {
702 let source = r#"t"{foo:{spec}}""#;
703 let suite = parse_suite(source).unwrap();
704 insta::assert_debug_snapshot!(suite);
705 }
706
707 #[test]
708 fn parse_tstring_nested_spec_grows_stack() {
709 assert!(parse_suite(&nested_format_spec('t', 200)).is_ok());
710 }
711
712 #[test]
713 fn test_parse_tstring_not_nested_spec() {
714 let source = r#"t"{foo:spec}""#;
715 let suite = parse_suite(source).unwrap();
716 insta::assert_debug_snapshot!(suite);
717 }
718
719 #[test]
720 fn test_parse_empty_tstring() {
721 let source = r#"t"""#;
722 let suite = parse_suite(source).unwrap();
723 insta::assert_debug_snapshot!(suite);
724 }
725
726 #[test]
727 fn test_tstring_parse_self_documenting_base() {
728 let source = r#"t"{user=}""#;
729 let suite = parse_suite(source).unwrap();
730 insta::assert_debug_snapshot!(suite);
731 }
732
733 #[test]
734 fn test_tstring_parse_self_documenting_base_more() {
735 let source = r#"t"mix {user=} with text and {second=}""#;
736 let suite = parse_suite(source).unwrap();
737 insta::assert_debug_snapshot!(suite);
738 }
739
740 #[test]
741 fn test_tstring_parse_self_documenting_format() {
742 let source = r#"t"{user=:>10}""#;
743 let suite = parse_suite(source).unwrap();
744 insta::assert_debug_snapshot!(suite);
745 }
746
747 fn parse_tstring_error(source: &str) -> InterpolatedStringErrorType {
748 parse_suite(source)
749 .map_err(|e| match e.error {
750 ParseErrorType::Lexical(LexicalErrorType::TStringError(e)) => e,
751 ParseErrorType::TStringError(e) => e,
752 e => unreachable!("Expected TStringError: {:?}", e),
753 })
754 .expect_err("Expected error")
755 }
756
757 #[test]
758 fn test_parse_invalid_tstring() {
759 use InterpolatedStringErrorType::{InvalidConversionFlag, LambdaWithoutParentheses};
760
761 assert_eq!(parse_tstring_error(r#"t"{5!x}""#), InvalidConversionFlag);
762 assert_eq!(
763 parse_tstring_error("t'{lambda x:{x}}'"),
764 LambdaWithoutParentheses
765 );
766 assert!(parse_suite(r#"t"{class}""#).is_err());
773 }
774
775 #[test]
776 fn test_parse_tstring_not_equals() {
777 let source = r#"t"{1 != 2}""#;
778 let suite = parse_suite(source).unwrap();
779 insta::assert_debug_snapshot!(suite);
780 }
781
782 #[test]
783 fn test_parse_tstring_equals() {
784 let source = r#"t"{42 == 42}""#;
785 let suite = parse_suite(source).unwrap();
786 insta::assert_debug_snapshot!(suite);
787 }
788
789 #[test]
790 fn test_parse_tstring_self_doc_prec_space() {
791 let source = r#"t"{x =}""#;
792 let suite = parse_suite(source).unwrap();
793 insta::assert_debug_snapshot!(suite);
794 }
795
796 #[test]
797 fn test_parse_tstring_self_doc_trailing_space() {
798 let source = r#"t"{x= }""#;
799 let suite = parse_suite(source).unwrap();
800 insta::assert_debug_snapshot!(suite);
801 }
802
803 #[test]
804 fn test_parse_tstring_yield_expr() {
805 let source = r#"t"{yield}""#;
806 let suite = parse_suite(source).unwrap();
807 insta::assert_debug_snapshot!(suite);
808 }
809
810 #[test]
811 fn test_parse_string_concat() {
812 let source = "'Hello ' 'world'";
813 let suite = parse_suite(source).unwrap();
814 insta::assert_debug_snapshot!(suite);
815 }
816
817 #[test]
818 fn test_parse_u_string_concat_1() {
819 let source = "'Hello ' u'world'";
820 let suite = parse_suite(source).unwrap();
821 insta::assert_debug_snapshot!(suite);
822 }
823
824 #[test]
825 fn test_parse_u_string_concat_2() {
826 let source = "u'Hello ' 'world'";
827 let suite = parse_suite(source).unwrap();
828 insta::assert_debug_snapshot!(suite);
829 }
830
831 #[test]
832 fn test_parse_f_string_concat_1() {
833 let source = "'Hello ' f'world'";
834 let suite = parse_suite(source).unwrap();
835 insta::assert_debug_snapshot!(suite);
836 }
837
838 #[test]
839 fn test_parse_f_string_concat_2() {
840 let source = "'Hello ' f'world'";
841 let suite = parse_suite(source).unwrap();
842 insta::assert_debug_snapshot!(suite);
843 }
844
845 #[test]
846 fn test_parse_f_string_concat_3() {
847 let source = "'Hello ' f'world{\"!\"}'";
848 let suite = parse_suite(source).unwrap();
849 insta::assert_debug_snapshot!(suite);
850 }
851
852 #[test]
853 fn test_parse_f_string_concat_4() {
854 let source = "'Hello ' f'world{\"!\"}' 'again!'";
855 let suite = parse_suite(source).unwrap();
856 insta::assert_debug_snapshot!(suite);
857 }
858
859 #[test]
860 fn test_parse_u_f_string_concat_1() {
861 let source = "u'Hello ' f'world'";
862 let suite = parse_suite(source).unwrap();
863 insta::assert_debug_snapshot!(suite);
864 }
865
866 #[test]
867 fn test_parse_u_f_string_concat_2() {
868 let source = "u'Hello ' f'world' '!'";
869 let suite = parse_suite(source).unwrap();
870 insta::assert_debug_snapshot!(suite);
871 }
872
873 #[test]
874 fn test_parse_t_string_concat_1_error() {
875 let source = "'Hello ' t'world'";
876 let suite = parse_suite(source).unwrap_err();
877 insta::assert_debug_snapshot!(suite);
878 }
879
880 #[test]
881 fn test_parse_t_string_concat_2_error() {
882 let source = "'Hello ' t'world'";
883 let suite = parse_suite(source).unwrap_err();
884 insta::assert_debug_snapshot!(suite);
885 }
886
887 #[test]
888 fn test_parse_t_string_concat_3_error() {
889 let source = "'Hello ' t'world{\"!\"}'";
890 let suite = parse_suite(source).unwrap_err();
891 insta::assert_debug_snapshot!(suite);
892 }
893
894 #[test]
895 fn test_parse_t_string_concat_4_error() {
896 let source = "'Hello ' t'world{\"!\"}' 'again!'";
897 let suite = parse_suite(source).unwrap_err();
898 insta::assert_debug_snapshot!(suite);
899 }
900
901 #[test]
902 fn test_parse_u_t_string_concat_1_error() {
903 let source = "u'Hello ' t'world'";
904 let suite = parse_suite(source).unwrap_err();
905 insta::assert_debug_snapshot!(suite);
906 }
907
908 #[test]
909 fn test_parse_u_t_string_concat_2_error() {
910 let source = "u'Hello ' t'world' '!'";
911 let suite = parse_suite(source).unwrap_err();
912 insta::assert_debug_snapshot!(suite);
913 }
914
915 #[test]
916 fn test_parse_f_t_string_concat_1_error() {
917 let source = "f'Hello ' t'world'";
918 let suite = parse_suite(source).unwrap_err();
919 insta::assert_debug_snapshot!(suite);
920 }
921
922 #[test]
923 fn test_parse_f_t_string_concat_2_error() {
924 let source = "f'Hello ' t'world' '!'";
925 let suite = parse_suite(source).unwrap_err();
926 insta::assert_debug_snapshot!(suite);
927 }
928
929 #[test]
930 fn test_parse_string_triple_quotes_with_kind() {
931 let source = "u'''Hello, world!'''";
932 let suite = parse_suite(source).unwrap();
933 insta::assert_debug_snapshot!(suite);
934 }
935
936 #[test]
937 fn test_single_quoted_byte() {
938 let source = r##"b'\x00\x01\x02\x03\x04\x05\x06\x07\x08\t\n\x0b\x0c\r\x0e\x0f\x10\x11\x12\x13\x14\x15\x16\x17\x18\x19\x1a\x1b\x1c\x1d\x1e\x1f !"#$%&\'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\\]^_`abcdefghijklmnopqrstuvwxyz{|}~\x7f\x80\x81\x82\x83\x84\x85\x86\x87\x88\x89\x8a\x8b\x8c\x8d\x8e\x8f\x90\x91\x92\x93\x94\x95\x96\x97\x98\x99\x9a\x9b\x9c\x9d\x9e\x9f\xa0\xa1\xa2\xa3\xa4\xa5\xa6\xa7\xa8\xa9\xaa\xab\xac\xad\xae\xaf\xb0\xb1\xb2\xb3\xb4\xb5\xb6\xb7\xb8\xb9\xba\xbb\xbc\xbd\xbe\xbf\xc0\xc1\xc2\xc3\xc4\xc5\xc6\xc7\xc8\xc9\xca\xcb\xcc\xcd\xce\xcf\xd0\xd1\xd2\xd3\xd4\xd5\xd6\xd7\xd8\xd9\xda\xdb\xdc\xdd\xde\xdf\xe0\xe1\xe2\xe3\xe4\xe5\xe6\xe7\xe8\xe9\xea\xeb\xec\xed\xee\xef\xf0\xf1\xf2\xf3\xf4\xf5\xf6\xf7\xf8\xf9\xfa\xfb\xfc\xfd\xfe\xff'"##;
940 let suite = parse_suite(source).unwrap();
941 insta::assert_debug_snapshot!(suite);
942 }
943
944 #[test]
945 fn test_double_quoted_byte() {
946 let source = r##"b"\x00\x01\x02\x03\x04\x05\x06\x07\x08\t\n\x0b\x0c\r\x0e\x0f\x10\x11\x12\x13\x14\x15\x16\x17\x18\x19\x1a\x1b\x1c\x1d\x1e\x1f !\"#$%&'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\\]^_`abcdefghijklmnopqrstuvwxyz{|}~\x7f\x80\x81\x82\x83\x84\x85\x86\x87\x88\x89\x8a\x8b\x8c\x8d\x8e\x8f\x90\x91\x92\x93\x94\x95\x96\x97\x98\x99\x9a\x9b\x9c\x9d\x9e\x9f\xa0\xa1\xa2\xa3\xa4\xa5\xa6\xa7\xa8\xa9\xaa\xab\xac\xad\xae\xaf\xb0\xb1\xb2\xb3\xb4\xb5\xb6\xb7\xb8\xb9\xba\xbb\xbc\xbd\xbe\xbf\xc0\xc1\xc2\xc3\xc4\xc5\xc6\xc7\xc8\xc9\xca\xcb\xcc\xcd\xce\xcf\xd0\xd1\xd2\xd3\xd4\xd5\xd6\xd7\xd8\xd9\xda\xdb\xdc\xdd\xde\xdf\xe0\xe1\xe2\xe3\xe4\xe5\xe6\xe7\xe8\xe9\xea\xeb\xec\xed\xee\xef\xf0\xf1\xf2\xf3\xf4\xf5\xf6\xf7\xf8\xf9\xfa\xfb\xfc\xfd\xfe\xff""##;
948 let suite = parse_suite(source).unwrap();
949 insta::assert_debug_snapshot!(suite);
950 }
951
952 #[test]
953 fn test_escape_char_in_byte_literal() {
954 let source = r#"b"omkmok\Xaa""#; let suite = parse_suite(source).unwrap();
957 insta::assert_debug_snapshot!(suite);
958 }
959
960 #[test]
961 fn test_raw_byte_literal_1() {
962 let source = r"rb'\x1z'";
963 let suite = parse_suite(source).unwrap();
964 insta::assert_debug_snapshot!(suite);
965 }
966
967 #[test]
968 fn test_raw_byte_literal_2() {
969 let source = r"rb'\\'";
970 let suite = parse_suite(source).unwrap();
971 insta::assert_debug_snapshot!(suite);
972 }
973
974 #[test]
975 fn test_escape_octet() {
976 let source = r"b'\43a\4\1234'";
977 let suite = parse_suite(source).unwrap();
978 insta::assert_debug_snapshot!(suite);
979 }
980
981 #[test]
982 fn test_fstring_escaped_newline() {
983 let source = r#"f"\n{x}""#;
984 let suite = parse_suite(source).unwrap();
985 insta::assert_debug_snapshot!(suite);
986 }
987
988 #[test]
989 fn test_fstring_constant_range() {
990 let source = r#"f"aaa{bbb}ccc{ddd}eee""#;
991 let suite = parse_suite(source).unwrap();
992 insta::assert_debug_snapshot!(suite);
993 }
994
995 #[test]
996 fn test_fstring_unescaped_newline() {
997 let source = r#"f"""
998{x}""""#;
999 let suite = parse_suite(source).unwrap();
1000 insta::assert_debug_snapshot!(suite);
1001 }
1002
1003 #[test]
1004 fn test_fstring_escaped_character() {
1005 let source = r#"f"\\{x}""#;
1006 let suite = parse_suite(source).unwrap();
1007 insta::assert_debug_snapshot!(suite);
1008 }
1009
1010 #[test]
1011 fn test_raw_fstring() {
1012 let source = r#"rf"{x}""#;
1013 let suite = parse_suite(source).unwrap();
1014 insta::assert_debug_snapshot!(suite);
1015 }
1016
1017 #[test]
1018 fn test_triple_quoted_raw_fstring() {
1019 let source = r#"rf"""{x}""""#;
1020 let suite = parse_suite(source).unwrap();
1021 insta::assert_debug_snapshot!(suite);
1022 }
1023
1024 #[test]
1025 fn test_fstring_line_continuation() {
1026 let source = r#"rf"\
1027{x}""#;
1028 let suite = parse_suite(source).unwrap();
1029 insta::assert_debug_snapshot!(suite);
1030 }
1031
1032 #[test]
1033 fn test_parse_fstring_nested_string_spec() {
1034 let source = r#"f"{foo:{''}}""#;
1035 let suite = parse_suite(source).unwrap();
1036 insta::assert_debug_snapshot!(suite);
1037 }
1038
1039 #[test]
1040 fn test_parse_fstring_nested_concatenation_string_spec() {
1041 let source = r#"f"{foo:{'' ''}}""#;
1042 let suite = parse_suite(source).unwrap();
1043 insta::assert_debug_snapshot!(suite);
1044 }
1045
1046 #[test]
1047 fn test_tstring_escaped_newline() {
1048 let source = r#"t"\n{x}""#;
1049 let suite = parse_suite(source).unwrap();
1050 insta::assert_debug_snapshot!(suite);
1051 }
1052
1053 #[test]
1054 fn test_tstring_constant_range() {
1055 let source = r#"t"aaa{bbb}ccc{ddd}eee""#;
1056 let suite = parse_suite(source).unwrap();
1057 insta::assert_debug_snapshot!(suite);
1058 }
1059
1060 #[test]
1061 fn test_tstring_unescaped_newline() {
1062 let source = r#"t"""
1063{x}""""#;
1064 let suite = parse_suite(source).unwrap();
1065 insta::assert_debug_snapshot!(suite);
1066 }
1067
1068 #[test]
1069 fn test_tstring_escaped_character() {
1070 let source = r#"t"\\{x}""#;
1071 let suite = parse_suite(source).unwrap();
1072 insta::assert_debug_snapshot!(suite);
1073 }
1074
1075 #[test]
1076 fn test_raw_tstring() {
1077 let source = r#"rt"{x}""#;
1078 let suite = parse_suite(source).unwrap();
1079 insta::assert_debug_snapshot!(suite);
1080 }
1081
1082 #[test]
1083 fn test_triple_quoted_raw_tstring() {
1084 let source = r#"rt"""{x}""""#;
1085 let suite = parse_suite(source).unwrap();
1086 insta::assert_debug_snapshot!(suite);
1087 }
1088
1089 #[test]
1090 fn test_tstring_line_continuation() {
1091 let source = r#"rt"\
1092{x}""#;
1093 let suite = parse_suite(source).unwrap();
1094 insta::assert_debug_snapshot!(suite);
1095 }
1096
1097 #[test]
1098 fn test_parse_tstring_nested_string_spec() {
1099 let source = r#"t"{foo:{''}}""#;
1100 let suite = parse_suite(source).unwrap();
1101 insta::assert_debug_snapshot!(suite);
1102 }
1103
1104 #[test]
1105 fn test_parse_tstring_nested_concatenation_string_spec() {
1106 let source = r#"t"{foo:{'' ''}}""#;
1107 let suite = parse_suite(source).unwrap();
1108 insta::assert_debug_snapshot!(suite);
1109 }
1110
1111 #[test]
1113 fn test_dont_panic_on_8_in_octal_escape() {
1114 let source = r"bold = '\038[1m'";
1115 let suite = parse_suite(source).unwrap();
1116 insta::assert_debug_snapshot!(suite);
1117 }
1118
1119 #[test]
1120 fn test_invalid_unicode_literal() {
1121 let source = r"'\x1ó34'";
1122 let error = parse_suite(source).unwrap_err();
1123 insta::assert_debug_snapshot!(error);
1124 }
1125
1126 #[test]
1127 fn test_missing_unicode_lbrace_error() {
1128 let source = r"'\N '";
1129 let error = parse_suite(source).unwrap_err();
1130 insta::assert_debug_snapshot!(error);
1131 }
1132
1133 #[test]
1134 fn test_missing_unicode_rbrace_error() {
1135 let source = r"'\N{SPACE'";
1136 let error = parse_suite(source).unwrap_err();
1137 insta::assert_debug_snapshot!(error);
1138 }
1139
1140 #[test]
1141 fn test_invalid_unicode_name_error() {
1142 let source = r"'\N{INVALID}'";
1143 let error = parse_suite(source).unwrap_err();
1144 insta::assert_debug_snapshot!(error);
1145 }
1146
1147 #[test]
1148 fn test_invalid_byte_literal_error() {
1149 let source = r"b'123a𝐁c'";
1150 let error = parse_suite(source).unwrap_err();
1151 insta::assert_debug_snapshot!(error);
1152 }
1153
1154 macro_rules! test_aliases_parse {
1155 ($($name:ident: $alias:expr,)*) => {
1156 $(
1157 #[test]
1158 fn $name() {
1159 let source = format!(r#""\N{{{0}}}""#, $alias);
1160 let suite = parse_suite(&source).unwrap();
1161 insta::assert_debug_snapshot!(suite);
1162 }
1163 )*
1164 }
1165 }
1166
1167 test_aliases_parse! {
1168 test_backspace_alias: "BACKSPACE",
1169 test_bell_alias: "BEL",
1170 test_carriage_return_alias: "CARRIAGE RETURN",
1171 test_delete_alias: "DELETE",
1172 test_escape_alias: "ESCAPE",
1173 test_form_feed_alias: "FORM FEED",
1174 test_hts_alias: "HTS",
1175 test_character_tabulation_with_justification_alias: "CHARACTER TABULATION WITH JUSTIFICATION",
1176 }
1177}