Skip to main content

ruff_python_ast/
token.rs

1//! Token kinds for Python source code created by the lexer and consumed by the `ruff_python_parser`.
2//!
3//! This module defines the tokens that the lexer recognizes. The tokens are
4//! loosely based on the token definitions found in the [CPython source].
5//!
6//! [CPython source]: https://github.com/python/cpython/blob/dfc2e065a2e71011017077e549cd2f9bf4944c54/Grammar/Tokens
7
8use std::fmt;
9
10use bitflags::bitflags;
11
12use crate::str::{Quote, TripleQuotes};
13use crate::str_prefix::{
14    AnyStringPrefix, ByteStringPrefix, FStringPrefix, StringLiteralPrefix, TStringPrefix,
15};
16use crate::{AnyStringFlags, BoolOp, Operator, StringFlags, UnaryOp};
17use ruff_text_size::{Ranged, TextRange};
18
19mod parentheses;
20mod tokens;
21
22pub use parentheses::{parentheses_iterator, parenthesized_range};
23pub use tokens::{TokenAt, TokenIterWithContext, Tokens};
24
25#[derive(Clone, Copy, PartialEq, Eq)]
26#[cfg_attr(feature = "get-size", derive(get_size2::GetSize))]
27pub struct Token {
28    /// The kind of the token.
29    kind: TokenKind,
30    /// The range of the token.
31    range: TextRange,
32    /// The set of flags describing this token.
33    flags: TokenFlags,
34}
35
36impl Token {
37    pub fn new(kind: TokenKind, range: TextRange, flags: TokenFlags) -> Token {
38        Self { kind, range, flags }
39    }
40
41    /// Returns the token kind.
42    #[inline]
43    pub const fn kind(&self) -> TokenKind {
44        self.kind
45    }
46
47    /// Returns the token as a tuple of (kind, range).
48    #[inline]
49    pub const fn as_tuple(&self) -> (TokenKind, TextRange) {
50        (self.kind, self.range)
51    }
52
53    /// Returns `true` if the current token is a triple-quoted string of any kind.
54    ///
55    /// # Panics
56    ///
57    /// If it isn't a string or any f/t-string tokens.
58    pub fn is_triple_quoted_string(self) -> bool {
59        self.unwrap_string_flags().is_triple_quoted()
60    }
61
62    /// Returns the [`Quote`] style for the current string token of any kind.
63    ///
64    /// # Panics
65    ///
66    /// If it isn't a string or any f/t-string tokens.
67    pub fn string_quote_style(self) -> Quote {
68        self.unwrap_string_flags().quote_style()
69    }
70
71    /// Returns the [`AnyStringFlags`] style for the current string token of any kind.
72    ///
73    /// # Panics
74    ///
75    /// If it isn't a string or any f/t-string tokens.
76    pub fn unwrap_string_flags(self) -> AnyStringFlags {
77        self.string_flags()
78            .unwrap_or_else(|| panic!("token to be a string"))
79    }
80
81    /// Returns true if the current token is a string and it is raw.
82    pub fn string_flags(self) -> Option<AnyStringFlags> {
83        if self.is_any_string() {
84            Some(self.flags.as_any_string_flags())
85        } else {
86            None
87        }
88    }
89
90    /// Returns `true` if this is any kind of string token - including
91    /// tokens in t-strings (which do not have type `str`).
92    const fn is_any_string(self) -> bool {
93        matches!(
94            self.kind,
95            TokenKind::String
96                | TokenKind::FStringStart
97                | TokenKind::FStringMiddle
98                | TokenKind::FStringEnd
99                | TokenKind::TStringStart
100                | TokenKind::TStringMiddle
101                | TokenKind::TStringEnd
102        )
103    }
104}
105
106impl Ranged for Token {
107    fn range(&self) -> TextRange {
108        self.range
109    }
110}
111
112impl fmt::Debug for Token {
113    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
114        write!(f, "{:?} {:?}", self.kind, self.range)?;
115        if !self.flags.is_empty() {
116            f.write_str(" (flags = ")?;
117            let mut first = true;
118            for (name, _) in self.flags.iter_names() {
119                if first {
120                    first = false;
121                } else {
122                    f.write_str(" | ")?;
123                }
124                f.write_str(name)?;
125            }
126            f.write_str(")")?;
127        }
128        Ok(())
129    }
130}
131
132/// A kind of a token.
133#[derive(Copy, Clone, PartialEq, Eq, Hash, Debug, PartialOrd, Ord)]
134#[cfg_attr(feature = "get-size", derive(get_size2::GetSize))]
135pub enum TokenKind {
136    /// Token kind for an identifier.
137    ///
138    /// The lexer emits separate token kinds for keywords and soft keywords. The parser
139    /// converts soft keyword tokens to this kind when they are used as identifiers.
140    Identifier,
141    /// Token kind for an integer.
142    Int,
143    /// Token kind for a floating point number.
144    Float,
145    /// Token kind for a complex number.
146    Complex,
147    /// Token kind for a string.
148    String,
149    /// Token kind for the start of an f-string. This includes the `f`/`F`/`fr` prefix
150    /// and the opening quote(s).
151    FStringStart,
152    /// Token kind that includes the portion of text inside the f-string that's not
153    /// part of the expression part and isn't an opening or closing brace.
154    FStringMiddle,
155    /// Token kind for the end of an f-string. This includes the closing quote.
156    FStringEnd,
157    /// Token kind for the start of a t-string. This includes the `t`/`T`/`tr` prefix
158    /// and the opening quote(s).
159    TStringStart,
160    /// Token kind that includes the portion of text inside the t-string that's not
161    /// part of the interpolation part and isn't an opening or closing brace.
162    TStringMiddle,
163    /// Token kind for the end of a t-string. This includes the closing quote.
164    TStringEnd,
165    /// Token kind for a IPython escape command.
166    IpyEscapeCommand,
167    /// Token kind for a comment. These are filtered out of the token stream prior to parsing.
168    Comment,
169    /// Token kind for a newline.
170    Newline,
171    /// Token kind for a newline that is not a logical line break. These are filtered out of
172    /// the token stream prior to parsing.
173    NonLogicalNewline,
174    /// Token kind for an indent.
175    Indent,
176    /// Token kind for a dedent.
177    Dedent,
178    EndOfFile,
179    /// Token kind for a question mark `?`.
180    Question,
181    /// Token kind for an exclamation mark `!`.
182    Exclamation,
183    /// Token kind for a left parenthesis `(`.
184    Lpar,
185    /// Token kind for a right parenthesis `)`.
186    Rpar,
187    /// Token kind for a left square bracket `[`.
188    Lsqb,
189    /// Token kind for a right square bracket `]`.
190    Rsqb,
191    /// Token kind for a colon `:`.
192    Colon,
193    /// Token kind for a comma `,`.
194    Comma,
195    /// Token kind for a semicolon `;`.
196    Semi,
197    /// Token kind for plus `+`.
198    Plus,
199    /// Token kind for minus `-`.
200    Minus,
201    /// Token kind for star `*`.
202    Star,
203    /// Token kind for slash `/`.
204    Slash,
205    /// Token kind for vertical bar `|`.
206    Vbar,
207    /// Token kind for ampersand `&`.
208    Amper,
209    /// Token kind for less than `<`.
210    Less,
211    /// Token kind for greater than `>`.
212    Greater,
213    /// Token kind for equal `=`.
214    Equal,
215    /// Token kind for dot `.`.
216    Dot,
217    /// Token kind for percent `%`.
218    Percent,
219    /// Token kind for left bracket `{`.
220    Lbrace,
221    /// Token kind for right bracket `}`.
222    Rbrace,
223    /// Token kind for double equal `==`.
224    EqEqual,
225    /// Token kind for not equal `!=`.
226    NotEqual,
227    /// Token kind for less than or equal `<=`.
228    LessEqual,
229    /// Token kind for greater than or equal `>=`.
230    GreaterEqual,
231    /// Token kind for tilde `~`.
232    Tilde,
233    /// Token kind for caret `^`.
234    CircumFlex,
235    /// Token kind for left shift `<<`.
236    LeftShift,
237    /// Token kind for right shift `>>`.
238    RightShift,
239    /// Token kind for double star `**`.
240    DoubleStar,
241    /// Token kind for double star equal `**=`.
242    DoubleStarEqual,
243    /// Token kind for plus equal `+=`.
244    PlusEqual,
245    /// Token kind for minus equal `-=`.
246    MinusEqual,
247    /// Token kind for star equal `*=`.
248    StarEqual,
249    /// Token kind for slash equal `/=`.
250    SlashEqual,
251    /// Token kind for percent equal `%=`.
252    PercentEqual,
253    /// Token kind for ampersand equal `&=`.
254    AmperEqual,
255    /// Token kind for vertical bar equal `|=`.
256    VbarEqual,
257    /// Token kind for caret equal `^=`.
258    CircumflexEqual,
259    /// Token kind for left shift equal `<<=`.
260    LeftShiftEqual,
261    /// Token kind for right shift equal `>>=`.
262    RightShiftEqual,
263    /// Token kind for double slash `//`.
264    DoubleSlash,
265    /// Token kind for double slash equal `//=`.
266    DoubleSlashEqual,
267    /// Token kind for colon equal `:=`.
268    ColonEqual,
269    /// Token kind for at `@`.
270    At,
271    /// Token kind for at equal `@=`.
272    AtEqual,
273    /// Token kind for arrow `->`.
274    Rarrow,
275    /// Token kind for ellipsis `...`.
276    Ellipsis,
277
278    // The keywords should be sorted in alphabetical order. If the boundary tokens for the
279    // "Keywords" and "Soft keywords" group change, update the related methods on `TokenKind`.
280
281    // Keywords
282    And,
283    As,
284    Assert,
285    Async,
286    Await,
287    Break,
288    Class,
289    Continue,
290    Def,
291    Del,
292    Elif,
293    Else,
294    Except,
295    False,
296    Finally,
297    For,
298    From,
299    Global,
300    If,
301    Import,
302    In,
303    Is,
304    Lambda,
305    None,
306    Nonlocal,
307    Not,
308    Or,
309    Pass,
310    Raise,
311    Return,
312    True,
313    Try,
314    While,
315    With,
316    Yield,
317
318    // Soft keywords
319    Case,
320    Lazy,
321    Match,
322    Type,
323
324    Unknown,
325}
326
327impl TokenKind {
328    /// Returns `true` if this is an end of file token.
329    #[inline]
330    pub const fn is_eof(self) -> bool {
331        matches!(self, TokenKind::EndOfFile)
332    }
333
334    /// Returns `true` if this is a dot token (`.`).
335    #[inline]
336    pub const fn is_dot(self) -> bool {
337        matches!(self, TokenKind::Dot)
338    }
339
340    /// Returns `true` if this is a left brace token (`{`).
341    #[inline]
342    pub const fn is_lbrace(self) -> bool {
343        matches!(self, TokenKind::Lbrace)
344    }
345
346    /// Returns `true` if this is either a newline or non-logical newline token.
347    #[inline]
348    pub const fn is_any_newline(self) -> bool {
349        matches!(self, TokenKind::Newline | TokenKind::NonLogicalNewline)
350    }
351
352    /// Returns `true` if the token is a keyword (including soft keywords).
353    ///
354    /// See also [`is_soft_keyword`], [`is_non_soft_keyword`].
355    ///
356    /// [`is_soft_keyword`]: TokenKind::is_soft_keyword
357    /// [`is_non_soft_keyword`]: TokenKind::is_non_soft_keyword
358    #[inline]
359    pub fn is_keyword(self) -> bool {
360        TokenKind::And <= self && self <= TokenKind::Type
361    }
362
363    /// Returns `true` if the token is strictly a soft keyword.
364    ///
365    /// See also [`is_keyword`], [`is_non_soft_keyword`].
366    ///
367    /// [`is_keyword`]: TokenKind::is_keyword
368    /// [`is_non_soft_keyword`]: TokenKind::is_non_soft_keyword
369    #[inline]
370    pub fn is_soft_keyword(self) -> bool {
371        TokenKind::Case <= self && self <= TokenKind::Type
372    }
373
374    /// Returns `true` if the token is strictly a non-soft keyword.
375    ///
376    /// See also [`is_keyword`], [`is_soft_keyword`].
377    ///
378    /// [`is_keyword`]: TokenKind::is_keyword
379    /// [`is_soft_keyword`]: TokenKind::is_soft_keyword
380    #[inline]
381    pub fn is_non_soft_keyword(self) -> bool {
382        TokenKind::And <= self && self <= TokenKind::Yield
383    }
384
385    #[inline]
386    pub const fn is_operator(self) -> bool {
387        matches!(
388            self,
389            TokenKind::Lpar
390                | TokenKind::Rpar
391                | TokenKind::Lsqb
392                | TokenKind::Rsqb
393                | TokenKind::Comma
394                | TokenKind::Semi
395                | TokenKind::Plus
396                | TokenKind::Minus
397                | TokenKind::Star
398                | TokenKind::Slash
399                | TokenKind::Vbar
400                | TokenKind::Amper
401                | TokenKind::Less
402                | TokenKind::Greater
403                | TokenKind::Equal
404                | TokenKind::Dot
405                | TokenKind::Percent
406                | TokenKind::Lbrace
407                | TokenKind::Rbrace
408                | TokenKind::EqEqual
409                | TokenKind::NotEqual
410                | TokenKind::LessEqual
411                | TokenKind::GreaterEqual
412                | TokenKind::Tilde
413                | TokenKind::CircumFlex
414                | TokenKind::LeftShift
415                | TokenKind::RightShift
416                | TokenKind::DoubleStar
417                | TokenKind::PlusEqual
418                | TokenKind::MinusEqual
419                | TokenKind::StarEqual
420                | TokenKind::SlashEqual
421                | TokenKind::PercentEqual
422                | TokenKind::AmperEqual
423                | TokenKind::VbarEqual
424                | TokenKind::CircumflexEqual
425                | TokenKind::LeftShiftEqual
426                | TokenKind::RightShiftEqual
427                | TokenKind::DoubleStarEqual
428                | TokenKind::DoubleSlash
429                | TokenKind::DoubleSlashEqual
430                | TokenKind::At
431                | TokenKind::AtEqual
432                | TokenKind::Rarrow
433                | TokenKind::Ellipsis
434                | TokenKind::ColonEqual
435                | TokenKind::Colon
436                | TokenKind::And
437                | TokenKind::Or
438                | TokenKind::Not
439                | TokenKind::In
440                | TokenKind::Is
441        )
442    }
443
444    /// Returns `true` if this is a singleton token i.e., `True`, `False`, or `None`.
445    #[inline]
446    pub const fn is_singleton(self) -> bool {
447        matches!(self, TokenKind::False | TokenKind::True | TokenKind::None)
448    }
449
450    /// Returns `true` if this is a trivia token i.e., a comment or a non-logical newline.
451    #[inline]
452    pub const fn is_trivia(&self) -> bool {
453        matches!(self, TokenKind::Comment | TokenKind::NonLogicalNewline)
454    }
455
456    /// Returns `true` if this is a comment token.
457    #[inline]
458    pub const fn is_comment(&self) -> bool {
459        matches!(self, TokenKind::Comment)
460    }
461
462    #[inline]
463    pub const fn is_arithmetic(self) -> bool {
464        matches!(
465            self,
466            TokenKind::DoubleStar
467                | TokenKind::Star
468                | TokenKind::Plus
469                | TokenKind::Minus
470                | TokenKind::Slash
471                | TokenKind::DoubleSlash
472                | TokenKind::At
473        )
474    }
475
476    #[inline]
477    pub const fn is_bitwise_or_shift(self) -> bool {
478        matches!(
479            self,
480            TokenKind::LeftShift
481                | TokenKind::LeftShiftEqual
482                | TokenKind::RightShift
483                | TokenKind::RightShiftEqual
484                | TokenKind::Amper
485                | TokenKind::AmperEqual
486                | TokenKind::Vbar
487                | TokenKind::VbarEqual
488                | TokenKind::CircumFlex
489                | TokenKind::CircumflexEqual
490                | TokenKind::Tilde
491        )
492    }
493
494    /// Returns `true` if the current token is a unary arithmetic operator.
495    #[inline]
496    pub const fn is_unary_arithmetic_operator(self) -> bool {
497        matches!(self, TokenKind::Plus | TokenKind::Minus)
498    }
499
500    #[inline]
501    pub const fn is_interpolated_string_end(self) -> bool {
502        matches!(self, TokenKind::FStringEnd | TokenKind::TStringEnd)
503    }
504
505    /// Returns the [`UnaryOp`] that corresponds to this token kind, if it is a unary arithmetic
506    /// operator, otherwise return [None].
507    ///
508    /// Use [`as_unary_operator`] to match against any unary operator.
509    ///
510    /// [`as_unary_operator`]: TokenKind::as_unary_operator
511    #[inline]
512    pub const fn as_unary_arithmetic_operator(self) -> Option<UnaryOp> {
513        Some(match self {
514            TokenKind::Plus => UnaryOp::UAdd,
515            TokenKind::Minus => UnaryOp::USub,
516            _ => return None,
517        })
518    }
519
520    /// Returns the [`UnaryOp`] that corresponds to this token kind, if it is a unary operator,
521    /// otherwise return [None].
522    ///
523    /// Use [`as_unary_arithmetic_operator`] to match against only an arithmetic unary operator.
524    ///
525    /// [`as_unary_arithmetic_operator`]: TokenKind::as_unary_arithmetic_operator
526    #[inline]
527    pub const fn as_unary_operator(self) -> Option<UnaryOp> {
528        Some(match self {
529            TokenKind::Plus => UnaryOp::UAdd,
530            TokenKind::Minus => UnaryOp::USub,
531            TokenKind::Tilde => UnaryOp::Invert,
532            TokenKind::Not => UnaryOp::Not,
533            _ => return None,
534        })
535    }
536
537    /// Returns the [`BoolOp`] that corresponds to this token kind, if it is a boolean operator,
538    /// otherwise return [None].
539    #[inline]
540    pub const fn as_bool_operator(self) -> Option<BoolOp> {
541        Some(match self {
542            TokenKind::And => BoolOp::And,
543            TokenKind::Or => BoolOp::Or,
544            _ => return None,
545        })
546    }
547
548    /// Returns the binary [`Operator`] that corresponds to the current token, if it's a binary
549    /// operator, otherwise return [None].
550    ///
551    /// Use [`as_augmented_assign_operator`] to match against an augmented assignment token.
552    ///
553    /// [`as_augmented_assign_operator`]: TokenKind::as_augmented_assign_operator
554    pub const fn as_binary_operator(self) -> Option<Operator> {
555        Some(match self {
556            TokenKind::Plus => Operator::Add,
557            TokenKind::Minus => Operator::Sub,
558            TokenKind::Star => Operator::Mult,
559            TokenKind::At => Operator::MatMult,
560            TokenKind::DoubleStar => Operator::Pow,
561            TokenKind::Slash => Operator::Div,
562            TokenKind::DoubleSlash => Operator::FloorDiv,
563            TokenKind::Percent => Operator::Mod,
564            TokenKind::Amper => Operator::BitAnd,
565            TokenKind::Vbar => Operator::BitOr,
566            TokenKind::CircumFlex => Operator::BitXor,
567            TokenKind::LeftShift => Operator::LShift,
568            TokenKind::RightShift => Operator::RShift,
569            _ => return None,
570        })
571    }
572
573    /// Returns the [`Operator`] that corresponds to this token kind, if it is
574    /// an augmented assignment operator, or [`None`] otherwise.
575    #[inline]
576    pub const fn as_augmented_assign_operator(self) -> Option<Operator> {
577        Some(match self {
578            TokenKind::PlusEqual => Operator::Add,
579            TokenKind::MinusEqual => Operator::Sub,
580            TokenKind::StarEqual => Operator::Mult,
581            TokenKind::AtEqual => Operator::MatMult,
582            TokenKind::DoubleStarEqual => Operator::Pow,
583            TokenKind::SlashEqual => Operator::Div,
584            TokenKind::DoubleSlashEqual => Operator::FloorDiv,
585            TokenKind::PercentEqual => Operator::Mod,
586            TokenKind::AmperEqual => Operator::BitAnd,
587            TokenKind::VbarEqual => Operator::BitOr,
588            TokenKind::CircumflexEqual => Operator::BitXor,
589            TokenKind::LeftShiftEqual => Operator::LShift,
590            TokenKind::RightShiftEqual => Operator::RShift,
591            _ => return None,
592        })
593    }
594}
595
596impl From<BoolOp> for TokenKind {
597    #[inline]
598    fn from(op: BoolOp) -> Self {
599        match op {
600            BoolOp::And => TokenKind::And,
601            BoolOp::Or => TokenKind::Or,
602        }
603    }
604}
605
606impl From<UnaryOp> for TokenKind {
607    #[inline]
608    fn from(op: UnaryOp) -> Self {
609        match op {
610            UnaryOp::Invert => TokenKind::Tilde,
611            UnaryOp::Not => TokenKind::Not,
612            UnaryOp::UAdd => TokenKind::Plus,
613            UnaryOp::USub => TokenKind::Minus,
614        }
615    }
616}
617
618impl From<Operator> for TokenKind {
619    #[inline]
620    fn from(op: Operator) -> Self {
621        match op {
622            Operator::Add => TokenKind::Plus,
623            Operator::Sub => TokenKind::Minus,
624            Operator::Mult => TokenKind::Star,
625            Operator::MatMult => TokenKind::At,
626            Operator::Div => TokenKind::Slash,
627            Operator::Mod => TokenKind::Percent,
628            Operator::Pow => TokenKind::DoubleStar,
629            Operator::LShift => TokenKind::LeftShift,
630            Operator::RShift => TokenKind::RightShift,
631            Operator::BitOr => TokenKind::Vbar,
632            Operator::BitXor => TokenKind::CircumFlex,
633            Operator::BitAnd => TokenKind::Amper,
634            Operator::FloorDiv => TokenKind::DoubleSlash,
635        }
636    }
637}
638
639impl fmt::Display for TokenKind {
640    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
641        let value = match self {
642            TokenKind::Unknown => "Unknown",
643            TokenKind::Newline => "newline",
644            TokenKind::NonLogicalNewline => "NonLogicalNewline",
645            TokenKind::Indent => "indent",
646            TokenKind::Dedent => "dedent",
647            TokenKind::EndOfFile => "end of file",
648            TokenKind::Identifier => "identifier",
649            TokenKind::Int => "int",
650            TokenKind::Float => "float",
651            TokenKind::Complex => "complex",
652            TokenKind::String => "string",
653            TokenKind::FStringStart => "FStringStart",
654            TokenKind::FStringMiddle => "FStringMiddle",
655            TokenKind::FStringEnd => "FStringEnd",
656            TokenKind::TStringStart => "TStringStart",
657            TokenKind::TStringMiddle => "TStringMiddle",
658            TokenKind::TStringEnd => "TStringEnd",
659            TokenKind::IpyEscapeCommand => "IPython escape command",
660            TokenKind::Comment => "comment",
661            TokenKind::Question => "`?`",
662            TokenKind::Exclamation => "`!`",
663            TokenKind::Lpar => "`(`",
664            TokenKind::Rpar => "`)`",
665            TokenKind::Lsqb => "`[`",
666            TokenKind::Rsqb => "`]`",
667            TokenKind::Lbrace => "`{`",
668            TokenKind::Rbrace => "`}`",
669            TokenKind::Equal => "`=`",
670            TokenKind::ColonEqual => "`:=`",
671            TokenKind::Dot => "`.`",
672            TokenKind::Colon => "`:`",
673            TokenKind::Semi => "`;`",
674            TokenKind::Comma => "`,`",
675            TokenKind::Rarrow => "`->`",
676            TokenKind::Plus => "`+`",
677            TokenKind::Minus => "`-`",
678            TokenKind::Star => "`*`",
679            TokenKind::DoubleStar => "`**`",
680            TokenKind::Slash => "`/`",
681            TokenKind::DoubleSlash => "`//`",
682            TokenKind::Percent => "`%`",
683            TokenKind::Vbar => "`|`",
684            TokenKind::Amper => "`&`",
685            TokenKind::CircumFlex => "`^`",
686            TokenKind::LeftShift => "`<<`",
687            TokenKind::RightShift => "`>>`",
688            TokenKind::Tilde => "`~`",
689            TokenKind::At => "`@`",
690            TokenKind::Less => "`<`",
691            TokenKind::Greater => "`>`",
692            TokenKind::EqEqual => "`==`",
693            TokenKind::NotEqual => "`!=`",
694            TokenKind::LessEqual => "`<=`",
695            TokenKind::GreaterEqual => "`>=`",
696            TokenKind::PlusEqual => "`+=`",
697            TokenKind::MinusEqual => "`-=`",
698            TokenKind::StarEqual => "`*=`",
699            TokenKind::DoubleStarEqual => "`**=`",
700            TokenKind::SlashEqual => "`/=`",
701            TokenKind::DoubleSlashEqual => "`//=`",
702            TokenKind::PercentEqual => "`%=`",
703            TokenKind::VbarEqual => "`|=`",
704            TokenKind::AmperEqual => "`&=`",
705            TokenKind::CircumflexEqual => "`^=`",
706            TokenKind::LeftShiftEqual => "`<<=`",
707            TokenKind::RightShiftEqual => "`>>=`",
708            TokenKind::AtEqual => "`@=`",
709            TokenKind::Ellipsis => "`...`",
710            TokenKind::False => "`False`",
711            TokenKind::None => "`None`",
712            TokenKind::True => "`True`",
713            TokenKind::And => "`and`",
714            TokenKind::As => "`as`",
715            TokenKind::Assert => "`assert`",
716            TokenKind::Async => "`async`",
717            TokenKind::Await => "`await`",
718            TokenKind::Break => "`break`",
719            TokenKind::Class => "`class`",
720            TokenKind::Continue => "`continue`",
721            TokenKind::Def => "`def`",
722            TokenKind::Del => "`del`",
723            TokenKind::Elif => "`elif`",
724            TokenKind::Else => "`else`",
725            TokenKind::Except => "`except`",
726            TokenKind::Finally => "`finally`",
727            TokenKind::For => "`for`",
728            TokenKind::From => "`from`",
729            TokenKind::Global => "`global`",
730            TokenKind::If => "`if`",
731            TokenKind::Import => "`import`",
732            TokenKind::In => "`in`",
733            TokenKind::Is => "`is`",
734            TokenKind::Lambda => "`lambda`",
735            TokenKind::Nonlocal => "`nonlocal`",
736            TokenKind::Not => "`not`",
737            TokenKind::Or => "`or`",
738            TokenKind::Pass => "`pass`",
739            TokenKind::Raise => "`raise`",
740            TokenKind::Return => "`return`",
741            TokenKind::Try => "`try`",
742            TokenKind::While => "`while`",
743            TokenKind::Lazy => "`lazy`",
744            TokenKind::Match => "`match`",
745            TokenKind::Type => "`type`",
746            TokenKind::Case => "`case`",
747            TokenKind::With => "`with`",
748            TokenKind::Yield => "`yield`",
749        };
750        f.write_str(value)
751    }
752}
753
754bitflags! {
755    #[derive(Clone, Copy, Debug, PartialEq, Eq)]
756    pub struct TokenFlags: u16 {
757        /// The token is a string with double quotes (`"`).
758        const DOUBLE_QUOTES = 1 << 0;
759        /// The token is a triple-quoted string i.e., it starts and ends with three consecutive
760        /// quote characters (`"""` or `'''`).
761        const TRIPLE_QUOTED_STRING = 1 << 1;
762
763        /// The token is a unicode string i.e., prefixed with `u` or `U`
764        const UNICODE_STRING = 1 << 2;
765        /// The token is a byte string i.e., prefixed with `b` or `B`
766        const BYTE_STRING = 1 << 3;
767        /// The token is an f-string i.e., prefixed with `f` or `F`
768        const F_STRING = 1 << 4;
769        /// The token is a t-string i.e., prefixed with `t` or `T`
770        const T_STRING = 1 << 5;
771        /// The token is a raw string and the prefix character is in lowercase.
772        const RAW_STRING_LOWERCASE = 1 << 6;
773        /// The token is a raw string and the prefix character is in uppercase.
774        const RAW_STRING_UPPERCASE = 1 << 7;
775        /// String without matching closing quote(s)
776        const UNCLOSED_STRING = 1 << 8;
777        /// The token is an identifier containing at least one non-ASCII codepoint.
778        const NON_ASCII_IDENTIFIER = 1 << 9;
779
780        /// The token is a raw string i.e., prefixed with `r` or `R`
781        const RAW_STRING = Self::RAW_STRING_LOWERCASE.bits() | Self::RAW_STRING_UPPERCASE.bits();
782
783    }
784}
785
786#[cfg(feature = "get-size")]
787impl get_size2::GetSize for TokenFlags {}
788
789impl StringFlags for TokenFlags {
790    fn quote_style(self) -> Quote {
791        if self.intersects(TokenFlags::DOUBLE_QUOTES) {
792            Quote::Double
793        } else {
794            Quote::Single
795        }
796    }
797
798    fn triple_quotes(self) -> TripleQuotes {
799        if self.intersects(TokenFlags::TRIPLE_QUOTED_STRING) {
800            TripleQuotes::Yes
801        } else {
802            TripleQuotes::No
803        }
804    }
805
806    fn prefix(self) -> AnyStringPrefix {
807        if self.intersects(TokenFlags::F_STRING) {
808            if self.intersects(TokenFlags::RAW_STRING_LOWERCASE) {
809                AnyStringPrefix::Format(FStringPrefix::Raw { uppercase_r: false })
810            } else if self.intersects(TokenFlags::RAW_STRING_UPPERCASE) {
811                AnyStringPrefix::Format(FStringPrefix::Raw { uppercase_r: true })
812            } else {
813                AnyStringPrefix::Format(FStringPrefix::Regular)
814            }
815        } else if self.intersects(TokenFlags::T_STRING) {
816            if self.intersects(TokenFlags::RAW_STRING_LOWERCASE) {
817                AnyStringPrefix::Template(TStringPrefix::Raw { uppercase_r: false })
818            } else if self.intersects(TokenFlags::RAW_STRING_UPPERCASE) {
819                AnyStringPrefix::Template(TStringPrefix::Raw { uppercase_r: true })
820            } else {
821                AnyStringPrefix::Template(TStringPrefix::Regular)
822            }
823        } else if self.intersects(TokenFlags::BYTE_STRING) {
824            if self.intersects(TokenFlags::RAW_STRING_LOWERCASE) {
825                AnyStringPrefix::Bytes(ByteStringPrefix::Raw { uppercase_r: false })
826            } else if self.intersects(TokenFlags::RAW_STRING_UPPERCASE) {
827                AnyStringPrefix::Bytes(ByteStringPrefix::Raw { uppercase_r: true })
828            } else {
829                AnyStringPrefix::Bytes(ByteStringPrefix::Regular)
830            }
831        } else if self.intersects(TokenFlags::RAW_STRING_LOWERCASE) {
832            AnyStringPrefix::Regular(StringLiteralPrefix::Raw { uppercase: false })
833        } else if self.intersects(TokenFlags::RAW_STRING_UPPERCASE) {
834            AnyStringPrefix::Regular(StringLiteralPrefix::Raw { uppercase: true })
835        } else if self.intersects(TokenFlags::UNICODE_STRING) {
836            AnyStringPrefix::Regular(StringLiteralPrefix::Unicode)
837        } else {
838            AnyStringPrefix::Regular(StringLiteralPrefix::Empty)
839        }
840    }
841
842    fn is_unclosed(self) -> bool {
843        self.intersects(TokenFlags::UNCLOSED_STRING)
844    }
845}
846
847impl TokenFlags {
848    /// Returns `true` if the token is an f-string.
849    pub const fn is_f_string(self) -> bool {
850        self.intersects(TokenFlags::F_STRING)
851    }
852
853    /// Returns `true` if the token is a t-string.
854    pub const fn is_t_string(self) -> bool {
855        self.intersects(TokenFlags::T_STRING)
856    }
857
858    /// Returns `true` if the token is a t-string.
859    pub const fn is_interpolated_string(self) -> bool {
860        self.intersects(TokenFlags::T_STRING.union(TokenFlags::F_STRING))
861    }
862
863    /// Returns `true` if the token is a triple-quoted t-string.
864    pub fn is_triple_quoted_interpolated_string(self) -> bool {
865        self.intersects(TokenFlags::TRIPLE_QUOTED_STRING) && self.is_interpolated_string()
866    }
867
868    /// Returns `true` if the token is a raw string.
869    pub const fn is_raw_string(self) -> bool {
870        self.intersects(TokenFlags::RAW_STRING)
871    }
872
873    /// Returns `true` if the token is an identifier containing at least one non-ASCII codepoint.
874    #[inline]
875    pub const fn is_non_ascii_identifier(self) -> bool {
876        self.intersects(TokenFlags::NON_ASCII_IDENTIFIER)
877    }
878}