Skip to main content

ruff_python_parser/
lib.rs

1//! This crate can be used to parse Python source code into an Abstract
2//! Syntax Tree.
3//!
4//! ## Overview
5//!
6//! The process by which source code is parsed into an AST can be broken down
7//! into two general stages: [lexical analysis] and [parsing].
8//!
9//! During lexical analysis, the source code is converted into a stream of lexical
10//! tokens that represent the smallest meaningful units of the language. For example,
11//! the source code `print("Hello world")` would _roughly_ be converted into the following
12//! stream of tokens:
13//!
14//! ```text
15//! Name("print"), LeftParen, String("Hello world"), RightParen
16//! ```
17//!
18//! These tokens are then consumed by the `ruff_python_parser`, which matches them against a set of
19//! grammar rules to verify that the source code is syntactically valid and to construct
20//! an AST that represents the source code.
21//!
22//! During parsing, the `ruff_python_parser` consumes the tokens generated by the lexer and constructs
23//! a tree representation of the source code. The tree is made up of nodes that represent
24//! the different syntactic constructs of the language. If the source code is syntactically
25//! invalid, parsing fails and an error is returned. After a successful parse, the AST can
26//! be used to perform further analysis on the source code. Continuing with the example
27//! above, the AST generated by the `ruff_python_parser` would _roughly_ look something like this:
28//!
29//! ```text
30//! node: Expr {
31//!     value: {
32//!         node: Call {
33//!             func: {
34//!                 node: Name {
35//!                     id: "print",
36//!                     ctx: Load,
37//!                 },
38//!             },
39//!             args: [
40//!                 node: Constant {
41//!                     value: Str("Hello World"),
42//!                     kind: None,
43//!                 },
44//!             ],
45//!             keywords: [],
46//!         },
47//!     },
48//! },
49//!```
50//!
51//! **Note:** The Tokens/ASTs shown above are not the exact tokens/ASTs generated by the `ruff_python_parser`.
52//! Refer to the [playground](https://play.ruff.rs) for the correct representation.
53//!
54//! ## Source code layout
55//!
56//! The functionality of this crate is split into several modules:
57//!
58//! - [lexer]: This module contains the lexer and is responsible for generating the tokens.
59//! - parser: This module contains an interface to the [Parsed] and is responsible for generating the AST.
60//! - mode: This module contains the definition of the different modes that the `ruff_python_parser` can be in.
61//!
62//! [lexical analysis]: https://en.wikipedia.org/wiki/Lexical_analysis
63//! [parsing]: https://en.wikipedia.org/wiki/Parsing
64//! [lexer]: crate::lexer
65
66pub use crate::error::{
67    InterpolatedStringErrorType, LexicalErrorType, ParseError, ParseErrorType,
68    UnsupportedSyntaxError, UnsupportedSyntaxErrorKind,
69};
70pub use crate::parser::ParseOptions;
71
72use crate::parser::Parser;
73
74use ruff_python_ast::token::{Token, TokenFlags, TokenKind, Tokens};
75use ruff_python_ast::{
76    AtomicNodeIndex, Expr, Mod, ModExpression, ModModule, PySourceType, StringFlags, StringLiteral,
77    Suite,
78};
79use ruff_text_size::{Ranged, TextRange};
80
81mod error;
82pub mod lexer;
83mod parser;
84pub mod semantic_errors;
85mod string;
86mod token_set;
87mod token_source;
88pub mod typing;
89
90/// Parse a full Python module usually consisting of multiple lines.
91///
92/// This is a convenience function that can be used to parse a full Python program without having to
93/// specify the [`Mode`] or the location. It is probably what you want to use most of the time.
94///
95/// # Example
96///
97/// For example, parsing a simple function definition and a call to that function:
98///
99/// ```
100/// use ruff_python_parser::parse_module;
101///
102/// let source = r#"
103/// def foo():
104///    return 42
105///
106/// print(foo())
107/// "#;
108///
109/// let module = parse_module(source);
110/// assert!(module.is_ok());
111/// ```
112pub fn parse_module(source: &str) -> Result<Parsed<ModModule>, ParseError> {
113    Parser::new(source, ParseOptions::from(Mode::Module))
114        .parse()
115        .try_into_module()
116        .unwrap()
117        .into_result()
118}
119
120/// Parses a single Python expression.
121///
122/// This convenience function can be used to parse a single expression without having to
123/// specify the Mode or the location.
124///
125/// # Example
126///
127/// For example, parsing a single expression denoting the addition of two numbers:
128///
129/// ```
130/// use ruff_python_parser::parse_expression;
131///
132/// let expr = parse_expression("1 + 2");
133/// assert!(expr.is_ok());
134/// ```
135pub fn parse_expression(source: &str) -> Result<Parsed<ModExpression>, ParseError> {
136    Parser::new(source, ParseOptions::from(Mode::Expression))
137        .parse()
138        .try_into_expression()
139        .unwrap()
140        .into_result()
141}
142
143/// Parses a Python expression for the given range in the source.
144///
145/// This function allows to specify the range of the expression in the source code, other than
146/// that, it behaves exactly like [`parse_expression`].
147///
148/// # Example
149///
150/// Parsing one of the numeric literal which is part of an addition expression:
151///
152/// ```
153/// use ruff_python_parser::parse_expression_range;
154/// # use ruff_text_size::{TextRange, TextSize};
155///
156/// let parsed = parse_expression_range("11 + 22 + 33", TextRange::new(TextSize::new(5), TextSize::new(7)));
157/// assert!(parsed.is_ok());
158/// ```
159pub fn parse_expression_range(
160    source: &str,
161    range: TextRange,
162) -> Result<Parsed<ModExpression>, ParseError> {
163    let source = &source[..range.end().to_usize()];
164    Parser::new_starts_at(source, range.start(), ParseOptions::from(Mode::Expression))
165        .parse()
166        .try_into_expression()
167        .unwrap()
168        .into_result()
169}
170
171/// Parses a Python expression as if it is parenthesized.
172///
173/// It behaves similarly to [`parse_expression_range`] but allows what would be valid within parenthesis
174///
175/// # Example
176///
177/// Parsing an expression that would be valid within parenthesis:
178///
179/// ```
180/// use ruff_python_parser::parse_parenthesized_expression_range;
181/// # use ruff_text_size::{TextRange, TextSize};
182///
183/// let parsed = parse_parenthesized_expression_range("'''\n int | str'''", TextRange::new(TextSize::new(3), TextSize::new(14)));
184/// assert!(parsed.is_ok());
185pub fn parse_parenthesized_expression_range(
186    source: &str,
187    range: TextRange,
188) -> Result<Parsed<ModExpression>, ParseError> {
189    let source = &source[..range.end().to_usize()];
190    let parsed = Parser::new_starts_at(
191        source,
192        range.start(),
193        ParseOptions::from(Mode::ParenthesizedExpression),
194    )
195    .parse();
196    parsed.try_into_expression().unwrap().into_result()
197}
198
199/// Parses a Python expression from a string annotation.
200///
201/// # Example
202///
203/// Parsing a string annotation:
204///
205/// ```
206/// use ruff_python_parser::parse_string_annotation;
207/// use ruff_python_ast::{StringLiteral, StringLiteralFlags, AtomicNodeIndex};
208/// use ruff_text_size::{TextRange, TextSize};
209///
210/// let string = StringLiteral {
211///     value: "'''\n int | str'''".to_string().into_boxed_str(),
212///     flags: StringLiteralFlags::empty(),
213///     range: TextRange::new(TextSize::new(0), TextSize::new(16)),
214///     node_index: AtomicNodeIndex::NONE
215/// };
216/// let parsed = parse_string_annotation("'''\n int | str'''", &string);
217/// assert!(!parsed.is_ok());
218/// ```
219pub fn parse_string_annotation(
220    source: &str,
221    string: &StringLiteral,
222) -> Result<Parsed<ModExpression>, ParseError> {
223    let range = string
224        .range()
225        .add_start(string.flags.opener_len())
226        .sub_end(string.flags.closer_len());
227    let source = &source[..range.end().to_usize()];
228    if string.flags.is_triple_quoted() {
229        parse_parenthesized_expression_range(source, range)
230    } else {
231        parse_expression_range(source, range)
232    }
233}
234
235/// Parse the given Python source code using the specified [`ParseOptions`].
236///
237/// This function is the most general function to parse Python code. Based on the [`Mode`] supplied
238/// via the [`ParseOptions`], it can be used to parse a single expression, a full Python program,
239/// an interactive expression or a Python program containing IPython escape commands.
240///
241/// # Example
242///
243/// If we want to parse a simple expression, we can use the [`Mode::Expression`] mode during
244/// parsing:
245///
246/// ```
247/// use ruff_python_parser::{parse, Mode, ParseOptions};
248///
249/// let parsed = parse("1 + 2", ParseOptions::from(Mode::Expression));
250/// assert!(parsed.is_ok());
251/// ```
252///
253/// Alternatively, we can parse a full Python program consisting of multiple lines:
254///
255/// ```
256/// use ruff_python_parser::{parse, Mode, ParseOptions};
257///
258/// let source = r#"
259/// class Greeter:
260///
261///   def greet(self):
262///    print("Hello, world!")
263/// "#;
264/// let parsed = parse(source, ParseOptions::from(Mode::Module));
265/// assert!(parsed.is_ok());
266/// ```
267///
268/// Additionally, we can parse a Python program containing IPython escapes:
269///
270/// ```
271/// use ruff_python_parser::{parse, Mode, ParseOptions};
272///
273/// let source = r#"
274/// %timeit 1 + 2
275/// ?str.replace
276/// !ls
277/// "#;
278/// let parsed = parse(source, ParseOptions::from(Mode::Ipython));
279/// assert!(parsed.is_ok());
280/// ```
281pub fn parse(source: &str, options: ParseOptions) -> Result<Parsed<Mod>, ParseError> {
282    parse_unchecked(source, options).into_result()
283}
284
285/// Parse the given Python source code using the specified [`ParseOptions`].
286///
287/// This is same as the [`parse`] function except that it doesn't check for any [`ParseError`]
288/// and returns the [`Parsed`] as is.
289pub fn parse_unchecked(source: &str, options: ParseOptions) -> Parsed<Mod> {
290    Parser::new(source, options).parse()
291}
292
293/// Parse the given Python source code using the specified [`PySourceType`].
294pub fn parse_unchecked_source(source: &str, source_type: PySourceType) -> Parsed<ModModule> {
295    // SAFETY: Safe because `PySourceType` always parses to a `ModModule`
296    Parser::new(source, ParseOptions::from(source_type))
297        .parse()
298        .try_into_module()
299        .unwrap()
300}
301
302/// Parses each `range` of `source` as an independent module and concatenates the results into a
303/// single [`Parsed<ModModule>`] whose nodes keep their offsets into `source`.
304///
305/// This validates sources such as Jupyter notebooks, where each cell must be syntactically valid on
306/// its own while later cells can still reference earlier definitions.
307/// The `ranges` must be ordered and non-overlapping.
308///
309/// Consecutive ranges must be separated by a single-byte `\n`: each range ends just before the
310/// separator and the next range starts just after it, as Ruff's notebook cells are. A syntax error
311/// anchored at a cell's trailing offset then lands on that separator, the cell's own last line, so
312/// it is attributed to that cell rather than to the following one.
313pub fn parse_cells_unchecked(
314    source: &str,
315    ranges: impl IntoIterator<Item = TextRange>,
316    options: &ParseOptions,
317) -> Parsed<ModModule> {
318    let mut ranges = ranges.into_iter().peekable();
319    let mut body = Suite::new();
320    let mut tokens = Vec::new();
321    let mut errors = Vec::new();
322    let mut unsupported_syntax_errors = Vec::new();
323    let mut module_range: Option<TextRange> = None;
324
325    while let Some(range) = ranges.next() {
326        if let Some(previous) = module_range {
327            assert!(previous.end() <= range.start());
328        }
329
330        // The cell is lexed from `range.start()`, so the slice must keep the leading text to
331        // preserve absolute offsets into the concatenated source.
332        let cell_source = &source[TextRange::up_to(range.end())];
333        let Parsed {
334            syntax,
335            tokens: cell_tokens,
336            errors: cell_errors,
337            unsupported_syntax_errors: cell_unsupported_syntax_errors,
338        } = Parser::new_starts_at(cell_source, range.start(), options.clone())
339            .parse()
340            .try_into_module()
341            .expect("module options should parse into a module");
342
343        body.extend(syntax.body);
344        tokens.extend(cell_tokens);
345        errors.extend(cell_errors);
346        unsupported_syntax_errors.extend(cell_unsupported_syntax_errors);
347
348        // Each range excludes its trailing `\n` separator (see the doc comment above), leaving a
349        // one-byte gap in the token stream. Cover it with a `NonLogicalNewline` so token-based
350        // checks don't treat the separator as another logical line terminator. The final cell's
351        // separator is the file-final newline and is deliberately left uncovered.
352        if let Some(next) = ranges.peek() {
353            let separator = TextRange::new(range.end(), next.start());
354            assert_eq!(&source[separator], "\n");
355            tokens.push(Token::new(
356                TokenKind::NonLogicalNewline,
357                separator,
358                TokenFlags::empty(),
359            ));
360        }
361
362        module_range = Some(match module_range {
363            Some(previous) => TextRange::new(previous.start(), range.end()),
364            None => range,
365        });
366    }
367
368    body.shrink_to_fit();
369    tokens.shrink_to_fit();
370    errors.shrink_to_fit();
371    unsupported_syntax_errors.shrink_to_fit();
372
373    Parsed {
374        syntax: ModModule {
375            node_index: AtomicNodeIndex::NONE,
376            range: module_range.unwrap_or_default(),
377            body,
378            runtime_body: None,
379        },
380        tokens: Tokens::new(tokens),
381        errors,
382        unsupported_syntax_errors,
383    }
384}
385
386/// Represents the parsed source code.
387#[derive(Debug, PartialEq, Clone, get_size2::GetSize)]
388pub struct Parsed<T> {
389    syntax: T,
390    tokens: Tokens,
391    errors: Vec<ParseError>,
392    unsupported_syntax_errors: Vec<UnsupportedSyntaxError>,
393}
394
395impl<T> Parsed<T> {
396    /// Returns the syntax node represented by this parsed output.
397    pub fn syntax(&self) -> &T {
398        &self.syntax
399    }
400
401    /// Returns all the tokens for the parsed output.
402    pub fn tokens(&self) -> &Tokens {
403        &self.tokens
404    }
405
406    /// Returns a list of syntax errors found during parsing.
407    pub fn errors(&self) -> &[ParseError] {
408        &self.errors
409    }
410
411    /// Returns a list of version-related syntax errors found during parsing.
412    pub fn unsupported_syntax_errors(&self) -> &[UnsupportedSyntaxError] {
413        &self.unsupported_syntax_errors
414    }
415
416    /// Consumes the [`Parsed`] output and returns the contained syntax node.
417    pub fn into_syntax(self) -> T {
418        self.syntax
419    }
420
421    /// Consumes the [`Parsed`] output and returns a list of syntax errors found during parsing.
422    fn into_errors(self) -> Vec<ParseError> {
423        self.errors
424    }
425
426    /// Returns `true` if the parsed source code is valid i.e., it has no [`ParseError`]s.
427    ///
428    /// Note that this does not include version-related [`UnsupportedSyntaxError`]s.
429    ///
430    /// See [`Parsed::has_no_syntax_errors`] for a version that takes these into account.
431    pub fn has_valid_syntax(&self) -> bool {
432        self.errors.is_empty()
433    }
434
435    /// Returns `true` if the parsed source code is invalid i.e., it has [`ParseError`]s.
436    ///
437    /// Note that this does not include version-related [`UnsupportedSyntaxError`]s.
438    ///
439    /// See [`Parsed::has_no_syntax_errors`] for a version that takes these into account.
440    pub fn has_invalid_syntax(&self) -> bool {
441        !self.has_valid_syntax()
442    }
443
444    /// Returns `true` if the parsed source code does not contain any [`ParseError`]s *or*
445    /// [`UnsupportedSyntaxError`]s.
446    ///
447    /// See [`Parsed::has_valid_syntax`] for a version specific to [`ParseError`]s.
448    pub fn has_no_syntax_errors(&self) -> bool {
449        self.has_valid_syntax() && self.unsupported_syntax_errors.is_empty()
450    }
451
452    /// Returns `true` if the parsed source code contains any [`ParseError`]s *or*
453    /// [`UnsupportedSyntaxError`]s.
454    ///
455    /// See [`Parsed::has_invalid_syntax`] for a version specific to [`ParseError`]s.
456    pub fn has_syntax_errors(&self) -> bool {
457        !self.has_no_syntax_errors()
458    }
459
460    /// Returns the [`Parsed`] output as a [`Result`], returning [`Ok`] if it has no syntax errors,
461    /// or [`Err`] containing the first [`ParseError`] encountered.
462    ///
463    /// Note that any [`unsupported_syntax_errors`](Parsed::unsupported_syntax_errors) will not
464    /// cause [`Err`] to be returned.
465    pub fn as_result(&self) -> Result<&Parsed<T>, &[ParseError]> {
466        if self.has_valid_syntax() {
467            Ok(self)
468        } else {
469            Err(&self.errors)
470        }
471    }
472
473    /// Consumes the [`Parsed`] output and returns a [`Result`] which is [`Ok`] if it has no syntax
474    /// errors, or [`Err`] containing the first [`ParseError`] encountered.
475    ///
476    /// Note that any [`unsupported_syntax_errors`](Parsed::unsupported_syntax_errors) will not
477    /// cause [`Err`] to be returned.
478    fn into_result(self) -> Result<Parsed<T>, ParseError> {
479        if self.has_valid_syntax() {
480            Ok(self)
481        } else {
482            Err(self.into_errors().into_iter().next().unwrap())
483        }
484    }
485}
486
487impl Parsed<Mod> {
488    /// Attempts to convert the [`Parsed<Mod>`] into a [`Parsed<ModModule>`].
489    ///
490    /// This method checks if the `syntax` field of the output is a [`Mod::Module`]. If it is, the
491    /// method returns [`Some(Parsed<ModModule>)`] with the contained module. Otherwise, it
492    /// returns [`None`].
493    ///
494    /// [`Some(Parsed<ModModule>)`]: Some
495    pub fn try_into_module(self) -> Option<Parsed<ModModule>> {
496        match self.syntax {
497            Mod::Module(module) => Some(Parsed {
498                syntax: module,
499                tokens: self.tokens,
500                errors: self.errors,
501                unsupported_syntax_errors: self.unsupported_syntax_errors,
502            }),
503            Mod::Expression(_) => None,
504        }
505    }
506
507    /// Attempts to convert the [`Parsed<Mod>`] into a [`Parsed<ModExpression>`].
508    ///
509    /// This method checks if the `syntax` field of the output is a [`Mod::Expression`]. If it is,
510    /// the method returns [`Some(Parsed<ModExpression>)`] with the contained expression.
511    /// Otherwise, it returns [`None`].
512    ///
513    /// [`Some(Parsed<ModExpression>)`]: Some
514    fn try_into_expression(self) -> Option<Parsed<ModExpression>> {
515        match self.syntax {
516            Mod::Module(_) => None,
517            Mod::Expression(expression) => Some(Parsed {
518                syntax: expression,
519                tokens: self.tokens,
520                errors: self.errors,
521                unsupported_syntax_errors: self.unsupported_syntax_errors,
522            }),
523        }
524    }
525}
526
527impl Parsed<ModModule> {
528    /// Returns the module body contained in this parsed output as a [`Suite`].
529    pub fn suite(&self) -> &Suite {
530        &self.syntax.body
531    }
532
533    /// Consumes the [`Parsed`] output and returns the module body as a [`Suite`].
534    pub fn into_suite(self) -> Suite {
535        self.syntax.body
536    }
537}
538
539impl Parsed<ModExpression> {
540    /// Returns the expression contained in this parsed output.
541    pub fn expr(&self) -> &Expr {
542        &self.syntax.body
543    }
544
545    /// Returns a mutable reference to the expression contained in this parsed output.
546    fn expr_mut(&mut self) -> &mut Expr {
547        &mut self.syntax.body
548    }
549
550    /// Consumes the [`Parsed`] output and returns the contained [`Expr`].
551    pub fn into_expr(self) -> Expr {
552        *self.syntax.body
553    }
554}
555
556/// Control in the different modes by which a source file can be parsed.
557///
558/// The mode argument specifies in what way code must be parsed.
559#[derive(Clone, Copy, Debug, Hash, PartialEq, Eq)]
560pub enum Mode {
561    /// The code consists of a sequence of statements.
562    Module,
563
564    /// The code consists of a single expression.
565    Expression,
566
567    /// The code consists of a single expression and is parsed as if it is parenthesized. The parentheses themselves aren't required.
568    /// This allows for having valid multiline expression without the need of parentheses
569    /// and is specifically useful for parsing string annotations.
570    ParenthesizedExpression,
571
572    /// The code consists of a sequence of statements which can include the
573    /// escape commands that are part of IPython syntax.
574    ///
575    /// ## Supported escape commands:
576    ///
577    /// - [Magic command system] which is limited to [line magics] and can start
578    ///   with `?` or `??`.
579    /// - [Dynamic object information] which can start with `?` or `??`.
580    /// - [System shell access] which can start with `!` or `!!`.
581    /// - [Automatic parentheses and quotes] which can start with `/`, `;`, or `,`.
582    ///
583    /// [Magic command system]: https://ipython.readthedocs.io/en/stable/interactive/reference.html#magic-command-system
584    /// [line magics]: https://ipython.readthedocs.io/en/stable/interactive/magics.html#line-magics
585    /// [Dynamic object information]: https://ipython.readthedocs.io/en/stable/interactive/reference.html#dynamic-object-information
586    /// [System shell access]: https://ipython.readthedocs.io/en/stable/interactive/reference.html#system-shell-access
587    /// [Automatic parentheses and quotes]: https://ipython.readthedocs.io/en/stable/interactive/reference.html#automatic-parentheses-and-quotes
588    Ipython,
589}
590
591impl std::str::FromStr for Mode {
592    type Err = ModeParseError;
593    fn from_str(s: &str) -> Result<Self, ModeParseError> {
594        match s {
595            "exec" | "single" => Ok(Mode::Module),
596            "eval" => Ok(Mode::Expression),
597            "ipython" => Ok(Mode::Ipython),
598            _ => Err(ModeParseError),
599        }
600    }
601}
602
603/// A type that can be represented as [Mode].
604pub trait AsMode {
605    fn as_mode(&self) -> Mode;
606}
607
608impl AsMode for PySourceType {
609    fn as_mode(&self) -> Mode {
610        match self {
611            PySourceType::Python | PySourceType::Stub => Mode::Module,
612            PySourceType::Ipynb => Mode::Ipython,
613        }
614    }
615}
616
617/// Returned when a given mode is not valid.
618#[derive(Debug)]
619pub struct ModeParseError;
620
621impl std::fmt::Display for ModeParseError {
622    fn fmt(&self, f: &mut std::fmt::Formatter) -> std::fmt::Result {
623        write!(f, r#"mode must be "exec", "eval", "ipython", or "single""#)
624    }
625}