Skip to main content

ruff_python_parser/
lib.rs

1//! This crate can be used to parse Python source code into an Abstract
2//! Syntax Tree.
3//!
4//! ## Overview
5//!
6//! The process by which source code is parsed into an AST can be broken down
7//! into two general stages: [lexical analysis] and [parsing].
8//!
9//! During lexical analysis, the source code is converted into a stream of lexical
10//! tokens that represent the smallest meaningful units of the language. For example,
11//! the source code `print("Hello world")` would _roughly_ be converted into the following
12//! stream of tokens:
13//!
14//! ```text
15//! Name("print"), LeftParen, String("Hello world"), RightParen
16//! ```
17//!
18//! These tokens are then consumed by the `ruff_python_parser`, which matches them against a set of
19//! grammar rules to verify that the source code is syntactically valid and to construct
20//! an AST that represents the source code.
21//!
22//! During parsing, the `ruff_python_parser` consumes the tokens generated by the lexer and constructs
23//! a tree representation of the source code. The tree is made up of nodes that represent
24//! the different syntactic constructs of the language. If the source code is syntactically
25//! invalid, parsing fails and an error is returned. After a successful parse, the AST can
26//! be used to perform further analysis on the source code. Continuing with the example
27//! above, the AST generated by the `ruff_python_parser` would _roughly_ look something like this:
28//!
29//! ```text
30//! node: Expr {
31//!     value: {
32//!         node: Call {
33//!             func: {
34//!                 node: Name {
35//!                     id: "print",
36//!                     ctx: Load,
37//!                 },
38//!             },
39//!             args: [
40//!                 node: Constant {
41//!                     value: Str("Hello World"),
42//!                     kind: None,
43//!                 },
44//!             ],
45//!             keywords: [],
46//!         },
47//!     },
48//! },
49//!```
50//!
51//! **Note:** The Tokens/ASTs shown above are not the exact tokens/ASTs generated by the `ruff_python_parser`.
52//! Refer to the [playground](https://play.ruff.rs) for the correct representation.
53//!
54//! ## Source code layout
55//!
56//! The functionality of this crate is split into several modules:
57//!
58//! - [lexer]: This module contains the lexer and is responsible for generating the tokens.
59//! - parser: This module contains an interface to the [Parsed] and is responsible for generating the AST.
60//! - mode: This module contains the definition of the different modes that the `ruff_python_parser` can be in.
61//!
62//! [lexical analysis]: https://en.wikipedia.org/wiki/Lexical_analysis
63//! [parsing]: https://en.wikipedia.org/wiki/Parsing
64//! [lexer]: crate::lexer
65
66pub use crate::error::{
67    BlockClause, ExpressionKind, InterpolatedStringErrorType, LexicalErrorType, NumberLiteralKind,
68    ParseError, ParseErrorType, UnicodeEscapeErrorKind, UnsupportedSyntaxError,
69    UnsupportedSyntaxErrorKind,
70};
71pub use crate::parser::ParseOptions;
72
73use crate::parser::Parser;
74
75use ruff_python_ast::token::{Token, TokenFlags, TokenKind, Tokens};
76use ruff_python_ast::{
77    AtomicNodeIndex, Expr, Mod, ModExpression, ModModule, PySourceType, StringFlags, StringLiteral,
78    Suite,
79};
80use ruff_text_size::{Ranged, TextRange};
81
82mod error;
83pub mod lexer;
84mod parser;
85pub mod semantic_errors;
86mod string;
87mod token_set;
88mod token_source;
89pub mod typing;
90
91/// Parse a full Python module usually consisting of multiple lines.
92///
93/// This is a convenience function that can be used to parse a full Python program without having to
94/// specify the [`Mode`] or the location. It is probably what you want to use most of the time.
95///
96/// # Example
97///
98/// For example, parsing a simple function definition and a call to that function:
99///
100/// ```
101/// use ruff_python_parser::parse_module;
102///
103/// let source = r#"
104/// def foo():
105///    return 42
106///
107/// print(foo())
108/// "#;
109///
110/// let module = parse_module(source);
111/// assert!(module.is_ok());
112/// ```
113pub fn parse_module(source: &str) -> Result<Parsed<ModModule>, ParseError> {
114    Parser::new(source, ParseOptions::from(Mode::Module))
115        .parse()
116        .try_into_module()
117        .unwrap()
118        .into_result()
119}
120
121/// Parses a single Python expression.
122///
123/// This convenience function can be used to parse a single expression without having to
124/// specify the Mode or the location.
125///
126/// # Example
127///
128/// For example, parsing a single expression denoting the addition of two numbers:
129///
130/// ```
131/// use ruff_python_parser::parse_expression;
132///
133/// let expr = parse_expression("1 + 2");
134/// assert!(expr.is_ok());
135/// ```
136pub fn parse_expression(source: &str) -> Result<Parsed<ModExpression>, ParseError> {
137    Parser::new(source, ParseOptions::from(Mode::Expression))
138        .parse()
139        .try_into_expression()
140        .unwrap()
141        .into_result()
142}
143
144/// Parses a Python expression for the given range in the source.
145///
146/// This function allows to specify the range of the expression in the source code, other than
147/// that, it behaves exactly like [`parse_expression`].
148///
149/// # Example
150///
151/// Parsing one of the numeric literal which is part of an addition expression:
152///
153/// ```
154/// use ruff_python_parser::parse_expression_range;
155/// # use ruff_text_size::{TextRange, TextSize};
156///
157/// let parsed = parse_expression_range("11 + 22 + 33", TextRange::new(TextSize::new(5), TextSize::new(7)));
158/// assert!(parsed.is_ok());
159/// ```
160pub fn parse_expression_range(
161    source: &str,
162    range: TextRange,
163) -> Result<Parsed<ModExpression>, ParseError> {
164    let source = &source[..range.end().to_usize()];
165    Parser::new_starts_at(source, range.start(), ParseOptions::from(Mode::Expression))
166        .parse()
167        .try_into_expression()
168        .unwrap()
169        .into_result()
170}
171
172/// Parses a Python expression as if it is parenthesized.
173///
174/// It behaves similarly to [`parse_expression_range`] but allows what would be valid within parenthesis
175///
176/// # Example
177///
178/// Parsing an expression that would be valid within parenthesis:
179///
180/// ```
181/// use ruff_python_parser::parse_parenthesized_expression_range;
182/// # use ruff_text_size::{TextRange, TextSize};
183///
184/// let parsed = parse_parenthesized_expression_range("'''\n int | str'''", TextRange::new(TextSize::new(3), TextSize::new(14)));
185/// assert!(parsed.is_ok());
186pub fn parse_parenthesized_expression_range(
187    source: &str,
188    range: TextRange,
189) -> Result<Parsed<ModExpression>, ParseError> {
190    let source = &source[..range.end().to_usize()];
191    let parsed = Parser::new_starts_at(
192        source,
193        range.start(),
194        ParseOptions::from(Mode::ParenthesizedExpression),
195    )
196    .parse();
197    parsed.try_into_expression().unwrap().into_result()
198}
199
200/// Parses a Python expression from a string annotation.
201///
202/// # Example
203///
204/// Parsing a string annotation:
205///
206/// ```
207/// use ruff_python_parser::parse_string_annotation;
208/// use ruff_python_ast::{StringLiteral, StringLiteralFlags, AtomicNodeIndex};
209/// use ruff_text_size::{TextRange, TextSize};
210///
211/// let string = StringLiteral {
212///     value: "'''\n int | str'''".to_string().into_boxed_str(),
213///     flags: StringLiteralFlags::empty(),
214///     range: TextRange::new(TextSize::new(0), TextSize::new(16)),
215///     node_index: AtomicNodeIndex::NONE
216/// };
217/// let parsed = parse_string_annotation("'''\n int | str'''", &string);
218/// assert!(!parsed.is_ok());
219/// ```
220pub fn parse_string_annotation(
221    source: &str,
222    string: &StringLiteral,
223) -> Result<Parsed<ModExpression>, ParseError> {
224    let range = string
225        .range()
226        .add_start(string.flags.opener_len())
227        .sub_end(string.flags.closer_len());
228    let source = &source[..range.end().to_usize()];
229    if string.flags.is_triple_quoted() {
230        parse_parenthesized_expression_range(source, range)
231    } else {
232        parse_expression_range(source, range)
233    }
234}
235
236/// Parse the given Python source code using the specified [`ParseOptions`].
237///
238/// This function is the most general function to parse Python code. Based on the [`Mode`] supplied
239/// via the [`ParseOptions`], it can be used to parse a single expression, a full Python program,
240/// an interactive expression or a Python program containing IPython escape commands.
241///
242/// # Example
243///
244/// If we want to parse a simple expression, we can use the [`Mode::Expression`] mode during
245/// parsing:
246///
247/// ```
248/// use ruff_python_parser::{parse, Mode, ParseOptions};
249///
250/// let parsed = parse("1 + 2", ParseOptions::from(Mode::Expression));
251/// assert!(parsed.is_ok());
252/// ```
253///
254/// Alternatively, we can parse a full Python program consisting of multiple lines:
255///
256/// ```
257/// use ruff_python_parser::{parse, Mode, ParseOptions};
258///
259/// let source = r#"
260/// class Greeter:
261///
262///   def greet(self):
263///    print("Hello, world!")
264/// "#;
265/// let parsed = parse(source, ParseOptions::from(Mode::Module));
266/// assert!(parsed.is_ok());
267/// ```
268///
269/// Additionally, we can parse a Python program containing IPython escapes:
270///
271/// ```
272/// use ruff_python_parser::{parse, Mode, ParseOptions};
273///
274/// let source = r#"
275/// %timeit 1 + 2
276/// ?str.replace
277/// !ls
278/// "#;
279/// let parsed = parse(source, ParseOptions::from(Mode::Ipython));
280/// assert!(parsed.is_ok());
281/// ```
282pub fn parse(source: &str, options: ParseOptions) -> Result<Parsed<Mod>, ParseError> {
283    parse_unchecked(source, options).into_result()
284}
285
286/// Parse the given Python source code using the specified [`ParseOptions`].
287///
288/// This is same as the [`parse`] function except that it doesn't check for any [`ParseError`]
289/// and returns the [`Parsed`] as is.
290pub fn parse_unchecked(source: &str, options: ParseOptions) -> Parsed<Mod> {
291    Parser::new(source, options).parse()
292}
293
294/// Parse the given Python source code using the specified [`PySourceType`].
295pub fn parse_unchecked_source(source: &str, source_type: PySourceType) -> Parsed<ModModule> {
296    // SAFETY: Safe because `PySourceType` always parses to a `ModModule`
297    Parser::new(source, ParseOptions::from(source_type))
298        .parse()
299        .try_into_module()
300        .unwrap()
301}
302
303/// Parses each `range` of `source` as an independent module and concatenates the results into a
304/// single [`Parsed<ModModule>`] whose nodes keep their offsets into `source`.
305///
306/// This validates sources such as Jupyter notebooks, where each cell must be syntactically valid on
307/// its own while later cells can still reference earlier definitions.
308/// The `ranges` must be ordered and non-overlapping.
309///
310/// Consecutive ranges must be separated by a single-byte `\n`: each range ends just before the
311/// separator and the next range starts just after it, as Ruff's notebook cells are. A syntax error
312/// anchored at a cell's trailing offset then lands on that separator, the cell's own last line, so
313/// it is attributed to that cell rather than to the following one.
314pub fn parse_cells_unchecked(
315    source: &str,
316    ranges: impl IntoIterator<Item = TextRange>,
317    options: &ParseOptions,
318) -> Parsed<ModModule> {
319    let mut ranges = ranges.into_iter().peekable();
320    let mut body = Suite::new();
321    let mut tokens = Vec::new();
322    let mut errors = Vec::new();
323    let mut unsupported_syntax_errors = Vec::new();
324    let mut module_range: Option<TextRange> = None;
325
326    while let Some(range) = ranges.next() {
327        if let Some(previous) = module_range {
328            assert!(previous.end() <= range.start());
329        }
330
331        // The cell is lexed from `range.start()`, so the slice must keep the leading text to
332        // preserve absolute offsets into the concatenated source.
333        let cell_source = &source[TextRange::up_to(range.end())];
334        let Parsed {
335            syntax,
336            tokens: cell_tokens,
337            errors: cell_errors,
338            unsupported_syntax_errors: cell_unsupported_syntax_errors,
339        } = Parser::new_starts_at(cell_source, range.start(), options.clone())
340            .parse()
341            .try_into_module()
342            .expect("module options should parse into a module");
343
344        body.extend(syntax.body);
345        tokens.extend(cell_tokens);
346        errors.extend(cell_errors);
347        unsupported_syntax_errors.extend(cell_unsupported_syntax_errors);
348
349        // Each range excludes its trailing `\n` separator (see the doc comment above), leaving a
350        // one-byte gap in the token stream. Cover it with a `NonLogicalNewline` so token-based
351        // checks don't treat the separator as another logical line terminator. The final cell's
352        // separator is the file-final newline and is deliberately left uncovered.
353        if let Some(next) = ranges.peek() {
354            let separator = TextRange::new(range.end(), next.start());
355            assert_eq!(&source[separator], "\n");
356            tokens.push(Token::new(
357                TokenKind::NonLogicalNewline,
358                separator,
359                TokenFlags::empty(),
360            ));
361        }
362
363        module_range = Some(match module_range {
364            Some(previous) => TextRange::new(previous.start(), range.end()),
365            None => range,
366        });
367    }
368
369    body.shrink_to_fit();
370    tokens.shrink_to_fit();
371    errors.shrink_to_fit();
372    unsupported_syntax_errors.shrink_to_fit();
373
374    Parsed {
375        syntax: ModModule {
376            node_index: AtomicNodeIndex::NONE,
377            range: module_range.unwrap_or_default(),
378            body,
379            runtime_body: None,
380        },
381        tokens: Tokens::new(tokens),
382        errors,
383        unsupported_syntax_errors,
384    }
385}
386
387/// Represents the parsed source code.
388#[derive(Debug, PartialEq, Clone, get_size2::GetSize)]
389pub struct Parsed<T> {
390    syntax: T,
391    tokens: Tokens,
392    errors: Vec<ParseError>,
393    unsupported_syntax_errors: Vec<UnsupportedSyntaxError>,
394}
395
396impl<T> Parsed<T> {
397    /// Returns the syntax node represented by this parsed output.
398    pub fn syntax(&self) -> &T {
399        &self.syntax
400    }
401
402    /// Returns all the tokens for the parsed output.
403    pub fn tokens(&self) -> &Tokens {
404        &self.tokens
405    }
406
407    /// Returns a list of syntax errors found during parsing.
408    pub fn errors(&self) -> &[ParseError] {
409        &self.errors
410    }
411
412    /// Returns a list of version-related syntax errors found during parsing.
413    pub fn unsupported_syntax_errors(&self) -> &[UnsupportedSyntaxError] {
414        &self.unsupported_syntax_errors
415    }
416
417    /// Consumes the [`Parsed`] output and returns the contained syntax node.
418    pub fn into_syntax(self) -> T {
419        self.syntax
420    }
421
422    /// Consumes the [`Parsed`] output and returns a list of syntax errors found during parsing.
423    fn into_errors(self) -> Vec<ParseError> {
424        self.errors
425    }
426
427    /// Returns `true` if the parsed source code is valid i.e., it has no [`ParseError`]s.
428    ///
429    /// Note that this does not include version-related [`UnsupportedSyntaxError`]s.
430    ///
431    /// See [`Parsed::has_no_syntax_errors`] for a version that takes these into account.
432    pub fn has_valid_syntax(&self) -> bool {
433        self.errors.is_empty()
434    }
435
436    /// Returns `true` if the parsed source code is invalid i.e., it has [`ParseError`]s.
437    ///
438    /// Note that this does not include version-related [`UnsupportedSyntaxError`]s.
439    ///
440    /// See [`Parsed::has_no_syntax_errors`] for a version that takes these into account.
441    pub fn has_invalid_syntax(&self) -> bool {
442        !self.has_valid_syntax()
443    }
444
445    /// Returns `true` if the parsed source code does not contain any [`ParseError`]s *or*
446    /// [`UnsupportedSyntaxError`]s.
447    ///
448    /// See [`Parsed::has_valid_syntax`] for a version specific to [`ParseError`]s.
449    pub fn has_no_syntax_errors(&self) -> bool {
450        self.has_valid_syntax() && self.unsupported_syntax_errors.is_empty()
451    }
452
453    /// Returns `true` if the parsed source code contains any [`ParseError`]s *or*
454    /// [`UnsupportedSyntaxError`]s.
455    ///
456    /// See [`Parsed::has_invalid_syntax`] for a version specific to [`ParseError`]s.
457    pub fn has_syntax_errors(&self) -> bool {
458        !self.has_no_syntax_errors()
459    }
460
461    /// Returns the [`Parsed`] output as a [`Result`], returning [`Ok`] if it has no syntax errors,
462    /// or [`Err`] containing the first [`ParseError`] encountered.
463    ///
464    /// Note that any [`unsupported_syntax_errors`](Parsed::unsupported_syntax_errors) will not
465    /// cause [`Err`] to be returned.
466    pub fn as_result(&self) -> Result<&Parsed<T>, &[ParseError]> {
467        if self.has_valid_syntax() {
468            Ok(self)
469        } else {
470            Err(&self.errors)
471        }
472    }
473
474    /// Consumes the [`Parsed`] output and returns a [`Result`] which is [`Ok`] if it has no syntax
475    /// errors, or [`Err`] containing the first [`ParseError`] encountered.
476    ///
477    /// Note that any [`unsupported_syntax_errors`](Parsed::unsupported_syntax_errors) will not
478    /// cause [`Err`] to be returned.
479    fn into_result(self) -> Result<Parsed<T>, ParseError> {
480        if self.has_valid_syntax() {
481            Ok(self)
482        } else {
483            Err(self.into_errors().into_iter().next().unwrap())
484        }
485    }
486}
487
488impl Parsed<Mod> {
489    /// Attempts to convert the [`Parsed<Mod>`] into a [`Parsed<ModModule>`].
490    ///
491    /// This method checks if the `syntax` field of the output is a [`Mod::Module`]. If it is, the
492    /// method returns [`Some(Parsed<ModModule>)`] with the contained module. Otherwise, it
493    /// returns [`None`].
494    ///
495    /// [`Some(Parsed<ModModule>)`]: Some
496    pub fn try_into_module(self) -> Option<Parsed<ModModule>> {
497        match self.syntax {
498            Mod::Module(module) => Some(Parsed {
499                syntax: module,
500                tokens: self.tokens,
501                errors: self.errors,
502                unsupported_syntax_errors: self.unsupported_syntax_errors,
503            }),
504            Mod::Expression(_) => None,
505        }
506    }
507
508    /// Attempts to convert the [`Parsed<Mod>`] into a [`Parsed<ModExpression>`].
509    ///
510    /// This method checks if the `syntax` field of the output is a [`Mod::Expression`]. If it is,
511    /// the method returns [`Some(Parsed<ModExpression>)`] with the contained expression.
512    /// Otherwise, it returns [`None`].
513    ///
514    /// [`Some(Parsed<ModExpression>)`]: Some
515    fn try_into_expression(self) -> Option<Parsed<ModExpression>> {
516        match self.syntax {
517            Mod::Module(_) => None,
518            Mod::Expression(expression) => Some(Parsed {
519                syntax: expression,
520                tokens: self.tokens,
521                errors: self.errors,
522                unsupported_syntax_errors: self.unsupported_syntax_errors,
523            }),
524        }
525    }
526}
527
528impl Parsed<ModModule> {
529    /// Returns the module body contained in this parsed output as a [`Suite`].
530    pub fn suite(&self) -> &Suite {
531        &self.syntax.body
532    }
533
534    /// Consumes the [`Parsed`] output and returns the module body as a [`Suite`].
535    pub fn into_suite(self) -> Suite {
536        self.syntax.body
537    }
538}
539
540impl Parsed<ModExpression> {
541    /// Returns the expression contained in this parsed output.
542    pub fn expr(&self) -> &Expr {
543        &self.syntax.body
544    }
545
546    /// Returns a mutable reference to the expression contained in this parsed output.
547    fn expr_mut(&mut self) -> &mut Expr {
548        &mut self.syntax.body
549    }
550
551    /// Consumes the [`Parsed`] output and returns the contained [`Expr`].
552    pub fn into_expr(self) -> Expr {
553        *self.syntax.body
554    }
555}
556
557/// Control in the different modes by which a source file can be parsed.
558///
559/// The mode argument specifies in what way code must be parsed.
560#[derive(Clone, Copy, Debug, Hash, PartialEq, Eq)]
561pub enum Mode {
562    /// The code consists of a sequence of statements.
563    Module,
564
565    /// The code consists of a single expression.
566    Expression,
567
568    /// The code consists of a single expression and is parsed as if it is parenthesized. The parentheses themselves aren't required.
569    /// This allows for having valid multiline expression without the need of parentheses
570    /// and is specifically useful for parsing string annotations.
571    ParenthesizedExpression,
572
573    /// The code consists of a sequence of statements which can include the
574    /// escape commands that are part of IPython syntax.
575    ///
576    /// ## Supported escape commands:
577    ///
578    /// - [Magic command system] which is limited to [line magics] and can start
579    ///   with `?` or `??`.
580    /// - [Dynamic object information] which can start with `?` or `??`.
581    /// - [System shell access] which can start with `!` or `!!`.
582    /// - [Automatic parentheses and quotes] which can start with `/`, `;`, or `,`.
583    ///
584    /// [Magic command system]: https://ipython.readthedocs.io/en/stable/interactive/reference.html#magic-command-system
585    /// [line magics]: https://ipython.readthedocs.io/en/stable/interactive/magics.html#line-magics
586    /// [Dynamic object information]: https://ipython.readthedocs.io/en/stable/interactive/reference.html#dynamic-object-information
587    /// [System shell access]: https://ipython.readthedocs.io/en/stable/interactive/reference.html#system-shell-access
588    /// [Automatic parentheses and quotes]: https://ipython.readthedocs.io/en/stable/interactive/reference.html#automatic-parentheses-and-quotes
589    Ipython,
590}
591
592impl std::str::FromStr for Mode {
593    type Err = ModeParseError;
594    fn from_str(s: &str) -> Result<Self, ModeParseError> {
595        match s {
596            "exec" | "single" => Ok(Mode::Module),
597            "eval" => Ok(Mode::Expression),
598            "ipython" => Ok(Mode::Ipython),
599            _ => Err(ModeParseError),
600        }
601    }
602}
603
604/// A type that can be represented as [Mode].
605pub trait AsMode {
606    fn as_mode(&self) -> Mode;
607}
608
609impl AsMode for PySourceType {
610    fn as_mode(&self) -> Mode {
611        match self {
612            PySourceType::Python | PySourceType::Stub => Mode::Module,
613            PySourceType::Ipynb => Mode::Ipython,
614        }
615    }
616}
617
618/// Returned when a given mode is not valid.
619#[derive(Debug)]
620pub struct ModeParseError;
621
622impl std::fmt::Display for ModeParseError {
623    fn fmt(&self, f: &mut std::fmt::Formatter) -> std::fmt::Result {
624        write!(f, r#"mode must be "exec", "eval", "ipython", or "single""#)
625    }
626}