Skip to main content

rustpython_compiler/
lib.rs

1extern crate alloc;
2
3use alloc::borrow::Cow;
4pub use ruff_python_ast::token::{TokenKind, Tokens};
5use ruff_python_parser::ParseErrorType;
6use ruff_source_file::{PositionEncoding, SourceFile, SourceFileBuilder, SourceLocation};
7use ruff_text_size::{Ranged, TextSize, TextSlice};
8use rustpython_codegen::{compile, symboltable};
9use thiserror::Error;
10
11pub use rustpython_codegen::compile::CompileOpts;
12pub use rustpython_compiler_core::{Mode, bytecode::CodeObject};
13
14// these modules are out of repository. re-exporting them here for convenience.
15pub use ruff_python_ast as ast;
16pub use ruff_python_parser as parser;
17pub use rustpython_codegen as codegen;
18pub use rustpython_compiler_core as core;
19
20#[derive(Error, Debug)]
21pub enum CompileErrorType {
22    #[error(transparent)]
23    Codegen(#[from] codegen::error::CodegenErrorType),
24    #[error(transparent)]
25    Parse(#[from] ParseErrorType),
26}
27
28#[derive(Error, Debug)]
29pub struct ParseError {
30    #[source]
31    pub error: ParseErrorType,
32    pub raw_location: ruff_text_size::TextRange,
33    pub location: SourceLocation,
34    pub end_location: SourceLocation,
35    pub source_path: String,
36    /// Set when the error is an unclosed bracket (converted from EOF).
37    pub is_unclosed_bracket: bool,
38    /// Set when a string is still open at EOF and more input could close it.
39    pub is_unclosed_string: bool,
40}
41
42impl ::core::fmt::Display for ParseError {
43    fn fmt(&self, f: &mut ::core::fmt::Formatter<'_>) -> ::core::fmt::Result {
44        self.error.fmt(f)
45    }
46}
47
48#[derive(Error, Debug)]
49pub enum CompileError {
50    #[error(transparent)]
51    Codegen(#[from] codegen::error::CodegenError),
52    #[error(transparent)]
53    Parse(#[from] ParseError),
54}
55
56impl CompileError {
57    #[must_use]
58    pub fn from_ruff_parse_error(
59        error: parser::ParseError,
60        source_file: &SourceFile,
61        mode: Mode,
62    ) -> Self {
63        let raw_location = error.location;
64        let diagnostic = match cpython_parse_diagnostic_override(&error, source_file, mode) {
65            Some(diagnostic) => diagnostic,
66            None => default_parse_diagnostic(error, source_file),
67        };
68
69        Self::Parse(ParseError {
70            error: diagnostic.error,
71            raw_location,
72            location: diagnostic.location,
73            end_location: diagnostic.end_location,
74            source_path: source_file.name().to_owned(),
75            is_unclosed_bracket: diagnostic.is_unclosed_bracket,
76            is_unclosed_string: diagnostic.is_unclosed_string,
77        })
78    }
79
80    fn from_source_error(source_file: &SourceFile, diagnostic: CpythonDiagnostic) -> Self {
81        let (location, end_location) = source_locations(
82            source_file,
83            diagnostic.range.start(),
84            diagnostic.range.end(),
85        );
86        Self::Parse(ParseError {
87            error: parser::ParseErrorType::OtherError(diagnostic.message),
88            raw_location: diagnostic.range,
89            location,
90            end_location,
91            source_path: source_file.name().to_owned(),
92            is_unclosed_bracket: diagnostic.is_unclosed_bracket,
93            is_unclosed_string: diagnostic.is_unclosed_string,
94        })
95    }
96
97    #[must_use]
98    pub const fn location(&self) -> Option<SourceLocation> {
99        match self {
100            Self::Codegen(codegen_error) => codegen_error.location,
101            Self::Parse(parse_error) => Some(parse_error.location),
102        }
103    }
104
105    #[must_use]
106    pub const fn python_location(&self) -> (usize, usize) {
107        if let Some(location) = self.location() {
108            (location.line.get(), location.character_offset.get())
109        } else {
110            (0, 0)
111        }
112    }
113
114    #[must_use]
115    pub fn python_end_location(&self) -> Option<(usize, usize)> {
116        match self {
117            Self::Codegen(codegen_error) => codegen_error
118                .end_location
119                .map(|end| (end.line.get(), end.character_offset.get())),
120            Self::Parse(parse_error) => Some((
121                parse_error.end_location.line.get(),
122                parse_error.end_location.character_offset.get(),
123            )),
124        }
125    }
126
127    #[must_use]
128    pub fn source_path(&self) -> &str {
129        match self {
130            Self::Codegen(codegen_error) => &codegen_error.source_path,
131            Self::Parse(parse_error) => &parse_error.source_path,
132        }
133    }
134}
135
136// A syntax error the parser reports counts its columns in characters rather
137// than in the bytes the range is measured in. An offset that lands inside a
138// character walks back to where that character starts, the way decoding the
139// line up to a truncated offset would.
140fn source_location(source_file: &SourceFile, offset: TextSize) -> SourceLocation {
141    let text = source_file.source_text();
142    let mut index = offset.to_usize().min(text.len());
143    while !text.is_char_boundary(index) {
144        index -= 1;
145    }
146    source_file
147        .to_source_code()
148        .source_location(TextSize::new(index as u32), PositionEncoding::Utf32)
149}
150
151fn source_locations(
152    source_file: &SourceFile,
153    start: TextSize,
154    end: TextSize,
155) -> (SourceLocation, SourceLocation) {
156    (
157        source_location(source_file, start),
158        source_location(source_file, end),
159    )
160}
161
162struct NormalizedParseDiagnostic {
163    error: parser::ParseErrorType,
164    location: SourceLocation,
165    end_location: SourceLocation,
166    is_unclosed_bracket: bool,
167    is_unclosed_string: bool,
168}
169
170impl NormalizedParseDiagnostic {
171    const fn new(
172        error: parser::ParseErrorType,
173        location: SourceLocation,
174        end_location: SourceLocation,
175    ) -> Self {
176        Self {
177            error,
178            location,
179            end_location,
180            is_unclosed_bracket: false,
181            is_unclosed_string: false,
182        }
183    }
184
185    fn other(source_file: &SourceFile, diagnostic: CpythonDiagnostic) -> Self {
186        let (location, end_location) = source_locations(
187            source_file,
188            diagnostic.range.start(),
189            diagnostic.range.end(),
190        );
191        let mut diagnostic_out = Self::new(
192            parser::ParseErrorType::OtherError(diagnostic.message),
193            location,
194            end_location,
195        );
196        diagnostic_out.is_unclosed_string = diagnostic.is_unclosed_string;
197        diagnostic_out.is_unclosed_bracket = diagnostic.is_unclosed_bracket;
198        diagnostic_out
199    }
200
201    const fn with_unclosed_bracket(mut self, is_unclosed_bracket: bool) -> Self {
202        self.is_unclosed_bracket = is_unclosed_bracket;
203        self
204    }
205}
206
207/// What CPython would have reported for a piece of source, before it is resolved to a line and
208/// column. These are reconstructed by re-scanning after ruff's parse has already failed, so they
209/// carry CPython's wording rather than a translation of ruff's own error, and they never reach
210/// ruff — `NormalizedParseDiagnostic` and `CompileError` are the only things that consume one.
211#[derive(Clone)]
212struct CpythonDiagnostic {
213    message: String,
214    range: ruff_text_size::TextRange,
215    is_unclosed_string: bool,
216    is_unclosed_bracket: bool,
217}
218
219impl CpythonDiagnostic {
220    /// The `u32` cast is ruff's invariant rather than one this adds: its `Lexer::new` asserts
221    /// that the source fits in a `u32` ("Lexer only supports files with a size up to 4GB") and
222    /// relies on that for its own offset arithmetic, and nothing here runs until that lexer has
223    /// read the source and the parse built on it has failed.
224    fn new(message: String, start: usize, end: usize) -> Self {
225        Self {
226            message,
227            range: ruff_text_size::TextRange::new(
228                TextSize::new(start as u32),
229                TextSize::new(end as u32),
230            ),
231            is_unclosed_string: false,
232            is_unclosed_bracket: false,
233        }
234    }
235
236    const fn with_unclosed_string(mut self) -> Self {
237        self.is_unclosed_string = true;
238        self
239    }
240
241    const fn with_unclosed_bracket(mut self) -> Self {
242        self.is_unclosed_bracket = true;
243        self
244    }
245}
246
247/// Lexer-class failures outrank print hints. Decode and f-string diagnostics
248/// compete with lexer failures by offset, and an unterminated quote is only
249/// a fallback when no decode/f-string diagnostic exists. Print is considered
250/// against the final winner: a later 0x still beats print, but an f-string
251/// that replaced that 0x must not hide an earlier print. A decode diagnostic
252/// also suppresses an EOF unclosed opener.
253#[derive(Clone, Copy, PartialEq, Eq)]
254enum OverrideClass {
255    Lexer,
256    Decode,
257    Print,
258}
259
260struct RankedOverride {
261    diagnostic: CpythonDiagnostic,
262    unclosed_bracket: bool,
263    class: OverrideClass,
264}
265
266fn consider_override(
267    best: &mut Option<RankedOverride>,
268    diagnostic: CpythonDiagnostic,
269    class: OverrideClass,
270) {
271    let unclosed_bracket = diagnostic.is_unclosed_bracket;
272    consider_ranked(best, diagnostic, unclosed_bracket, class);
273}
274
275fn consider_ranked(
276    best: &mut Option<RankedOverride>,
277    diagnostic: CpythonDiagnostic,
278    unclosed_bracket: bool,
279    class: OverrideClass,
280) {
281    if best
282        .as_ref()
283        .is_none_or(|current| diagnostic.range.start() < current.diagnostic.range.start())
284    {
285        *best = Some(RankedOverride {
286            diagnostic,
287            unclosed_bracket,
288            class,
289        });
290    }
291}
292
293fn cpython_parse_diagnostic_override(
294    error: &parser::ParseError,
295    source_file: &SourceFile,
296    mode: Mode,
297) -> Option<NormalizedParseDiagnostic> {
298    let source_text = source_file.source_text();
299
300    macro_rules! source_error {
301        ($expr:expr) => {
302            if let Some(error) = $expr {
303                return Some(NormalizedParseDiagnostic::other(source_file, error));
304            }
305        };
306    }
307
308    let mut earliest: Option<RankedOverride> = None;
309    if let Some(diagnostic) = invalid_number_literal_error(source_text) {
310        consider_override(&mut earliest, diagnostic, OverrideClass::Lexer);
311    }
312    if let Some(diagnostic) = incompatible_string_prefix_error(source_text) {
313        consider_override(&mut earliest, diagnostic, OverrideClass::Lexer);
314    }
315    if let Some(diagnostic) = non_printable_character_error(source_text) {
316        consider_override(&mut earliest, diagnostic, OverrideClass::Lexer);
317    }
318    let bracket = bracket_syntax_error(source_text);
319    if let Some(bracket) = bracket.as_ref() {
320        // Unclosed openers are reported at the opener and only become errors
321        // at EOF. A later token-time diagnostic (invalid number, prefix, …)
322        // must keep winning. Mismatched closers stay in the lexer-class
323        // positional ranking.
324        if !bracket.unclosed {
325            consider_ranked(
326                &mut earliest,
327                bracket.diagnostic.clone(),
328                false,
329                OverrideClass::Lexer,
330            );
331        }
332    }
333    let mut saw_decode = false;
334    if let Some(diagnostic) = malformed_unicode_n_escape_error(source_text) {
335        saw_decode = true;
336        consider_override(&mut earliest, diagnostic, OverrideClass::Decode);
337    }
338    if let Some(diagnostic) = invalid_interpolated_string_error(source_text) {
339        saw_decode = true;
340        consider_override(&mut earliest, diagnostic, OverrideClass::Decode);
341    }
342    if let Some(diagnostic) = mixed_tstring_literal_error(error, source_text) {
343        saw_decode = true;
344        consider_override(&mut earliest, diagnostic, OverrideClass::Decode);
345    }
346    // A later ordinary unterminated quote is only a fallback. Format-spec
347    // newlines and empty fields are decode diagnostics and must keep winning.
348    let line_continuation = matches!(
349        &error.error,
350        parser::ParseErrorType::Lexical(parser::LexicalErrorType::LineContinuationError)
351    );
352    if !saw_decode
353        && !line_continuation
354        && let Some(diagnostic) = unterminated_string_error(source_text, mode)
355    {
356        consider_override(&mut earliest, diagnostic, OverrideClass::Lexer);
357    }
358    // Indentation errors outrank a print/exec missing-parentheses rewrite.
359    let indent_error = matches!(
360        &error.error,
361        parser::ParseErrorType::Lexical(parser::LexicalErrorType::IndentationError)
362            | parser::ParseErrorType::UnexpectedIndentation
363    ) || expected_indented_block_error(error, source_text).is_some();
364    if !indent_error
365        && earliest
366            .as_ref()
367            .is_none_or(|current| current.class != OverrideClass::Lexer)
368        && let Some(diagnostic) = invalid_legacy_statement_error(source_text)
369    {
370        consider_override(&mut earliest, diagnostic, OverrideClass::Print);
371    }
372    if !saw_decode
373        && earliest
374            .as_ref()
375            .is_none_or(|current| current.class == OverrideClass::Print)
376        && let Some(bracket) = bracket.filter(|bracket| bracket.unclosed)
377    {
378        consider_ranked(
379            &mut earliest,
380            bracket.diagnostic,
381            true,
382            OverrideClass::Lexer,
383        );
384    }
385    if let Some(override_diag) = earliest {
386        return Some(
387            NormalizedParseDiagnostic::other(source_file, override_diag.diagnostic)
388                .with_unclosed_bracket(override_diag.unclosed_bracket),
389        );
390    }
391
392    if matches!(
393        &error.error,
394        parser::ParseErrorType::Lexical(parser::LexicalErrorType::LineContinuationError)
395    ) {
396        // exec input gets an implicit trailing newline, so a final `\` is a
397        // continuation that then hits EOF (`E_EOF`). single/eval see `\` at
398        // EOF as `E_LINECONT` instead.
399        let terminal_backslash = source_text.len().checked_sub(1);
400        if matches!(mode, Mode::Exec)
401            && terminal_backslash == Some(error.location.start().to_usize())
402        {
403            let loc = source_line_end_location(source_file, error.location.start());
404            return Some(NormalizedParseDiagnostic::new(
405                parser::ParseErrorType::OtherError("unexpected EOF while parsing".to_owned()),
406                loc,
407                loc,
408            ));
409        }
410        let loc = source_location(source_file, error.location.start() + TextSize::from(1));
411        return Some(NormalizedParseDiagnostic::new(
412            parser::ParseErrorType::OtherError(
413                "unexpected character after line continuation character".to_owned(),
414            ),
415            loc,
416            loc,
417        ));
418    }
419
420    source_error!(unterminated_string_error(source_text, mode));
421    source_error!(expected_indented_block_error(error, source_text));
422
423    if matches!(
424        &error.error,
425        parser::ParseErrorType::Lexical(parser::LexicalErrorType::Eof)
426    ) {
427        return Some(eof_parse_diagnostic(error, source_file));
428    }
429
430    source_error!(invalid_type_param_error(source_text));
431    source_error!(invalid_comprehension_error(source_text));
432    source_error!(invalid_parameter_star_annotation_error(source_text));
433    source_error!(invalid_parameter_list_error(source_text));
434    source_error!(invalid_call_argument_error(source_text));
435
436    if is_missing_comma_between_literals(error) {
437        let (loc, end_loc) = adjusted_error_locations(source_file, error.location);
438        let msg = "invalid syntax. Perhaps you forgot a comma?".into();
439        return Some(NormalizedParseDiagnostic::new(
440            parser::ParseErrorType::OtherError(msg),
441            loc,
442            end_loc,
443        ));
444    }
445
446    source_error!(invalid_dict_error(source_text));
447    if !matches!(mode, Mode::Eval) {
448        source_error!(invalid_collection_assignment_error(source_text));
449    }
450    source_error!(invalid_group_error(source_text));
451    source_error!(invalid_def_type_params_error(source_text));
452    source_error!(invalid_expression_error(source_text));
453    source_error!(invalid_named_expression_error(source_text));
454    // Assignment and other statements are not expressions. Eval keeps the
455    // generic parse error rather than a statement-target rewrite.
456    if !matches!(mode, Mode::Eval) {
457        source_error!(invalid_plain_assignment_error(source_text));
458        source_error!(expression_assignment_error(source_text));
459        source_error!(invalid_annotation_target_error(source_text));
460        source_error!(invalid_assignment_target_error(source_text));
461        source_error!(invalid_condition_assignment_error(
462            source_text,
463            error.location.start().to_usize()
464        ));
465        source_error!(invalid_augassign_target_error(source_text));
466        source_error!(invalid_for_target_error(source_text));
467        source_error!(invalid_with_target_error(source_text));
468        source_error!(invalid_delete_target_error(source_text));
469        source_error!(invalid_standalone_except_error(source_text));
470        source_error!(invalid_import_statement_error(source_text));
471        source_error!(invalid_import_target_error(source_text));
472        source_error!(invalid_except_as_target_error(source_text));
473        source_error!(invalid_match_mapping_rest_wildcard_error(source_text));
474        source_error!(invalid_match_as_target_error(source_text));
475        source_error!(invalid_for_if_clause_error(source_text));
476        source_error!(invalid_if_expression_statement_error(source_text));
477        source_error!(invalid_else_elif_error(source_text));
478        source_error!(mixed_except_handlers_error(source_text));
479    }
480
481    if matches!(
482        &error.error,
483        parser::ParseErrorType::Lexical(parser::LexicalErrorType::InvalidByteLiteral)
484    ) && let Some((start, end)) =
485        bytes_literal_span(source_text, error.location.start().to_usize())
486    {
487        let (loc, end_loc) = source_locations(
488            source_file,
489            TextSize::new(start as u32),
490            TextSize::new(end as u32),
491        );
492        return Some(NormalizedParseDiagnostic::new(
493            error.error.clone(),
494            loc,
495            end_loc,
496        ));
497    }
498
499    if matches!(
500        &error.error,
501        parser::ParseErrorType::Lexical(parser::LexicalErrorType::IndentationError)
502    ) {
503        let end_loc = source_line_end_location(source_file, error.location.start());
504        return Some(NormalizedParseDiagnostic::new(
505            error.error.clone(),
506            end_loc,
507            end_loc,
508        ));
509    }
510
511    if matches!(
512        &error.error,
513        parser::ParseErrorType::InvalidAssignmentTarget
514    ) {
515        return Some(invalid_assignment_target_diagnostic(error, source_file));
516    }
517
518    if matches!(
519        &error.error,
520        parser::ParseErrorType::InvalidNamedAssignmentTarget
521    ) {
522        let (loc, end_loc) = adjusted_error_locations(source_file, error.location);
523        let target = source_file.source_text().slice(error.location);
524        let msg = format!("cannot use assignment expressions with {target}");
525        return Some(NormalizedParseDiagnostic::new(
526            parser::ParseErrorType::OtherError(msg),
527            loc,
528            end_loc,
529        ));
530    }
531
532    // CPython's PEG parser collapses a bare "expected an expression" failure
533    // into the generic "invalid syntax" message. rustpython-vm's `vm_new.rs`
534    // does this same collapse for its own callers; rustpython-compiler has no
535    // vm dependency, so mirror it here.
536    if matches!(
537        &error.error,
538        parser::ParseErrorType::ExpectedExpression
539            | parser::ParseErrorType::UnexpectedExpressionToken
540    ) {
541        let (loc, end_loc) = adjusted_error_locations(source_file, error.location);
542        return Some(NormalizedParseDiagnostic::new(
543            parser::ParseErrorType::OtherError("invalid syntax".into()),
544            loc,
545            end_loc,
546        ));
547    }
548
549    None
550}
551
552fn eof_parse_diagnostic(
553    error: &parser::ParseError,
554    source_file: &SourceFile,
555) -> NormalizedParseDiagnostic {
556    let source_text = source_file.source_text();
557    if let Some((bracket_char, bracket_offset)) = find_unclosed_bracket(source_text) {
558        let loc = source_location(source_file, TextSize::new(bracket_offset as u32));
559        let end_loc = SourceLocation {
560            line: loc.line,
561            character_offset: loc.character_offset.saturating_add(1),
562        };
563        let msg = format!("'{bracket_char}' was never closed");
564        NormalizedParseDiagnostic::new(parser::ParseErrorType::OtherError(msg), loc, end_loc)
565            .with_unclosed_bracket(true)
566    } else {
567        let end_loc = source_line_end_location(source_file, error.location.start());
568        NormalizedParseDiagnostic::new(error.error.clone(), end_loc, end_loc)
569    }
570}
571
572fn invalid_assignment_target_diagnostic(
573    error: &parser::ParseError,
574    source_file: &SourceFile,
575) -> NormalizedParseDiagnostic {
576    let (loc, end_loc) = adjusted_error_locations(source_file, error.location);
577    let expr_str = source_file.source_text().slice(error.location);
578
579    let msg = parser::parse_expression(expr_str).map_or_else(
580        |_| match expr_str {
581            "yield" => "assignment to yield expression not possible".into(),
582            _ => format!("cannot assign to {expr_str}"),
583        },
584        |parsed| match *parsed.syntax().body {
585            ast::Expr::Call(_) => "cannot assign to function call".into(),
586            ast::Expr::BinOp(_) => "cannot assign to expression".into(),
587            ast::Expr::If(_) => "cannot assign to conditional expression".into(),
588            ast::Expr::Generator(_) => "cannot assign to generator expression".into(),
589            ast::Expr::Yield(_) | ast::Expr::YieldFrom(_) => {
590                "cannot assign to yield expression here. Maybe you meant '==' instead of '='?"
591                    .into()
592            }
593            ast::Expr::FString(_) => "invalid syntax".into(),
594            ast::Expr::StringLiteral(_)
595            | ast::Expr::BytesLiteral(_)
596            | ast::Expr::NumberLiteral(_) => {
597                "cannot assign to literal here. Maybe you meant '==' instead of '='?".into()
598            }
599            ast::Expr::EllipsisLiteral(_) => {
600                "cannot assign to ellipsis here. Maybe you meant '==' instead of '='?".into()
601            }
602            _ => format!("cannot assign to {expr_str}"),
603        },
604    );
605
606    NormalizedParseDiagnostic::new(parser::ParseErrorType::OtherError(msg), loc, end_loc)
607}
608
609fn default_parse_diagnostic(
610    error: parser::ParseError,
611    source_file: &SourceFile,
612) -> NormalizedParseDiagnostic {
613    let (loc, end_loc) = adjusted_error_locations(source_file, error.location);
614    NormalizedParseDiagnostic::new(error.error, loc, end_loc)
615}
616
617fn adjusted_error_locations(
618    source_file: &SourceFile,
619    range: ruff_text_size::TextRange,
620) -> (SourceLocation, SourceLocation) {
621    let mut locations = source_locations(source_file, range.start(), range.end());
622    if locations.1.character_offset.get() == 1 && locations.1.line > locations.0.line {
623        locations.1 = source_location(source_file, range.end() - TextSize::from(1));
624        locations.1.character_offset = locations.1.character_offset.saturating_add(1);
625    } else if range.is_empty() {
626        // The parser blames a token, and the narrowest token still covers a
627        // character, so an error reported between two of them spans one.
628        locations.1.character_offset = locations.1.character_offset.saturating_add(1);
629    }
630    locations
631}
632
633fn source_line_end_location(source_file: &SourceFile, offset: TextSize) -> SourceLocation {
634    let loc = source_location(source_file, offset);
635    let line_idx = loc.line.to_zero_indexed();
636    let line = source_file
637        .source_text()
638        .split('\n')
639        .nth(line_idx)
640        .unwrap_or("");
641    let line_end_col = line.chars().count() + 1;
642    SourceLocation {
643        line: loc.line,
644        character_offset: ruff_source_file::OneIndexed::new(line_end_col)
645            .unwrap_or(loc.character_offset),
646    }
647}
648
649fn is_missing_comma_between_literals(error: &parser::ParseError) -> bool {
650    matches!(
651        &error.error,
652        parser::ParseErrorType::ExpectedToken { expected, found }
653            if matches!((expected, found), (TokenKind::Comma, TokenKind::Int))
654    )
655}
656
657fn is_ascii_identifier_char(byte: u8) -> bool {
658    byte == b'_' || byte.is_ascii_alphanumeric()
659}
660
661fn identifier_continue_before(bytes: &[u8], index: usize) -> bool {
662    if index == 0 {
663        return false;
664    }
665    if bytes[index - 1].is_ascii() {
666        return is_ascii_identifier_char(bytes[index - 1]);
667    }
668    let mut start = index - 1;
669    while start > 0 && bytes[start] & 0b1100_0000 == 0b1000_0000 {
670        start -= 1;
671    }
672    ::core::str::from_utf8(&bytes[start..index])
673        .ok()
674        .and_then(|text| text.chars().next_back())
675        .is_some_and(|ch| ch == '_' || ch.is_alphanumeric())
676}
677
678fn numeric_keyword_suffix(rest: &[u8]) -> bool {
679    rest.starts_with(b"and")
680        || rest.starts_with(b"else")
681        || rest.starts_with(b"for")
682        || rest.starts_with(b"if")
683        || rest.starts_with(b"in")
684        || rest.starts_with(b"is")
685        || rest.starts_with(b"or")
686        || rest.starts_with(b"not")
687}
688
689fn consume_decimal_digits(bytes: &[u8], mut index: usize) -> usize {
690    while index < bytes.len() {
691        match bytes[index] {
692            b'0'..=b'9' => index += 1,
693            b'_' if bytes
694                .get(index + 1)
695                .is_some_and(|byte| byte.is_ascii_digit()) =>
696            {
697                index += 2;
698            }
699            _ => break,
700        }
701    }
702    index
703}
704
705fn consume_radix_digits(bytes: &[u8], mut index: usize, is_digit: impl Fn(u8) -> bool) -> usize {
706    while index < bytes.len() {
707        if is_digit(bytes[index]) {
708            index += 1;
709        } else if bytes.get(index) == Some(&b'_')
710            && bytes.get(index + 1).is_some_and(|&byte| is_digit(byte))
711        {
712            index += 2;
713        } else {
714            break;
715        }
716    }
717    index
718}
719
720fn invalid_radix_literal_error(
721    bytes: &[u8],
722    start: usize,
723    kind: &'static str,
724    is_digit: impl Fn(u8) -> bool,
725) -> Option<(String, usize)> {
726    let mut index = start + 2;
727    let mut has_digit = false;
728    loop {
729        let Some(&byte) = bytes.get(index) else {
730            return if has_digit {
731                None
732            } else {
733                Some((format!("invalid {kind} literal"), start + 1))
734            };
735        };
736        if byte == b'_' {
737            let Some(&next) = bytes.get(index + 1) else {
738                return Some((format!("invalid {kind} literal"), index));
739            };
740            if is_digit(next) {
741                has_digit = true;
742                index += 2;
743                continue;
744            }
745            if next.is_ascii_digit() && matches!(kind, "binary" | "octal") {
746                return Some((
747                    format!("invalid digit '{}' in {kind} literal", next as char),
748                    index + 1,
749                ));
750            }
751            return Some((format!("invalid {kind} literal"), index));
752        }
753        if is_digit(byte) {
754            has_digit = true;
755            index += 1;
756            continue;
757        }
758        if byte.is_ascii_digit() && matches!(kind, "binary" | "octal") {
759            return Some((
760                format!("invalid digit '{}' in {kind} literal", byte as char),
761                index,
762            ));
763        }
764        if has_digit {
765            return None;
766        }
767        return Some((format!("invalid {kind} literal"), start + 1));
768    }
769}
770
771fn decimal_tail_error(bytes: &[u8], mut index: usize) -> Option<usize> {
772    loop {
773        while bytes.get(index).is_some_and(|byte| byte.is_ascii_digit()) {
774            index += 1;
775        }
776        if bytes.get(index) != Some(&b'_') {
777            return None;
778        }
779        let underscore = index;
780        index += 1;
781        if !bytes.get(index).is_some_and(|byte| byte.is_ascii_digit()) {
782            return Some(underscore);
783        }
784    }
785}
786
787fn decimal_tail_end(bytes: &[u8], mut index: usize) -> usize {
788    loop {
789        while bytes.get(index).is_some_and(|byte| byte.is_ascii_digit()) {
790            index += 1;
791        }
792        if bytes.get(index) == Some(&b'_')
793            && bytes
794                .get(index + 1)
795                .is_some_and(|byte| byte.is_ascii_digit())
796        {
797            index += 2;
798        } else {
799            return index;
800        }
801    }
802}
803
804fn invalid_decimal_literal_error(bytes: &[u8], start: usize) -> Option<(String, usize)> {
805    if bytes.get(start) == Some(&b'.') {
806        return None;
807    }
808    let message = "invalid decimal literal".to_owned();
809    if let Some(offset) = decimal_tail_error(bytes, start) {
810        return Some((message, offset));
811    }
812
813    let mut index = decimal_tail_end(bytes, start);
814    if bytes.get(index) == Some(&b'.') {
815        if bytes.get(index + 1) == Some(&b'_') {
816            return Some((message, index));
817        }
818        if let Some(offset) = decimal_tail_error(bytes, index + 1) {
819            return Some((message, offset));
820        }
821        index = decimal_tail_end(bytes, index + 1);
822    }
823    if matches!(bytes.get(index), Some(b'e' | b'E')) {
824        let exponent = index;
825        index += 1;
826        let sign = if matches!(bytes.get(index), Some(b'+' | b'-')) {
827            let sign = index;
828            index += 1;
829            Some(sign)
830        } else {
831            None
832        };
833        if !bytes.get(index).is_some_and(|byte| byte.is_ascii_digit()) {
834            // Without a sign the exponent letter is put back, so the position
835            // is the digit before it rather than the letter.
836            return Some((message, sign.unwrap_or_else(|| exponent.saturating_sub(1))));
837        }
838        if let Some(offset) = decimal_tail_error(bytes, index) {
839            return Some((message, offset));
840        }
841    }
842    None
843}
844
845fn leading_zero_decimal_literal_error(bytes: &[u8], start: usize) -> Option<CpythonDiagnostic> {
846    if bytes.get(start) != Some(&b'0') {
847        return None;
848    }
849    let mut index = start;
850    loop {
851        match bytes.get(index) {
852            Some(b'0') => index += 1,
853            Some(b'_')
854                if bytes
855                    .get(index + 1)
856                    .is_some_and(|byte| byte.is_ascii_digit()) =>
857            {
858                index += 1;
859            }
860            _ => break,
861        }
862    }
863    if bytes.get(index).is_some_and(|byte| byte.is_ascii_digit()) {
864        let after_digits = decimal_tail_end(bytes, index);
865        if !matches!(
866            bytes.get(after_digits),
867            Some(b'.' | b'e' | b'E' | b'j' | b'J')
868        ) {
869            return Some(CpythonDiagnostic::new(
870                "leading zeros in decimal integer literals are not permitted; use an 0o prefix for octal integers"
871                    .to_owned(),
872                start,
873                index,
874            ));
875        }
876    }
877    None
878}
879
880fn invalid_numeric_literal_error(bytes: &[u8], start: usize) -> Option<CpythonDiagnostic> {
881    if bytes.get(start) == Some(&b'0') {
882        let radix = match bytes.get(start + 1) {
883            Some(b'x' | b'X') => invalid_radix_literal_error(bytes, start, "hexadecimal", |byte| {
884                byte.is_ascii_hexdigit()
885            }),
886            Some(b'o' | b'O') => invalid_radix_literal_error(bytes, start, "octal", |byte| {
887                matches!(byte, b'0'..=b'7')
888            }),
889            Some(b'b' | b'B') => invalid_radix_literal_error(bytes, start, "binary", |byte| {
890                matches!(byte, b'0' | b'1')
891            }),
892            _ => None,
893        };
894        if let Some(radix) = radix {
895            return Some(point_span(radix));
896        }
897        if let Some(err) = leading_zero_decimal_literal_error(bytes, start) {
898            return Some(err);
899        }
900    }
901    invalid_decimal_literal_error(bytes, start).map(point_span)
902}
903
904fn consume_exponent(bytes: &[u8], index: usize) -> usize {
905    if !matches!(bytes.get(index), Some(b'e' | b'E')) {
906        return index;
907    }
908    let mut cursor = index + 1;
909    if matches!(bytes.get(cursor), Some(b'+' | b'-')) {
910        cursor += 1;
911    }
912    if bytes.get(cursor).is_some_and(|byte| byte.is_ascii_digit()) {
913        consume_decimal_digits(bytes, cursor)
914    } else {
915        index
916    }
917}
918
919fn number_literal_end(bytes: &[u8], start: usize) -> Option<(&'static str, usize)> {
920    if bytes.get(start) == Some(&b'.') {
921        if !bytes
922            .get(start + 1)
923            .is_some_and(|byte| byte.is_ascii_digit())
924        {
925            return None;
926        }
927        let mut index = consume_decimal_digits(bytes, start + 1);
928        index = consume_exponent(bytes, index);
929        if matches!(bytes.get(index), Some(b'j' | b'J')) {
930            return Some(("imaginary", index + 1));
931        }
932        return Some(("decimal", index));
933    }
934
935    if !bytes.get(start).is_some_and(|byte| byte.is_ascii_digit()) {
936        return None;
937    }
938
939    if bytes.get(start) == Some(&b'0') {
940        match bytes.get(start + 1) {
941            Some(b'x' | b'X') => {
942                let end = consume_radix_digits(bytes, start + 2, |byte| byte.is_ascii_hexdigit());
943                return Some(("hexadecimal", end));
944            }
945            Some(b'o' | b'O') => {
946                let end =
947                    consume_radix_digits(bytes, start + 2, |byte| matches!(byte, b'0'..=b'7'));
948                return Some(("octal", end));
949            }
950            Some(b'b' | b'B') => {
951                let end =
952                    consume_radix_digits(bytes, start + 2, |byte| matches!(byte, b'0' | b'1'));
953                return Some(("binary", end));
954            }
955            _ => {}
956        }
957    }
958
959    let mut index = consume_decimal_digits(bytes, start);
960    if bytes.get(index) == Some(&b'.') {
961        index = consume_decimal_digits(bytes, index + 1);
962    }
963    index = consume_exponent(bytes, index);
964    if matches!(bytes.get(index), Some(b'j' | b'J')) {
965        return Some(("imaginary", index + 1));
966    }
967    Some(("decimal", index))
968}
969
970fn quoted_string_is_closed(bytes: &[u8], start: usize) -> bool {
971    let quote = bytes[start];
972    let triple = bytes.get(start + 1) == Some(&quote) && bytes.get(start + 2) == Some(&quote);
973    let end = skip_quoted_string(bytes, start);
974    if triple {
975        end >= start + 6
976            && bytes[end - 3] == quote
977            && bytes[end - 2] == quote
978            && bytes[end - 1] == quote
979    } else {
980        end > start + 1 && bytes[end - 1] == quote
981    }
982}
983
984fn bytes_literal_span(source: &str, error_at: usize) -> Option<(usize, usize)> {
985    let bytes = source.as_bytes();
986    if error_at > bytes.len() || bytes.is_empty() {
987        return None;
988    }
989    let mut quote_idx = error_at.min(bytes.len().saturating_sub(1));
990    loop {
991        if matches!(bytes[quote_idx], b'\'' | b'"') {
992            break;
993        }
994        if quote_idx == 0 {
995            return None;
996        }
997        quote_idx -= 1;
998    }
999    let mut start = quote_idx;
1000    while start > 0 && matches!(bytes[start - 1], b'b' | b'B' | b'r' | b'R') {
1001        start -= 1;
1002    }
1003    if !matches!(bytes.get(start), Some(b'b' | b'B' | b'r' | b'R')) {
1004        return None;
1005    }
1006    let end = skip_quoted_string(bytes, quote_idx);
1007    Some((start, end))
1008}
1009
1010fn skip_quoted_string(bytes: &[u8], mut index: usize) -> usize {
1011    let quote = bytes[index];
1012    let triple = bytes.get(index + 1) == Some(&quote) && bytes.get(index + 2) == Some(&quote);
1013    let quote_len = if triple { 3 } else { 1 };
1014    index += quote_len;
1015    while index < bytes.len() {
1016        if bytes[index] == b'\\' {
1017            index = (index + 2).min(bytes.len());
1018        } else if triple
1019            && bytes.get(index) == Some(&quote)
1020            && bytes.get(index + 1) == Some(&quote)
1021            && bytes.get(index + 2) == Some(&quote)
1022        {
1023            return index + 3;
1024        } else if !triple && bytes[index] == quote {
1025            return index + 1;
1026        } else {
1027            index += 1;
1028        }
1029    }
1030    index
1031}
1032
1033// An error the tokenizer reports at a single position spans nothing.
1034fn point_span((message, offset): (String, usize)) -> CpythonDiagnostic {
1035    CpythonDiagnostic::new(message, offset, offset)
1036}
1037
1038fn invalid_number_literal_error(source: &str) -> Option<CpythonDiagnostic> {
1039    let bytes = source.as_bytes();
1040    let mut index = 0;
1041    while index < bytes.len() {
1042        match bytes[index] {
1043            b'#' => {
1044                while index < bytes.len() && bytes[index] != b'\n' {
1045                    index += 1;
1046                }
1047            }
1048            b'\'' | b'"' => {
1049                index = skip_quoted_string(bytes, index);
1050            }
1051            byte if byte >= 0x80 || byte == b'_' || byte.is_ascii_alphabetic() => {
1052                index += 1;
1053                while index < bytes.len()
1054                    && (bytes[index] >= 0x80 || is_ascii_identifier_char(bytes[index]))
1055                {
1056                    index += 1;
1057                }
1058            }
1059            b'.' | b'0'..=b'9' => {
1060                if let Some(err) = invalid_numeric_literal_error(bytes, index) {
1061                    return Some(err);
1062                }
1063                let Some((kind, end)) = number_literal_end(bytes, index) else {
1064                    index += 1;
1065                    continue;
1066                };
1067                if end > index {
1068                    if source[end..].starts_with('⁄') {
1069                        return Some(CpythonDiagnostic::new(
1070                            "invalid character '⁄' (U+2044)".to_owned(),
1071                            end,
1072                            end,
1073                        ));
1074                    }
1075                    if bytes
1076                        .get(end)
1077                        .is_some_and(|byte| *byte < 128 && is_ascii_identifier_char(*byte))
1078                        && !numeric_keyword_suffix(&bytes[end..])
1079                    {
1080                        let offset = end.saturating_sub(1);
1081                        return Some(CpythonDiagnostic::new(
1082                            format!("invalid {kind} literal"),
1083                            offset,
1084                            offset,
1085                        ));
1086                    }
1087                }
1088                index = end.max(index + 1);
1089            }
1090            _ => index += 1,
1091        }
1092    }
1093    None
1094}
1095
1096fn cpython_indented_block_clause(message: &str) -> Option<&'static str> {
1097    let clause = message.strip_prefix("Expected an indented block after ")?;
1098    Some(match clause {
1099        "`if` statement" => "'if' statement",
1100        "`elif` clause" => "'elif' statement",
1101        "`else` clause" => "'else' statement",
1102        "`for` statement" => "'for' statement",
1103        "`with` statement" => "'with' statement",
1104        "`while` statement" => "'while' statement",
1105        "`try` statement" => "'try' statement",
1106        "`except` clause" => "'except' statement",
1107        "`finally` clause" => "'finally' statement",
1108        "`match` statement" => "'match' statement",
1109        "`case` block" => "'case' statement",
1110        "`class` definition" => "class definition",
1111        "function definition" => "function definition",
1112        _ => return None,
1113    })
1114}
1115
1116fn previous_non_empty_line_number(source: &str, offset: usize) -> Option<usize> {
1117    let bytes = source.as_bytes();
1118    let mut index = offset.min(bytes.len());
1119    while index > 0 {
1120        let line_end = index;
1121        while index > 0 && bytes[index - 1] != b'\n' {
1122            index -= 1;
1123        }
1124        let line_start = index;
1125        let content_start = skip_horizontal_whitespace(bytes, line_start);
1126        let mut content_end = line_end;
1127        while content_end > content_start
1128            && matches!(
1129                bytes.get(content_end - 1),
1130                Some(b' ' | b'\t' | b'\r' | b'\x0c')
1131            )
1132        {
1133            content_end -= 1;
1134        }
1135        if content_start < content_end {
1136            return Some(
1137                source[..line_start]
1138                    .bytes()
1139                    .filter(|byte| *byte == b'\n')
1140                    .count()
1141                    + 1,
1142            );
1143        }
1144        index = line_start.saturating_sub(1);
1145    }
1146    None
1147}
1148
1149fn expected_indented_block_error(
1150    error: &parser::ParseError,
1151    source: &str,
1152) -> Option<CpythonDiagnostic> {
1153    let parser::ParseErrorType::OtherError(message) = &error.error else {
1154        return None;
1155    };
1156    let mut clause = cpython_indented_block_clause(message)?;
1157    let start = error.location.start().to_usize();
1158    let end = error.location.end().to_usize();
1159    let line = previous_non_empty_line_number(source, start)?;
1160    if clause == "'except' statement"
1161        && let Some(previous_line) = previous_non_empty_line(source, start)
1162        && matches!(
1163            previous_line.trim_start(),
1164            line if line.starts_with("except*") || line.starts_with("except *")
1165        )
1166    {
1167        clause = "'except*' statement";
1168    }
1169    Some(CpythonDiagnostic::new(
1170        format!("expected an indented block after {clause} on line {line}"),
1171        start,
1172        end,
1173    ))
1174}
1175
1176fn previous_non_empty_line(source: &str, offset: usize) -> Option<&str> {
1177    let bytes = source.as_bytes();
1178    let mut index = offset.min(bytes.len());
1179    while index > 0 {
1180        let line_end = index;
1181        while index > 0 && bytes[index - 1] != b'\n' {
1182            index -= 1;
1183        }
1184        let line_start = index;
1185        let mut content_start = line_start;
1186        while content_start < line_end
1187            && matches!(bytes[content_start], b' ' | b'\t' | b'\n' | b'\r' | b'\x0c')
1188        {
1189            content_start += 1;
1190        }
1191        let mut content_end = line_end;
1192        while content_end > content_start
1193            && matches!(bytes[content_end - 1], b' ' | b'\t' | b'\r' | b'\x0c')
1194        {
1195            content_end -= 1;
1196        }
1197        if content_start < content_end {
1198            return source.get(line_start..line_end);
1199        }
1200        index = line_start.saturating_sub(1);
1201    }
1202    None
1203}
1204
1205fn starts_identifier(bytes: &[u8], index: usize, word: &[u8]) -> bool {
1206    bytes.get(index..index + word.len()) == Some(word)
1207        && index
1208            .checked_sub(1)
1209            .and_then(|before| bytes.get(before))
1210            .is_none_or(|byte| !is_ascii_identifier_char(*byte))
1211        && bytes
1212            .get(index + word.len())
1213            .is_none_or(|byte| !is_ascii_identifier_char(*byte))
1214}
1215
1216fn is_plain_assignment_operator(bytes: &[u8], index: usize) -> bool {
1217    bytes.get(index) == Some(&b'=')
1218        && bytes.get(index + 1) != Some(&b'=')
1219        && !matches!(
1220            index.checked_sub(1).and_then(|before| bytes.get(before)),
1221            Some(b'=' | b'!' | b'<' | b'>' | b':')
1222        )
1223}
1224
1225fn is_simple_keyword_name(bytes: &[u8], mut start: usize, mut end: usize) -> bool {
1226    while matches!(
1227        bytes.get(start),
1228        Some(b' ' | b'\t' | b'\n' | b'\r' | b'\x0c')
1229    ) {
1230        start += 1;
1231    }
1232    while end > start
1233        && matches!(
1234            bytes.get(end - 1),
1235            Some(b' ' | b'\t' | b'\n' | b'\r' | b'\x0c')
1236        )
1237    {
1238        end -= 1;
1239    }
1240    let Some(&first) = bytes.get(start) else {
1241        return false;
1242    };
1243    if !(first == b'_' || first.is_ascii_alphabetic() || first >= 0x80) {
1244        return false;
1245    }
1246    let mut index = start + 1;
1247    while index < end {
1248        if bytes[index] < 0x80 && !is_ascii_identifier_char(bytes[index]) {
1249            return false;
1250        }
1251        index += 1;
1252    }
1253    true
1254}
1255
1256fn is_function_parameter_list(bytes: &[u8], paren: usize) -> bool {
1257    let mut cursor = paren;
1258    while cursor > 0 && matches!(bytes.get(cursor - 1), Some(b' ' | b'\t' | b'\x0c')) {
1259        cursor -= 1;
1260    }
1261    if cursor > 0 && bytes.get(cursor - 1) == Some(&b']') {
1262        let mut bracket = cursor;
1263        let mut level = 0usize;
1264        while bracket > 0 {
1265            bracket -= 1;
1266            match bytes[bracket] {
1267                b']' => level += 1,
1268                b'[' => {
1269                    level = level.saturating_sub(1);
1270                    if level == 0 {
1271                        cursor = bracket;
1272                        break;
1273                    }
1274                }
1275                _ => {}
1276            }
1277        }
1278        while cursor > 0 && matches!(bytes.get(cursor - 1), Some(b' ' | b'\t' | b'\x0c')) {
1279            cursor -= 1;
1280        }
1281    }
1282    while cursor > 0
1283        && bytes
1284            .get(cursor - 1)
1285            .is_some_and(|byte| *byte >= 0x80 || is_ascii_identifier_char(*byte))
1286    {
1287        cursor -= 1;
1288    }
1289    while cursor > 0 && matches!(bytes.get(cursor - 1), Some(b' ' | b'\t' | b'\x0c')) {
1290        cursor -= 1;
1291    }
1292    cursor >= 3
1293        && starts_identifier(bytes, cursor - 3, b"def")
1294        && cursor
1295            .checked_sub(4)
1296            .and_then(|before| bytes.get(before))
1297            .is_none_or(|byte| !is_ascii_identifier_char(*byte))
1298}
1299
1300#[derive(Clone, Copy)]
1301enum ParameterListKind {
1302    Function,
1303    Lambda,
1304}
1305
1306fn matching_delimiter(bytes: &[u8], open: usize, close: u8) -> Option<usize> {
1307    let mut index = open;
1308    let mut level = 0usize;
1309    while index < bytes.len() {
1310        match bytes[index] {
1311            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1312            b'(' | b'[' | b'{' => {
1313                level += 1;
1314                index += 1;
1315            }
1316            byte if byte == close => {
1317                level = level.saturating_sub(1);
1318                if level == 0 {
1319                    return Some(index);
1320                }
1321                index += 1;
1322            }
1323            b')' | b']' | b'}' => {
1324                level = level.saturating_sub(1);
1325                index += 1;
1326            }
1327            _ => index += 1,
1328        }
1329    }
1330    None
1331}
1332
1333fn find_lambda_parameter_end(bytes: &[u8], mut index: usize) -> Option<usize> {
1334    let mut level = 0usize;
1335    while index < bytes.len() {
1336        match bytes[index] {
1337            b'#' if level == 0 => return None,
1338            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1339            b'(' | b'[' | b'{' => {
1340                level += 1;
1341                index += 1;
1342            }
1343            b')' | b']' | b'}' => {
1344                level = level.saturating_sub(1);
1345                index += 1;
1346            }
1347            b':' if level == 0 => return Some(index),
1348            _ => index += 1,
1349        }
1350    }
1351    None
1352}
1353
1354fn top_level_byte(bytes: &[u8], mut index: usize, end: usize, needle: u8) -> Option<usize> {
1355    let mut level = 0usize;
1356    while index < end {
1357        match bytes[index] {
1358            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1359            byte if level == 0 && byte == needle => return Some(index),
1360            b'(' | b'[' | b'{' => {
1361                level += 1;
1362                index += 1;
1363            }
1364            b')' | b']' | b'}' => {
1365                level = level.saturating_sub(1);
1366                index += 1;
1367            }
1368            _ => index += 1,
1369        }
1370    }
1371    None
1372}
1373
1374fn identifier_end(bytes: &[u8], mut index: usize, end: usize) -> usize {
1375    if !bytes
1376        .get(index)
1377        .is_some_and(|byte| *byte >= 0x80 || *byte == b'_' || byte.is_ascii_alphabetic())
1378    {
1379        return index;
1380    }
1381    index += 1;
1382    while index < end
1383        && bytes
1384            .get(index)
1385            .is_some_and(|byte| *byte >= 0x80 || is_ascii_identifier_char(*byte))
1386    {
1387        index += 1;
1388    }
1389    index
1390}
1391
1392fn expression_slice_is_tuple(source: &str, start: usize, end: usize) -> bool {
1393    let bytes = source.as_bytes();
1394    let (start, end) = trim_target_range(bytes, start, end);
1395    if start >= end {
1396        return false;
1397    }
1398    let Ok(parsed) = parser::parse(&source[start..end], parser::Mode::Expression.into()) else {
1399        return false;
1400    };
1401    matches!(parsed.into_syntax(), ast::Mod::Expression(expression) if matches!(*expression.body, ast::Expr::Tuple(_)))
1402}
1403
1404fn type_param_list_open(bytes: &[u8], open: usize) -> bool {
1405    let mut cursor = open;
1406    while cursor > 0 && matches!(bytes.get(cursor - 1), Some(b' ' | b'\t' | b'\x0c')) {
1407        cursor -= 1;
1408    }
1409    while cursor > 0
1410        && bytes
1411            .get(cursor - 1)
1412            .is_some_and(|byte| *byte >= 0x80 || is_ascii_identifier_char(*byte))
1413    {
1414        cursor -= 1;
1415    }
1416    while cursor > 0 && matches!(bytes.get(cursor - 1), Some(b' ' | b'\t' | b'\x0c')) {
1417        cursor -= 1;
1418    }
1419    (cursor >= 3 && starts_identifier(bytes, cursor - 3, b"def"))
1420        || (cursor >= 5 && starts_identifier(bytes, cursor - 5, b"class"))
1421        || (cursor >= 4 && starts_identifier(bytes, cursor - 4, b"type"))
1422}
1423
1424fn invalid_type_param_item_error(
1425    source: &str,
1426    start: usize,
1427    end: usize,
1428) -> Option<CpythonDiagnostic> {
1429    let bytes = source.as_bytes();
1430    let (start, end) = trim_target_range(bytes, start, end);
1431    if start >= end || bytes.get(start) != Some(&b'*') {
1432        return None;
1433    }
1434    let is_param_spec = bytes.get(start + 1) == Some(&b'*');
1435    let name_start = start + if is_param_spec { 2 } else { 1 };
1436    let name_end = identifier_end(bytes, name_start, end);
1437    if name_start == name_end {
1438        return None;
1439    }
1440    let colon = next_non_horizontal_whitespace(bytes, name_end);
1441    if colon >= end || bytes.get(colon) != Some(&b':') {
1442        return None;
1443    }
1444    let has_constraints = expression_slice_is_tuple(source, colon + 1, end);
1445    let message = match (is_param_spec, has_constraints) {
1446        (false, false) => "cannot use bound with TypeVarTuple",
1447        (false, true) => "cannot use constraints with TypeVarTuple",
1448        (true, false) => "cannot use bound with ParamSpec",
1449        (true, true) => "cannot use constraints with ParamSpec",
1450    };
1451    Some(CpythonDiagnostic::new(message.to_owned(), colon, colon + 1))
1452}
1453
1454fn invalid_type_param_list_error(
1455    source: &str,
1456    open: usize,
1457    close: usize,
1458) -> Option<CpythonDiagnostic> {
1459    let bytes = source.as_bytes();
1460    let mut item_start = open + 1;
1461    let mut index = item_start;
1462    let mut level = 0usize;
1463    while index <= close {
1464        if index == close || (level == 0 && bytes.get(index) == Some(&b',')) {
1465            if let Some(error) = invalid_type_param_item_error(source, item_start, index) {
1466                return Some(error);
1467            }
1468            item_start = index + 1;
1469            index += 1;
1470            continue;
1471        }
1472        match bytes[index] {
1473            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1474            b'(' | b'[' | b'{' => {
1475                level += 1;
1476                index += 1;
1477            }
1478            b')' | b']' | b'}' => {
1479                level = level.saturating_sub(1);
1480                index += 1;
1481            }
1482            _ => index += 1,
1483        }
1484    }
1485    None
1486}
1487
1488fn invalid_type_param_error(source: &str) -> Option<CpythonDiagnostic> {
1489    let bytes = source.as_bytes();
1490    let mut index = 0usize;
1491    while index < bytes.len() {
1492        match bytes[index] {
1493            b'#' => {
1494                while index < bytes.len() && bytes[index] != b'\n' {
1495                    index += 1;
1496                }
1497            }
1498            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1499            b'[' if type_param_list_open(bytes, index) => {
1500                let Some(close) = matching_delimiter(bytes, index, b']') else {
1501                    index += 1;
1502                    continue;
1503                };
1504                if let Some(error) = invalid_type_param_list_error(source, index, close) {
1505                    return Some(error);
1506                }
1507                index = close + 1;
1508            }
1509            _ => index += 1,
1510        }
1511    }
1512    None
1513}
1514
1515fn invalid_comprehension_in_slice(
1516    bytes: &[u8],
1517    open: usize,
1518    close: usize,
1519) -> Option<CpythonDiagnostic> {
1520    let for_index = find_keyword_at_level(bytes, open + 1, close, b"for")?;
1521    let item_start = next_non_horizontal_whitespace(bytes, open + 1);
1522    if item_start >= for_index {
1523        return None;
1524    }
1525    if bytes.get(item_start..item_start + 2) == Some(b"**") && bytes.get(open) == Some(&b'{') {
1526        return Some(CpythonDiagnostic::new(
1527            "dict unpacking cannot be used in dict comprehension".to_owned(),
1528            item_start,
1529            item_start + 2,
1530        ));
1531    }
1532    if bytes.get(item_start..item_start + 2) == Some(b"**") && bytes.get(open) == Some(&b'(') {
1533        return Some(CpythonDiagnostic::new(
1534            "invalid syntax".to_owned(),
1535            for_index,
1536            for_index + 3,
1537        ));
1538    }
1539    if bytes.get(item_start) == Some(&b'*') {
1540        return Some(CpythonDiagnostic::new(
1541            "iterable unpacking cannot be used in comprehension".to_owned(),
1542            item_start,
1543            item_start + 1,
1544        ));
1545    }
1546    if !matches!(bytes.get(open), Some(b'[' | b'{')) {
1547        return None;
1548    }
1549    if top_level_colon(bytes, open + 1, for_index).is_none()
1550        && let Some(comma) = top_level_byte(bytes, open + 1, for_index, b',')
1551    {
1552        let (start, _) = trim_target_range(bytes, open + 1, comma);
1553        return Some(CpythonDiagnostic::new(
1554            "did you forget parentheses around the comprehension target?".to_owned(),
1555            start,
1556            comma + 1,
1557        ));
1558    }
1559    None
1560}
1561
1562fn invalid_comprehension_error(source: &str) -> Option<CpythonDiagnostic> {
1563    let bytes = source.as_bytes();
1564    let mut index = 0usize;
1565    while index < bytes.len() {
1566        match bytes[index] {
1567            b'#' => {
1568                while index < bytes.len() && bytes[index] != b'\n' {
1569                    index += 1;
1570                }
1571            }
1572            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1573            b'(' | b'[' | b'{' => {
1574                let close_byte = match bytes[index] {
1575                    b'(' => b')',
1576                    b'[' => b']',
1577                    _ => b'}',
1578                };
1579                let Some(close) = matching_delimiter(bytes, index, close_byte) else {
1580                    index += 1;
1581                    continue;
1582                };
1583                if let Some(error) = invalid_comprehension_in_slice(bytes, index, close) {
1584                    return Some(error);
1585                }
1586                index = close + 1;
1587            }
1588            _ => index += 1,
1589        }
1590    }
1591    None
1592}
1593
1594fn invalid_group_in_slice(bytes: &[u8], open: usize, close: usize) -> Option<CpythonDiagnostic> {
1595    let (item_start, item_end) = trim_target_range(bytes, open + 1, close);
1596    if item_start >= item_end
1597        || top_level_byte(bytes, item_start, item_end, b',').is_some()
1598        || top_level_colon(bytes, item_start, item_end).is_some()
1599        || find_keyword_at_level(bytes, item_start, item_end, b"for").is_some()
1600    {
1601        return None;
1602    }
1603    if bytes.get(item_start..item_start + 2) == Some(b"**") {
1604        return Some(CpythonDiagnostic::new(
1605            "cannot use double starred expression here".to_owned(),
1606            item_start,
1607            item_start + 2,
1608        ));
1609    }
1610    if bytes.get(item_start) == Some(&b'*') {
1611        return Some(CpythonDiagnostic::new(
1612            "cannot use starred expression here".to_owned(),
1613            item_start,
1614            item_start + 1,
1615        ));
1616    }
1617    None
1618}
1619
1620fn invalid_group_error(source: &str) -> Option<CpythonDiagnostic> {
1621    let bytes = source.as_bytes();
1622    let mut index = 0usize;
1623    while index < bytes.len() {
1624        match bytes[index] {
1625            b'#' => {
1626                while index < bytes.len() && bytes[index] != b'\n' {
1627                    index += 1;
1628                }
1629            }
1630            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1631            b'(' => {
1632                let Some(close) = matching_delimiter(bytes, index, b')') else {
1633                    index += 1;
1634                    continue;
1635                };
1636                if let Some(error) = invalid_group_in_slice(bytes, index, close) {
1637                    return Some(error);
1638                }
1639                index = close + 1;
1640            }
1641            _ => index += 1,
1642        }
1643    }
1644    None
1645}
1646
1647fn invalid_parameter_star_annotation_error(source: &str) -> Option<CpythonDiagnostic> {
1648    let bytes = source.as_bytes();
1649    let mut index = 0usize;
1650    while index < bytes.len() {
1651        match bytes[index] {
1652            b'#' => {
1653                while index < bytes.len() && bytes[index] != b'\n' {
1654                    index += 1;
1655                }
1656            }
1657            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1658            b'(' => {
1659                let Some(close) = matching_delimiter(bytes, index, b')') else {
1660                    index += 1;
1661                    continue;
1662                };
1663                let mut param_start = index + 1;
1664                while param_start < close {
1665                    let param_end =
1666                        find_byte_at_level(bytes, param_start, close, b',').unwrap_or(close);
1667                    if let Some(colon) = top_level_colon(bytes, param_start, param_end) {
1668                        let value_start = next_non_horizontal_whitespace(bytes, colon + 1);
1669                        if bytes.get(value_start) == Some(&b'*') {
1670                            return Some(CpythonDiagnostic::new(
1671                                "invalid syntax".to_owned(),
1672                                value_start,
1673                                value_start + 1,
1674                            ));
1675                        }
1676                    }
1677                    param_start = param_end.saturating_add(1);
1678                }
1679                index = close + 1;
1680            }
1681            _ => index += 1,
1682        }
1683    }
1684    None
1685}
1686
1687fn invalid_def_type_params_error(source: &str) -> Option<CpythonDiagnostic> {
1688    let bytes = source.as_bytes();
1689    let mut index = 0;
1690    while index < bytes.len() {
1691        match bytes[index] {
1692            b'#' => {
1693                while index < bytes.len() && bytes[index] != b'\n' {
1694                    index += 1;
1695                }
1696            }
1697            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1698            _ if starts_identifier(bytes, index, b"def") => {
1699                let name_start = skip_horizontal_whitespace(bytes, index + 3);
1700                let name_end = identifier_end(bytes, name_start, bytes.len());
1701                let bracket = skip_horizontal_whitespace(bytes, name_end);
1702                if bytes.get(bracket) == Some(&b'[') {
1703                    let Some(close) = matching_delimiter(bytes, bracket, b']') else {
1704                        index = bracket + 1;
1705                        continue;
1706                    };
1707                    let after_close = skip_horizontal_whitespace(bytes, close + 1);
1708                    if bytes.get(after_close) == Some(&b'(')
1709                        && type_param_list_is_malformed(bytes, bracket + 1, close)
1710                    {
1711                        return Some(CpythonDiagnostic::new(
1712                            "expected '('".to_owned(),
1713                            bracket,
1714                            bracket + 1,
1715                        ));
1716                    }
1717                }
1718                index = name_end.max(index + 3);
1719            }
1720            _ => index += 1,
1721        }
1722    }
1723    None
1724}
1725
1726fn type_param_list_is_malformed(bytes: &[u8], start: usize, end: usize) -> bool {
1727    let mut index = start;
1728    let mut expect_item = true;
1729    while index < end {
1730        index = skip_horizontal_whitespace(bytes, index);
1731        if index >= end {
1732            break;
1733        }
1734        if bytes[index] == b',' {
1735            if expect_item {
1736                return true;
1737            }
1738            expect_item = true;
1739            index += 1;
1740            continue;
1741        }
1742        if !expect_item {
1743            return true;
1744        }
1745        if bytes.get(index..index + 2) == Some(b"**") {
1746            index += 2;
1747        } else if bytes.get(index) == Some(&b'*') {
1748            index += 1;
1749        }
1750        let item_start = skip_horizontal_whitespace(bytes, index);
1751        let item_end = identifier_end(bytes, item_start, end);
1752        if item_end == item_start {
1753            return true;
1754        }
1755        index = item_end;
1756        if bytes.get(skip_horizontal_whitespace(bytes, index)) == Some(&b':') {
1757            index = skip_horizontal_whitespace(bytes, index) + 1;
1758            while index < end && bytes[index] != b',' {
1759                index = match bytes[index] {
1760                    b'\'' | b'"' => skip_quoted_string(bytes, index),
1761                    _ => index + 1,
1762                };
1763            }
1764        }
1765        expect_item = false;
1766    }
1767    false
1768}
1769
1770fn invalid_parameter_list_slice_error(
1771    source: &str,
1772    start: usize,
1773    end: usize,
1774    kind: ParameterListKind,
1775) -> Option<CpythonDiagnostic> {
1776    let bytes = source.as_bytes();
1777    let mut index = start;
1778    let mut level = 0usize;
1779    let mut default_seen = false;
1780    let mut keyword_only = false;
1781    let mut slash_seen = false;
1782    let mut var_keyword_seen = false;
1783    while index < end {
1784        match bytes[index] {
1785            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1786            _ if level == 0
1787                && var_keyword_seen
1788                && bytes.get(index).is_some_and(|byte| {
1789                    *byte >= 0x80 || *byte == b'_' || byte.is_ascii_alphabetic()
1790                }) =>
1791            {
1792                let name_end = identifier_end(bytes, index, end);
1793                return Some(CpythonDiagnostic::new(
1794                    "arguments cannot follow var-keyword argument".to_owned(),
1795                    index,
1796                    name_end,
1797                ));
1798            }
1799            _ if level == 0
1800                && !keyword_only
1801                && bytes.get(index).is_some_and(|byte| {
1802                    *byte >= 0x80 || *byte == b'_' || byte.is_ascii_alphabetic()
1803                }) =>
1804            {
1805                let param_end = find_byte_at_level(bytes, index, end, b',')
1806                    .or_else(|| top_level_byte(bytes, index, end, b')'))
1807                    .or_else(|| {
1808                        matches!(kind, ParameterListKind::Lambda)
1809                            .then(|| top_level_byte(bytes, index, end, b':'))
1810                            .flatten()
1811                    })
1812                    .unwrap_or(end);
1813                let name_end = identifier_end(bytes, index, param_end);
1814                if let Some(eq) = top_level_byte(bytes, index, param_end, b'=') {
1815                    let value_start = next_non_horizontal_whitespace(bytes, eq + 1);
1816                    if value_start >= param_end
1817                        || matches!(bytes.get(value_start), Some(b',' | b')' | b':'))
1818                    {
1819                        let message = if matches!(kind, ParameterListKind::Lambda)
1820                            && matches!(bytes.get(value_start), Some(b':'))
1821                        {
1822                            "invalid syntax"
1823                        } else {
1824                            "expected default value expression"
1825                        };
1826                        return Some(CpythonDiagnostic::new(message.to_owned(), eq, eq + 1));
1827                    }
1828                    default_seen = true;
1829                } else if default_seen {
1830                    return Some(CpythonDiagnostic::new(
1831                        "parameter without a default follows parameter with a default".to_owned(),
1832                        index,
1833                        name_end,
1834                    ));
1835                }
1836                index = param_end;
1837            }
1838            b'(' if level == 0 => {
1839                let close = matching_delimiter(bytes, index, b')')
1840                    .filter(|close| *close <= end)
1841                    .unwrap_or(index + 1);
1842                let message = match kind {
1843                    ParameterListKind::Function => "Function parameters cannot be parenthesized",
1844                    ParameterListKind::Lambda => {
1845                        "Lambda expression parameters cannot be parenthesized"
1846                    }
1847                };
1848                return Some(CpythonDiagnostic::new(message.to_owned(), index, close + 1));
1849            }
1850            b'(' | b'[' | b'{' => {
1851                level += 1;
1852                index += 1;
1853            }
1854            b')' | b']' | b'}' => {
1855                level = level.saturating_sub(1);
1856                index += 1;
1857            }
1858            b'/' if level == 0 => {
1859                if var_keyword_seen {
1860                    return Some(CpythonDiagnostic::new(
1861                        "arguments cannot follow var-keyword argument".to_owned(),
1862                        index,
1863                        index + 1,
1864                    ));
1865                }
1866                if slash_seen {
1867                    return Some(CpythonDiagnostic::new(
1868                        "/ may appear only once".to_owned(),
1869                        index,
1870                        index + 1,
1871                    ));
1872                }
1873                slash_seen = true;
1874                let next = next_non_horizontal_whitespace(bytes, index + 1);
1875                if bytes.get(next) == Some(&b'*') {
1876                    return Some(CpythonDiagnostic::new(
1877                        "expected comma between / and *".to_owned(),
1878                        next,
1879                        next + 1,
1880                    ));
1881                }
1882                index += 1;
1883            }
1884            b'*' if level == 0 => {
1885                if var_keyword_seen {
1886                    return Some(CpythonDiagnostic::new(
1887                        "arguments cannot follow var-keyword argument".to_owned(),
1888                        index,
1889                        index + 1,
1890                    ));
1891                }
1892                keyword_only = true;
1893                let stars = usize::from(bytes.get(index + 1) == Some(&b'*')) + 1;
1894                let name_start = next_non_horizontal_whitespace(bytes, index + stars);
1895                for keyword in [b"True".as_slice(), b"False".as_slice(), b"None".as_slice()] {
1896                    if starts_identifier(bytes, name_start, keyword) {
1897                        return Some(CpythonDiagnostic::new(
1898                            "invalid syntax".to_owned(),
1899                            name_start,
1900                            name_start + keyword.len(),
1901                        ));
1902                    }
1903                }
1904                let param_end = find_byte_at_level(bytes, name_start, end, b',')
1905                    .or_else(|| top_level_byte(bytes, name_start, end, b')'))
1906                    .or_else(|| {
1907                        matches!(kind, ParameterListKind::Lambda)
1908                            .then(|| top_level_byte(bytes, name_start, end, b':'))
1909                            .flatten()
1910                    })
1911                    .unwrap_or(end);
1912                if stars == 1 && matches!(bytes.get(name_start), Some(b')' | b',' | b':')) {
1913                    return Some(CpythonDiagnostic::new(
1914                        "named arguments must follow bare *".to_owned(),
1915                        index,
1916                        index + 1,
1917                    ));
1918                }
1919                if stars == 1 && top_level_byte(bytes, name_start, param_end, b'=').is_some() {
1920                    return Some(CpythonDiagnostic::new(
1921                        "var-positional argument cannot have default value".to_owned(),
1922                        index,
1923                        index + 1,
1924                    ));
1925                }
1926                if stars == 2 && top_level_byte(bytes, name_start, param_end, b'=').is_some() {
1927                    return Some(CpythonDiagnostic::new(
1928                        "var-keyword argument cannot have default value".to_owned(),
1929                        index,
1930                        index + 2,
1931                    ));
1932                }
1933                if stars == 2 {
1934                    var_keyword_seen = true;
1935                    index = param_end;
1936                    continue;
1937                }
1938                index += stars;
1939            }
1940            b'=' if level == 0 => {
1941                let value_start = next_non_horizontal_whitespace(bytes, index + 1);
1942                if value_start >= end || matches!(bytes.get(value_start), Some(b',' | b')' | b':'))
1943                {
1944                    if matches!(kind, ParameterListKind::Lambda)
1945                        && matches!(bytes.get(value_start), Some(b':'))
1946                    {
1947                        return Some(CpythonDiagnostic::new(
1948                            "invalid syntax".to_owned(),
1949                            index,
1950                            index + 1,
1951                        ));
1952                    }
1953                    return Some(CpythonDiagnostic::new(
1954                        "expected default value expression".to_owned(),
1955                        index,
1956                        index + 1,
1957                    ));
1958                }
1959                index += 1;
1960            }
1961            _ => index += 1,
1962        }
1963    }
1964    None
1965}
1966
1967fn invalid_parameter_list_error(source: &str) -> Option<CpythonDiagnostic> {
1968    let bytes = source.as_bytes();
1969    let mut index = 0usize;
1970    while index < bytes.len() {
1971        match bytes[index] {
1972            b'#' => {
1973                while index < bytes.len() && bytes[index] != b'\n' {
1974                    index += 1;
1975                }
1976            }
1977            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1978            _ if starts_identifier(bytes, index, b"def") => {
1979                let Some(paren) = top_level_byte(bytes, index + 3, bytes.len(), b'(') else {
1980                    index += 3;
1981                    continue;
1982                };
1983                let Some(close) = matching_delimiter(bytes, paren, b')') else {
1984                    index = paren + 1;
1985                    continue;
1986                };
1987                if let Some(error) = invalid_parameter_list_slice_error(
1988                    source,
1989                    paren + 1,
1990                    close,
1991                    ParameterListKind::Function,
1992                ) {
1993                    return Some(error);
1994                }
1995                index = close + 1;
1996            }
1997            _ if starts_identifier(bytes, index, b"lambda") => {
1998                let params_start = index + 6;
1999                let Some(params_end) = find_lambda_parameter_end(bytes, params_start) else {
2000                    index = params_start;
2001                    continue;
2002                };
2003                if let Some(error) = invalid_parameter_list_slice_error(
2004                    source,
2005                    params_start,
2006                    params_end,
2007                    ParameterListKind::Lambda,
2008                ) {
2009                    return Some(error);
2010                }
2011                index = params_end + 1;
2012            }
2013            _ => index += 1,
2014        }
2015    }
2016    None
2017}
2018
2019#[derive(Clone, Copy)]
2020struct CallArgFrame {
2021    level: usize,
2022    arg_start: Option<usize>,
2023    in_call: bool,
2024}
2025
2026fn next_non_horizontal_whitespace(bytes: &[u8], mut index: usize) -> usize {
2027    while matches!(bytes.get(index), Some(b' ' | b'\t' | b'\x0c')) {
2028        index += 1;
2029    }
2030    index
2031}
2032
2033fn invalid_call_argument_assignment_error(
2034    source: &str,
2035    arg_start: usize,
2036    equal: usize,
2037) -> Option<CpythonDiagnostic> {
2038    let bytes = source.as_bytes();
2039    let start = bytes[arg_start..equal]
2040        .iter()
2041        .rposition(|byte| *byte == b'\n')
2042        .map_or(arg_start, |newline| arg_start + newline + 1);
2043    let (target_start, target_end) = trim_target_range(bytes, start, equal);
2044    if target_start >= target_end {
2045        return None;
2046    }
2047    let value_start = next_non_horizontal_whitespace(bytes, equal + 1);
2048    if matches!(bytes.get(value_start), None | Some(b',' | b')')) {
2049        return Some(CpythonDiagnostic::new(
2050            "expected argument value expression".to_owned(),
2051            target_start,
2052            equal + 1,
2053        ));
2054    }
2055    if bytes.get(target_start..target_start + 2) == Some(b"**") {
2056        return Some(CpythonDiagnostic::new(
2057            "cannot assign to keyword argument unpacking".to_owned(),
2058            target_start,
2059            value_start,
2060        ));
2061    }
2062    if bytes.get(target_start) == Some(&b'*') {
2063        return Some(CpythonDiagnostic::new(
2064            "cannot assign to iterable argument unpacking".to_owned(),
2065            target_start,
2066            value_start,
2067        ));
2068    }
2069    for keyword in [b"True".as_slice(), b"False".as_slice(), b"None".as_slice()] {
2070        if bytes.get(target_start..target_end) == Some(keyword) {
2071            let keyword = ::core::str::from_utf8(keyword).ok()?;
2072            return Some(CpythonDiagnostic::new(
2073                format!("cannot assign to {keyword}"),
2074                target_start,
2075                target_end,
2076            ));
2077        }
2078    }
2079    if is_simple_keyword_name(bytes, target_start, target_end) {
2080        // NAME '=' expression for_if_clauses
2081        if unparenthesized_comprehension(bytes, equal + 1) {
2082            return Some(CpythonDiagnostic::new(
2083                "invalid syntax. Maybe you meant '==' or ':=' instead of '='?".to_owned(),
2084                target_start,
2085                equal + 1,
2086            ));
2087        }
2088        return None;
2089    }
2090    Some(CpythonDiagnostic::new(
2091        "expression cannot contain assignment, perhaps you meant \"==\"?".to_owned(),
2092        target_start,
2093        equal,
2094    ))
2095}
2096
2097fn unparenthesized_comprehension(bytes: &[u8], mut index: usize) -> bool {
2098    index = next_non_horizontal_whitespace(bytes, index);
2099    if index >= bytes.len() {
2100        return false;
2101    }
2102    let start = index;
2103    let mut level = 0usize;
2104    while index < bytes.len() {
2105        match bytes[index] {
2106            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2107            b'(' | b'[' | b'{' => {
2108                level += 1;
2109                index += 1;
2110            }
2111            b')' | b']' | b'}' => {
2112                if level == 0 {
2113                    return false;
2114                }
2115                level -= 1;
2116                index += 1;
2117            }
2118            b',' | b':' if level == 0 => return false,
2119            _ if level == 0 && index > start && starts_identifier(bytes, index, b"for") => {
2120                return true;
2121            }
2122            _ => index += 1,
2123        }
2124    }
2125    false
2126}
2127
2128fn invalid_call_star_expression_error(
2129    bytes: &[u8],
2130    arg_start: usize,
2131    index: usize,
2132) -> Option<CpythonDiagnostic> {
2133    let start = next_non_horizontal_whitespace(bytes, arg_start);
2134    if start != index || bytes.get(index) != Some(&b'*') {
2135        return None;
2136    }
2137    let after_star = next_non_horizontal_whitespace(bytes, index + 1);
2138    if matches!(bytes.get(after_star), None | Some(b',' | b')' | b':')) {
2139        return Some(CpythonDiagnostic::new(
2140            "Invalid star expression".to_owned(),
2141            index,
2142            (index + 1).min(bytes.len()),
2143        ));
2144    }
2145    None
2146}
2147
2148fn invalid_call_argument_error(source: &str) -> Option<CpythonDiagnostic> {
2149    let bytes = source.as_bytes();
2150    let mut index = 0usize;
2151    let mut level = 0usize;
2152    let mut frames: Vec<CallArgFrame> = Vec::new();
2153    while index < bytes.len() {
2154        match bytes[index] {
2155            b'#' => {
2156                while index < bytes.len() && bytes[index] != b'\n' {
2157                    index += 1;
2158                }
2159            }
2160            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2161            _ if starts_identifier(bytes, index, b"lambda") => {
2162                let params_start = index + 6;
2163                if let Some(params_end) = find_lambda_parameter_end(bytes, params_start) {
2164                    index = params_end + 1;
2165                } else {
2166                    index = params_start;
2167                }
2168            }
2169            b'(' => {
2170                level += 1;
2171                let in_call = opening_paren_is_call(bytes, index)
2172                    || frames.last().is_some_and(|frame| frame.in_call);
2173                frames.push(CallArgFrame {
2174                    level,
2175                    arg_start: (in_call && !is_function_parameter_list(bytes, index))
2176                        .then_some(index + 1),
2177                    in_call,
2178                });
2179                index += 1;
2180            }
2181            b')' => {
2182                if matches!(frames.last(), Some(frame) if frame.level == level) {
2183                    frames.pop();
2184                }
2185                level = level.saturating_sub(1);
2186                index += 1;
2187            }
2188            b'[' | b'{' => {
2189                level += 1;
2190                index += 1;
2191            }
2192            b']' | b'}' => {
2193                level = level.saturating_sub(1);
2194                index += 1;
2195            }
2196            b',' => {
2197                if let Some(frame) = frames.last_mut()
2198                    && frame.level == level
2199                    && frame.arg_start.is_some()
2200                {
2201                    frame.arg_start = Some(index + 1);
2202                }
2203                index += 1;
2204            }
2205            b'*' => {
2206                if let Some(CallArgFrame {
2207                    level: frame_level,
2208                    arg_start: Some(arg_start),
2209                    in_call: true,
2210                }) = frames.last().copied()
2211                    && frame_level == level
2212                    && let Some(error) = invalid_call_star_expression_error(bytes, arg_start, index)
2213                {
2214                    return Some(error);
2215                }
2216                index += 1;
2217            }
2218            b'=' if is_plain_assignment_operator(bytes, index) => {
2219                if let Some(CallArgFrame {
2220                    level: frame_level,
2221                    arg_start: Some(arg_start),
2222                    in_call: true,
2223                }) = frames.last().copied()
2224                    && frame_level == level
2225                    && let Some(error) =
2226                        invalid_call_argument_assignment_error(source, arg_start, index)
2227                {
2228                    return Some(error);
2229                }
2230                index += 1;
2231            }
2232            _ => index += 1,
2233        }
2234    }
2235    None
2236}
2237
2238fn top_level_colon(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
2239    let mut level = 0usize;
2240    while index < end {
2241        match bytes[index] {
2242            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2243            b'(' | b'[' | b'{' => {
2244                level += 1;
2245                index += 1;
2246            }
2247            b')' | b']' | b'}' => {
2248                level = level.saturating_sub(1);
2249                index += 1;
2250            }
2251            b':' if level == 0 => return Some(index),
2252            _ => index += 1,
2253        }
2254    }
2255    None
2256}
2257
2258fn expression_slice_is_valid(source: &str, start: usize, end: usize) -> bool {
2259    let bytes = source.as_bytes();
2260    let (start, end) = trim_target_range(bytes, start, end);
2261    start < end
2262        && parser::parse(&source[start..end], parser::Mode::Expression.into())
2263            .is_ok_and(|parsed| matches!(parsed.into_syntax(), ast::Mod::Expression(_)))
2264}
2265
2266fn invalid_dict_entry_error(
2267    source: &str,
2268    item_start: usize,
2269    item_end: usize,
2270    colon: Option<usize>,
2271    saw_dict_item: bool,
2272) -> Option<CpythonDiagnostic> {
2273    let bytes = source.as_bytes();
2274    let (item_start, item_end) = trim_target_range(bytes, item_start, item_end);
2275    if item_start >= item_end {
2276        return None;
2277    }
2278    if let Some(colon) = colon {
2279        let value_start = next_non_horizontal_whitespace(bytes, colon + 1);
2280        if value_start >= item_end {
2281            return Some(CpythonDiagnostic::new(
2282                "expression expected after dictionary key and ':'".to_owned(),
2283                colon,
2284                colon + 1,
2285            ));
2286        }
2287        if bytes.get(value_start) == Some(&b'*') {
2288            return Some(CpythonDiagnostic::new(
2289                "cannot use a starred expression in a dictionary value".to_owned(),
2290                value_start,
2291                value_start + 1,
2292            ));
2293        }
2294        if !expression_slice_is_valid(source, value_start, item_end) {
2295            return Some(CpythonDiagnostic::new(
2296                "invalid syntax".to_owned(),
2297                value_start,
2298                value_start,
2299            ));
2300        }
2301    } else if saw_dict_item {
2302        return Some(CpythonDiagnostic::new(
2303            "':' expected after dictionary key".to_owned(),
2304            item_end.saturating_sub(1),
2305            item_end,
2306        ));
2307    }
2308    None
2309}
2310
2311fn invalid_dict_literal_error(
2312    source: &str,
2313    open: usize,
2314    close: usize,
2315) -> Option<CpythonDiagnostic> {
2316    let bytes = source.as_bytes();
2317    let mut item_start = open + 1;
2318    let mut index = item_start;
2319    let mut level = 0usize;
2320    let mut saw_dict_item = false;
2321    let mut item_colon = None;
2322    while index <= close {
2323        if index == close || (level == 0 && bytes.get(index) == Some(&b',')) {
2324            if let Some(error) =
2325                invalid_dict_entry_error(source, item_start, index, item_colon, saw_dict_item)
2326            {
2327                return Some(error);
2328            }
2329            saw_dict_item |= item_colon.is_some();
2330            item_start = index + 1;
2331            item_colon = None;
2332            index += 1;
2333            continue;
2334        }
2335        match bytes[index] {
2336            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2337            b'(' | b'[' | b'{' => {
2338                level += 1;
2339                index += 1;
2340            }
2341            b')' | b']' | b'}' => {
2342                level = level.saturating_sub(1);
2343                index += 1;
2344            }
2345            b':' if level == 0 && item_colon.is_none() => {
2346                item_colon = Some(index);
2347                saw_dict_item = true;
2348                index += 1;
2349            }
2350            _ => index += 1,
2351        }
2352    }
2353    None
2354}
2355
2356fn invalid_dict_error(source: &str) -> Option<CpythonDiagnostic> {
2357    let bytes = source.as_bytes();
2358    let mut index = 0usize;
2359    while index < bytes.len() {
2360        match bytes[index] {
2361            b'#' => {
2362                while index < bytes.len() && bytes[index] != b'\n' {
2363                    index += 1;
2364                }
2365            }
2366            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2367            b'{' => {
2368                let Some(close) = matching_delimiter(bytes, index, b'}') else {
2369                    index += 1;
2370                    continue;
2371                };
2372                if top_level_colon(bytes, index + 1, close).is_some()
2373                    && let Some(error) = invalid_dict_literal_error(source, index, close)
2374                {
2375                    return Some(error);
2376                }
2377                index = close + 1;
2378            }
2379            _ => index += 1,
2380        }
2381    }
2382    None
2383}
2384
2385fn collection_open_is_call(bytes: &[u8], open: usize) -> bool {
2386    if bytes.get(open) != Some(&b'(') {
2387        return false;
2388    }
2389    let mut cursor = open;
2390    while cursor > 0 && matches!(bytes.get(cursor - 1), Some(b' ' | b'\t' | b'\x0c')) {
2391        cursor -= 1;
2392    }
2393    matches!(
2394        cursor.checked_sub(1).and_then(|before| bytes.get(before)),
2395        Some(b')' | b']' | b'_' | b'a'..=b'z' | b'A'..=b'Z' | 0x80..=0xff)
2396    )
2397}
2398
2399fn invalid_collection_assignment_in_slice(
2400    source: &str,
2401    bytes: &[u8],
2402    start: usize,
2403    end: usize,
2404) -> Option<CpythonDiagnostic> {
2405    let mut item_start = start;
2406    let mut index = start;
2407    let mut level = 0usize;
2408    while index < end {
2409        match bytes[index] {
2410            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2411            b'(' | b'[' | b'{' => {
2412                level += 1;
2413                index += 1;
2414            }
2415            b')' | b']' | b'}' => {
2416                level = level.saturating_sub(1);
2417                index += 1;
2418            }
2419            b',' if level == 0 => {
2420                item_start = index + 1;
2421                index += 1;
2422            }
2423            b'=' if level == 0 && is_plain_assignment_operator(bytes, index) => {
2424                if top_level_colon(bytes, item_start, index).is_none() {
2425                    let start = next_non_horizontal_whitespace(bytes, item_start);
2426                    let target_end = trim_end_horizontal_whitespace(bytes, start, index);
2427                    if start < target_end
2428                        && let Some((expr_name, expr_start, expr_end, _)) =
2429                            expression_name_and_range(&source[start..target_end])
2430                    {
2431                        if matches!(expr_name, "list" | "tuple") {
2432                            return None;
2433                        }
2434                        if matches!(expr_name, "expression" | "attribute" | "subscript") {
2435                            return Some(CpythonDiagnostic::new(
2436                                format!(
2437                                    "cannot assign to {expr_name} here. Maybe you meant '==' instead of '='?"
2438                                ),
2439                                start + expr_start,
2440                                start + expr_end,
2441                            ));
2442                        }
2443                    }
2444                    return Some(CpythonDiagnostic::new(
2445                        "invalid syntax. Maybe you meant '==' or ':=' instead of '='?".to_owned(),
2446                        start,
2447                        index + 1,
2448                    ));
2449                }
2450                index += 1;
2451            }
2452            _ => index += 1,
2453        }
2454    }
2455    None
2456}
2457
2458fn invalid_collection_assignment_error(source: &str) -> Option<CpythonDiagnostic> {
2459    let bytes = source.as_bytes();
2460    let mut index = 0usize;
2461    while index < bytes.len() {
2462        match bytes[index] {
2463            b'#' => {
2464                while index < bytes.len() && bytes[index] != b'\n' {
2465                    index += 1;
2466                }
2467            }
2468            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2469            b'(' | b'[' | b'{' => {
2470                let close_byte = match bytes[index] {
2471                    b'(' => b')',
2472                    b'[' => b']',
2473                    _ => b'}',
2474                };
2475                let Some(close) = matching_delimiter(bytes, index, close_byte) else {
2476                    index += 1;
2477                    continue;
2478                };
2479                if !collection_open_is_call(bytes, index)
2480                    && let Some(error) =
2481                        invalid_collection_assignment_in_slice(source, bytes, index + 1, close)
2482                {
2483                    return Some(error);
2484                }
2485                index = close + 1;
2486            }
2487            _ => index += 1,
2488        }
2489    }
2490    None
2491}
2492
2493fn expression_assignment_error(source: &str) -> Option<CpythonDiagnostic> {
2494    let bytes = source.as_bytes();
2495    let mut index = 0;
2496    let mut paren_arg_starts: Vec<(Option<usize>, bool)> = Vec::new();
2497    while index < bytes.len() {
2498        match bytes[index] {
2499            b'#' => {
2500                while index < bytes.len() && bytes[index] != b'\n' {
2501                    index += 1;
2502                }
2503            }
2504            b'\'' | b'"' => {
2505                index = skip_quoted_string(bytes, index);
2506            }
2507            _ if starts_identifier(bytes, index, b"lambda") => {
2508                let params_start = index + 6;
2509                if let Some(params_end) = find_lambda_parameter_end(bytes, params_start) {
2510                    index = params_end + 1;
2511                } else {
2512                    index = params_start;
2513                }
2514            }
2515            b'(' => {
2516                let in_call_context = opening_paren_is_call(bytes, index)
2517                    || paren_arg_starts.last().is_some_and(|(_, in_call)| *in_call);
2518                paren_arg_starts.push((
2519                    (!is_function_parameter_list(bytes, index)).then_some(index + 1),
2520                    in_call_context,
2521                ));
2522                index += 1;
2523            }
2524            b')' => {
2525                paren_arg_starts.pop();
2526                index += 1;
2527            }
2528            b',' => {
2529                if let Some((start, _)) = paren_arg_starts.last_mut()
2530                    && start.is_some()
2531                {
2532                    *start = Some(index + 1);
2533                }
2534                index += 1;
2535            }
2536            b'=' if is_plain_assignment_operator(bytes, index) => {
2537                if let Some((Some(start), true)) = paren_arg_starts.last().copied()
2538                    && !is_simple_keyword_name(bytes, start, index)
2539                {
2540                    let mut expr_start = start;
2541                    while matches!(bytes.get(expr_start), Some(b' ' | b'\t' | b'\x0c')) {
2542                        expr_start += 1;
2543                    }
2544                    return Some(CpythonDiagnostic::new(
2545                        "expression cannot contain assignment, perhaps you meant \"==\"?"
2546                            .to_owned(),
2547                        expr_start,
2548                        index,
2549                    ));
2550                }
2551                index += 1;
2552            }
2553            _ => index += 1,
2554        }
2555    }
2556    None
2557}
2558
2559fn invalid_named_expression_error(source: &str) -> Option<CpythonDiagnostic> {
2560    let bytes = source.as_bytes();
2561    let mut index = 0;
2562    while index + 1 < bytes.len() {
2563        match bytes[index] {
2564            b'#' => {
2565                while index < bytes.len() && bytes[index] != b'\n' {
2566                    index += 1;
2567                }
2568            }
2569            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2570            b':' if bytes.get(index + 1) == Some(&b'=') => {
2571                let target_start = named_expression_target_start(bytes, index);
2572                let target_end = trim_end_horizontal_whitespace(bytes, target_start, index);
2573                if target_start < target_end
2574                    && let Some((expr_name, start, end, is_name)) =
2575                        expression_name_and_range(&source[target_start..target_end])
2576                    && !is_name
2577                {
2578                    return Some(CpythonDiagnostic::new(
2579                        format!("cannot use assignment expressions with {expr_name}"),
2580                        target_start + start,
2581                        target_start + end,
2582                    ));
2583                }
2584                index += 2;
2585            }
2586            _ => index += 1,
2587        }
2588    }
2589    None
2590}
2591
2592#[derive(Clone, Copy)]
2593struct AssignmentContext {
2594    start: usize,
2595    call: bool,
2596}
2597
2598fn invalid_plain_assignment_error(source: &str) -> Option<CpythonDiagnostic> {
2599    let bytes = source.as_bytes();
2600    let mut stack: Vec<AssignmentContext> = Vec::new();
2601    let mut index = 0;
2602    while index < bytes.len() {
2603        match bytes[index] {
2604            b'#' => {
2605                while index < bytes.len() && bytes[index] != b'\n' {
2606                    index += 1;
2607                }
2608            }
2609            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2610            b'(' | b'[' | b'{' => {
2611                stack.push(AssignmentContext {
2612                    start: index + 1,
2613                    call: bytes[index] == b'(' && opening_paren_is_call(bytes, index),
2614                });
2615                index += 1;
2616            }
2617            b')' | b']' | b'}' => {
2618                stack.pop();
2619                index += 1;
2620            }
2621            b',' => {
2622                if let Some(context) = stack.last_mut()
2623                    && !context.call
2624                {
2625                    context.start = index + 1;
2626                }
2627                index += 1;
2628            }
2629            b'=' if is_plain_assignment_operator(bytes, index) => {
2630                if let Some(context) = stack.last().copied()
2631                    && !context.call
2632                {
2633                    let target_start = skip_horizontal_whitespace(bytes, context.start);
2634                    let target_end = trim_end_horizontal_whitespace(bytes, target_start, index);
2635                    if target_start < target_end
2636                        && let Some((expr_name, start, end, _)) =
2637                            expression_name_and_range(&source[target_start..target_end])
2638                        && matches!(expr_name, "expression" | "attribute" | "subscript")
2639                    {
2640                        return Some(CpythonDiagnostic::new(
2641                            format!(
2642                                "cannot assign to {expr_name} here. Maybe you meant '==' instead of '='?"
2643                            ),
2644                            target_start + start,
2645                            target_start + end,
2646                        ));
2647                    }
2648                }
2649                index += 1;
2650            }
2651            _ => index += 1,
2652        }
2653    }
2654    None
2655}
2656
2657fn opening_paren_is_call(bytes: &[u8], paren: usize) -> bool {
2658    let mut cursor = paren;
2659    while cursor > 0 && matches!(bytes[cursor - 1], b' ' | b'\t' | b'\x0c') {
2660        cursor -= 1;
2661    }
2662    cursor > 0
2663        && (bytes[cursor - 1] >= 0x80
2664            || is_ascii_identifier_char(bytes[cursor - 1])
2665            || matches!(bytes[cursor - 1], b')' | b']'))
2666}
2667
2668fn named_expression_target_start(bytes: &[u8], walrus: usize) -> usize {
2669    let mut index = walrus;
2670    let mut level = 0usize;
2671    while index > 0 {
2672        index -= 1;
2673        match bytes[index] {
2674            b')' | b']' | b'}' => level += 1,
2675            b'(' | b'[' | b'{' if level > 0 => level -= 1,
2676            b'(' | b'[' | b'{' if level == 0 => return index + 1,
2677            b',' | b'\n' | b';' if level == 0 => return index + 1,
2678            _ => {}
2679        }
2680    }
2681    0
2682}
2683
2684fn trim_end_horizontal_whitespace(bytes: &[u8], start: usize, mut end: usize) -> usize {
2685    while end > start && matches!(bytes[end - 1], b' ' | b'\t' | b'\x0c') {
2686        end -= 1;
2687    }
2688    end
2689}
2690
2691fn annotation_target_error_for_slice(
2692    source: &str,
2693    start: usize,
2694    colon: usize,
2695) -> Option<CpythonDiagnostic> {
2696    let bytes = source.as_bytes();
2697    let (target_start, target_end) = trim_target_range(bytes, start, colon);
2698    if target_start >= target_end {
2699        return None;
2700    }
2701    let target_text = &source[target_start..target_end];
2702    let Ok(parsed) = parser::parse(target_text, parser::Mode::Expression.into()) else {
2703        return None;
2704    };
2705    let ast::Mod::Expression(expression) = parsed.into_syntax() else {
2706        return None;
2707    };
2708    match expression.body.as_ref() {
2709        ast::Expr::Name(_) | ast::Expr::Attribute(_) | ast::Expr::Subscript(_) => None,
2710        ast::Expr::List(_) => Some(CpythonDiagnostic::new(
2711            "only single target (not list) can be annotated".to_owned(),
2712            target_start,
2713            target_end,
2714        )),
2715        ast::Expr::Tuple(_) => Some(CpythonDiagnostic::new(
2716            "only single target (not tuple) can be annotated".to_owned(),
2717            target_start,
2718            target_end,
2719        )),
2720        _ => Some(CpythonDiagnostic::new(
2721            "illegal target for annotation".to_owned(),
2722            target_start,
2723            target_end,
2724        )),
2725    }
2726}
2727
2728fn invalid_annotation_line_start(bytes: &[u8], line_start: usize) -> bool {
2729    let column = skip_horizontal_whitespace(bytes, line_start);
2730    for keyword in [
2731        b"async".as_slice(),
2732        b"case",
2733        b"class",
2734        b"def",
2735        b"elif",
2736        b"else",
2737        b"except",
2738        b"finally",
2739        b"for",
2740        b"if",
2741        b"match",
2742        b"try",
2743        b"while",
2744        b"with",
2745    ] {
2746        if starts_identifier(bytes, column, keyword) {
2747            return false;
2748        }
2749    }
2750    true
2751}
2752
2753fn invalid_annotation_target_error(source: &str) -> Option<CpythonDiagnostic> {
2754    let bytes = source.as_bytes();
2755    let mut line_start = 0usize;
2756    for line in source.split_inclusive('\n') {
2757        let line_end = line_start + line.len();
2758        if invalid_annotation_line_start(bytes, line_start)
2759            && let Some(colon) = find_byte_at_level(bytes, line_start, line_end, b':')
2760            && bytes.get(colon + 1) != Some(&b'=')
2761            && colon.checked_sub(1).and_then(|before| bytes.get(before)) != Some(&b':')
2762            && let Some(error) = annotation_target_error_for_slice(source, line_start, colon)
2763        {
2764            return Some(error);
2765        }
2766        line_start = line_end;
2767    }
2768    None
2769}
2770
2771fn statement_target_end(bytes: &[u8], mut index: usize) -> usize {
2772    let mut level = 0usize;
2773    while index < bytes.len() {
2774        match bytes[index] {
2775            b'#' if level == 0 => return index,
2776            b'\n' | b';' if level == 0 => return index,
2777            b'\'' | b'"' => {
2778                index = skip_quoted_string(bytes, index);
2779            }
2780            b'(' | b'[' | b'{' => {
2781                level += 1;
2782                index += 1;
2783            }
2784            b')' | b']' | b'}' => {
2785                level = level.saturating_sub(1);
2786                index += 1;
2787            }
2788            _ => index += 1,
2789        }
2790    }
2791    index
2792}
2793
2794fn invalid_assignment_target(expression: &ast::Expr) -> Option<&ast::Expr> {
2795    match expression {
2796        ast::Expr::List(ast::ExprList { elts, .. })
2797        | ast::Expr::Tuple(ast::ExprTuple { elts, .. }) => {
2798            elts.iter().find_map(invalid_assignment_target)
2799        }
2800        ast::Expr::Starred(ast::ExprStarred { value, .. }) => invalid_assignment_target(value),
2801        ast::Expr::Name(_) | ast::Expr::Subscript(_) | ast::Expr::Attribute(_) => None,
2802        _ => Some(expression),
2803    }
2804}
2805
2806fn invalid_for_target(expression: &ast::Expr) -> Option<&ast::Expr> {
2807    match expression {
2808        ast::Expr::List(ast::ExprList { elts, .. })
2809        | ast::Expr::Tuple(ast::ExprTuple { elts, .. }) => elts.iter().find_map(invalid_for_target),
2810        ast::Expr::Starred(ast::ExprStarred { value, .. }) => invalid_for_target(value),
2811        ast::Expr::Compare(ast::ExprCompare { left, ops, .. }) => {
2812            if matches!(ops.first(), Some(ast::CmpOp::In)) {
2813                invalid_for_target(left)
2814            } else {
2815                None
2816            }
2817        }
2818        ast::Expr::Name(_) | ast::Expr::Subscript(_) | ast::Expr::Attribute(_) => None,
2819        _ => Some(expression),
2820    }
2821}
2822
2823fn invalid_delete_target(expression: &ast::Expr) -> Option<&ast::Expr> {
2824    match expression {
2825        ast::Expr::List(ast::ExprList { elts, .. })
2826        | ast::Expr::Tuple(ast::ExprTuple { elts, .. }) => {
2827            elts.iter().find_map(invalid_delete_target)
2828        }
2829        ast::Expr::Name(_) | ast::Expr::Subscript(_) | ast::Expr::Attribute(_) => None,
2830        ast::Expr::Starred(_) => Some(expression),
2831        ast::Expr::Compare(_) => Some(expression),
2832        _ => Some(expression),
2833    }
2834}
2835
2836fn delete_target_expr_name(expression: &ast::Expr) -> &'static str {
2837    match expression {
2838        ast::Expr::Attribute(_) => "attribute",
2839        ast::Expr::Subscript(_) => "subscript",
2840        ast::Expr::Starred(_) => "starred",
2841        ast::Expr::Name(_) => "name",
2842        ast::Expr::List(_) => "list",
2843        ast::Expr::Tuple(_) => "tuple",
2844        ast::Expr::Lambda(_) => "lambda",
2845        ast::Expr::Call(_) => "function call",
2846        ast::Expr::BoolOp(_) | ast::Expr::BinOp(_) | ast::Expr::UnaryOp(_) => "expression",
2847        ast::Expr::Generator(_) => "generator expression",
2848        ast::Expr::Yield(_) | ast::Expr::YieldFrom(_) => "yield expression",
2849        ast::Expr::Await(_) => "await expression",
2850        ast::Expr::ListComp(_) => "list comprehension",
2851        ast::Expr::SetComp(_) => "set comprehension",
2852        ast::Expr::DictComp(_) => "dict comprehension",
2853        ast::Expr::Dict(_) => "dict literal",
2854        ast::Expr::Set(_) => "set display",
2855        ast::Expr::FString(_) => "f-string expression",
2856        ast::Expr::TString(_) => "t-string expression",
2857        ast::Expr::NumberLiteral(_) | ast::Expr::StringLiteral(_) | ast::Expr::BytesLiteral(_) => {
2858            "literal"
2859        }
2860        ast::Expr::Constant(expr) => match &expr.value {
2861            ast::ConstantValue::None => "None",
2862            ast::ConstantValue::Boolean(true) => "True",
2863            ast::ConstantValue::Boolean(false) => "False",
2864            ast::ConstantValue::Ellipsis => "ellipsis",
2865            ast::ConstantValue::Tuple(_) => "tuple",
2866            ast::ConstantValue::Frozenset(_) => "literal",
2867            ast::ConstantValue::Str(_)
2868            | ast::ConstantValue::Bytes(_)
2869            | ast::ConstantValue::Integer(_)
2870            | ast::ConstantValue::Float(_)
2871            | ast::ConstantValue::Complex { .. } => "literal",
2872        },
2873        ast::Expr::BooleanLiteral(boolean) => {
2874            if boolean.value {
2875                "True"
2876            } else {
2877                "False"
2878            }
2879        }
2880        ast::Expr::NoneLiteral(_) => "None",
2881        ast::Expr::EllipsisLiteral(_) => "ellipsis",
2882        ast::Expr::Compare(_) => "comparison",
2883        ast::Expr::If(_) => "conditional expression",
2884        ast::Expr::Named(_) => "named expression",
2885        ast::Expr::Slice(_) | ast::Expr::IpyEscapeCommand(_) => "expression",
2886    }
2887}
2888
2889fn parenthesized_single_starred_delete_target(bytes: &[u8], start: usize, end: usize) -> bool {
2890    let mut cursor = start;
2891    while matches!(bytes.get(cursor), Some(b' ' | b'\t' | b'\x0c')) {
2892        cursor += 1;
2893    }
2894    if bytes.get(cursor) != Some(&b'(') {
2895        return false;
2896    }
2897    cursor += 1;
2898    while matches!(bytes.get(cursor), Some(b' ' | b'\t' | b'\x0c')) {
2899        cursor += 1;
2900    }
2901    if bytes.get(cursor) != Some(&b'*') {
2902        return false;
2903    }
2904    let mut level = 1usize;
2905    cursor += 1;
2906    while cursor < end {
2907        match bytes[cursor] {
2908            b'\'' | b'"' => {
2909                cursor = skip_quoted_string(bytes, cursor);
2910            }
2911            b'(' | b'[' | b'{' => {
2912                level += 1;
2913                cursor += 1;
2914            }
2915            b')' => {
2916                level = level.saturating_sub(1);
2917                if level == 0 {
2918                    cursor += 1;
2919                    while matches!(bytes.get(cursor), Some(b' ' | b'\t' | b'\x0c')) {
2920                        cursor += 1;
2921                    }
2922                    return cursor == end;
2923                }
2924                cursor += 1;
2925            }
2926            b',' if level == 1 => return false,
2927            b']' | b'}' => {
2928                level = level.saturating_sub(1);
2929                cursor += 1;
2930            }
2931            _ => cursor += 1,
2932        }
2933    }
2934    false
2935}
2936
2937fn assignment_target_expr_range(source: &str, start: usize, end: usize) -> Option<(usize, usize)> {
2938    let bytes = source.as_bytes();
2939    let (target_start, target_end) = trim_target_range(bytes, start, end);
2940    if target_start >= target_end {
2941        return None;
2942    }
2943    if parser::parse(
2944        &source[target_start..target_end],
2945        parser::Mode::Expression.into(),
2946    )
2947    .is_ok()
2948    {
2949        return Some((target_start, target_end));
2950    }
2951    // `def f(): (yield bar)` — skip the suite header so the remaining
2952    // text is the assignment target expression. Other colons (`x: int += 1`)
2953    // are not suite headers for this diagnostic. Grouping parentheses may
2954    // wrap the yield expression.
2955    let colon = top_level_colon(bytes, target_start, target_end)?;
2956    let after = skip_horizontal_whitespace(bytes, colon + 1);
2957    let yield_at = skip_opening_parentheses(bytes, after, target_end);
2958    if yield_at >= target_end || !starts_identifier(bytes, yield_at, b"yield") {
2959        return None;
2960    }
2961    Some((after, target_end))
2962}
2963
2964fn skip_opening_parentheses(bytes: &[u8], mut index: usize, end: usize) -> usize {
2965    loop {
2966        index = skip_horizontal_whitespace(bytes, index);
2967        if index >= end || bytes[index] != b'(' {
2968            return index;
2969        }
2970        index += 1;
2971    }
2972}
2973
2974fn trim_target_range(bytes: &[u8], mut start: usize, mut end: usize) -> (usize, usize) {
2975    while start < end
2976        && matches!(
2977            bytes.get(start),
2978            Some(b' ' | b'\t' | b'\n' | b'\r' | b'\x0c')
2979        )
2980    {
2981        start += 1;
2982    }
2983    while end > start
2984        && matches!(
2985            bytes.get(end - 1),
2986            Some(b' ' | b'\t' | b'\n' | b'\r' | b'\x0c')
2987        )
2988    {
2989        end -= 1;
2990    }
2991    (start, end)
2992}
2993
2994fn invalid_assignment_message(name: &'static str, top_level_bitwise: bool) -> String {
2995    if top_level_bitwise {
2996        format!("cannot assign to {name} here. Maybe you meant '==' instead of '='?")
2997    } else {
2998        format!("cannot assign to {name}")
2999    }
3000}
3001
3002fn assignment_target_error_for_slice(
3003    source: &str,
3004    start: usize,
3005    end: usize,
3006) -> Option<CpythonDiagnostic> {
3007    let bytes = source.as_bytes();
3008    let (target_start, target_end) = assignment_target_expr_range(source, start, end)?;
3009    if starts_identifier(bytes, target_start, b"yield") {
3010        return Some(CpythonDiagnostic::new(
3011            "assignment to yield expression not possible".to_owned(),
3012            target_start,
3013            target_start + 5,
3014        ));
3015    }
3016    let target_text = &source[target_start..target_end];
3017    let Ok(parsed) = parser::parse(target_text, parser::Mode::Expression.into()) else {
3018        return None;
3019    };
3020    let ast::Mod::Expression(expression) = parsed.into_syntax() else {
3021        return None;
3022    };
3023    let invalid_target = invalid_assignment_target(&expression.body)?;
3024    let invalid_start = target_start + invalid_target.range().start().to_usize();
3025    let invalid_end = target_start + invalid_target.range().end().to_usize();
3026    let name = delete_target_expr_name(invalid_target);
3027    let top_level = invalid_target.range() == expression.body.range();
3028    let bitwise_like = matches!(
3029        invalid_target,
3030        ast::Expr::Call(_)
3031            | ast::Expr::BoolOp(_)
3032            | ast::Expr::BinOp(_)
3033            | ast::Expr::UnaryOp(_)
3034            | ast::Expr::NumberLiteral(_)
3035            | ast::Expr::StringLiteral(_)
3036            | ast::Expr::BytesLiteral(_)
3037            | ast::Expr::EllipsisLiteral(_)
3038            | ast::Expr::Yield(_)
3039            | ast::Expr::YieldFrom(_)
3040            | ast::Expr::Set(_)
3041            | ast::Expr::Dict(_)
3042            | ast::Expr::FString(_)
3043            | ast::Expr::TString(_)
3044    );
3045    Some(CpythonDiagnostic::new(
3046        invalid_assignment_message(name, top_level && bitwise_like),
3047        invalid_start,
3048        invalid_end,
3049    ))
3050}
3051
3052fn star_target_error_for_slice(
3053    source: &str,
3054    start: usize,
3055    end: usize,
3056) -> Option<CpythonDiagnostic> {
3057    invalid_target_error_for_slice(source, start, end, invalid_assignment_target)
3058}
3059
3060fn for_target_error_for_slice(source: &str, start: usize, end: usize) -> Option<CpythonDiagnostic> {
3061    invalid_target_error_for_slice(source, start, end, invalid_for_target)
3062}
3063
3064fn invalid_target_error_for_slice(
3065    source: &str,
3066    start: usize,
3067    end: usize,
3068    invalid_target: for<'a> fn(&'a ast::Expr) -> Option<&'a ast::Expr>,
3069) -> Option<CpythonDiagnostic> {
3070    let bytes = source.as_bytes();
3071    let (target_start, target_end) = trim_target_range(bytes, start, end);
3072    if target_start >= target_end {
3073        return None;
3074    }
3075    let target_text = &source[target_start..target_end];
3076    let Ok(parsed) = parser::parse(target_text, parser::Mode::Expression.into()) else {
3077        return None;
3078    };
3079    let ast::Mod::Expression(expression) = parsed.into_syntax() else {
3080        return None;
3081    };
3082    let invalid_target = invalid_target(&expression.body)?;
3083    let name = delete_target_expr_name(invalid_target);
3084    let invalid_start = target_start + invalid_target.range().start().to_usize();
3085    let invalid_end = target_start + invalid_target.range().end().to_usize();
3086    Some(CpythonDiagnostic::new(
3087        format!("cannot assign to {name}"),
3088        invalid_start,
3089        invalid_end,
3090    ))
3091}
3092
3093fn first_compare_operator_at_level(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
3094    let mut level = 0usize;
3095    while index < end {
3096        match bytes[index] {
3097            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3098            b'(' | b'[' | b'{' => {
3099                level += 1;
3100                index += 1;
3101            }
3102            b')' | b']' | b'}' => {
3103                level = level.saturating_sub(1);
3104                index += 1;
3105            }
3106            b'<' | b'>' if level == 0 => return Some(index),
3107            b'=' if level == 0 && bytes.get(index + 1) == Some(&b'=') => return Some(index),
3108            b'!' if level == 0 && bytes.get(index + 1) == Some(&b'=') => return Some(index),
3109            _ if level == 0 && starts_identifier(bytes, index, b"is") => return Some(index),
3110            _ if level == 0 && starts_identifier(bytes, index, b"not") => return Some(index),
3111            _ => index += 1,
3112        }
3113    }
3114    None
3115}
3116
3117fn non_in_compare_for_target_error(
3118    source: &str,
3119    start: usize,
3120    end: usize,
3121) -> Option<CpythonDiagnostic> {
3122    let bytes = source.as_bytes();
3123    let (target_start, target_end) = trim_target_range(bytes, start, end);
3124    if target_start >= target_end {
3125        return None;
3126    }
3127    let target_text = &source[target_start..target_end];
3128    let Ok(parsed) = parser::parse(target_text, parser::Mode::Expression.into()) else {
3129        return None;
3130    };
3131    let ast::Mod::Expression(expression) = parsed.into_syntax() else {
3132        return None;
3133    };
3134    let ast::Expr::Compare(ast::ExprCompare { ops, .. }) = expression.body.as_ref() else {
3135        return None;
3136    };
3137    if matches!(ops.first(), Some(ast::CmpOp::In)) {
3138        return None;
3139    }
3140    let operator = first_compare_operator_at_level(bytes, target_start, target_end)?;
3141    Some(CpythonDiagnostic::new(
3142        "invalid syntax".to_owned(),
3143        operator,
3144        (operator + 1).min(target_end),
3145    ))
3146}
3147
3148fn top_level_plain_assignment_offsets(bytes: &[u8]) -> Vec<usize> {
3149    let mut offsets = Vec::new();
3150    let mut index = 0usize;
3151    let mut level = 0usize;
3152    while index < bytes.len() {
3153        match bytes[index] {
3154            b'#' if level == 0 => {
3155                while index < bytes.len() && bytes[index] != b'\n' {
3156                    index += 1;
3157                }
3158            }
3159            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3160            b'(' | b'[' | b'{' => {
3161                level += 1;
3162                index += 1;
3163            }
3164            b')' | b']' | b'}' => {
3165                level = level.saturating_sub(1);
3166                index += 1;
3167            }
3168            b'=' if level == 0 && is_plain_assignment_operator(bytes, index) => {
3169                offsets.push(index);
3170                index += 1;
3171            }
3172            _ => index += 1,
3173        }
3174    }
3175    offsets
3176}
3177
3178fn invalid_condition_assignment_error(
3179    source: &str,
3180    parse_error_offset: usize,
3181) -> Option<CpythonDiagnostic> {
3182    let bytes = source.as_bytes();
3183    let mut index = 0usize;
3184    let mut line_start = 0usize;
3185    let mut level = 0usize;
3186    while index < bytes.len() {
3187        match bytes[index] {
3188            b'#' if level == 0 => {
3189                while index < bytes.len() && bytes[index] != b'\n' {
3190                    index += 1;
3191                }
3192            }
3193            b'\n' => {
3194                line_start = index + 1;
3195                index += 1;
3196            }
3197            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3198            b'(' | b'[' | b'{' => {
3199                level += 1;
3200                index += 1;
3201            }
3202            b')' | b']' | b'}' => {
3203                level = level.saturating_sub(1);
3204                index += 1;
3205            }
3206            _ if level == 0
3207                && index == skip_horizontal_whitespace(bytes, line_start)
3208                && (starts_identifier(bytes, index, b"if")
3209                    || starts_identifier(bytes, index, b"elif")
3210                    || starts_identifier(bytes, index, b"while")) =>
3211            {
3212                let keyword_len = if starts_identifier(bytes, index, b"while") {
3213                    5
3214                } else if starts_identifier(bytes, index, b"elif") {
3215                    4
3216                } else {
3217                    2
3218                };
3219                let cond_start = skip_horizontal_whitespace(bytes, index + keyword_len);
3220                let Some(colon) = top_level_colon(bytes, cond_start, bytes.len()) else {
3221                    index += keyword_len;
3222                    continue;
3223                };
3224                if parse_error_offset < index || parse_error_offset > colon {
3225                    index += keyword_len;
3226                    continue;
3227                }
3228                let Some(equal) = condition_plain_assignment(bytes, cond_start, colon) else {
3229                    index += keyword_len;
3230                    continue;
3231                };
3232                let (target_start, target_end) = trim_target_range(bytes, cond_start, equal);
3233                if target_start >= target_end {
3234                    index += keyword_len;
3235                    continue;
3236                }
3237                if is_simple_keyword_name(bytes, target_start, target_end) {
3238                    return Some(CpythonDiagnostic::new(
3239                        "invalid syntax. Maybe you meant '==' or ':=' instead of '='?".to_owned(),
3240                        target_start,
3241                        equal + 1,
3242                    ));
3243                }
3244                if let Some((expr_name, start, end, _)) =
3245                    expression_name_and_range(&source[target_start..target_end])
3246                {
3247                    return Some(CpythonDiagnostic::new(
3248                        format!(
3249                            "cannot assign to {expr_name} here. Maybe you meant '==' instead of '='?"
3250                        ),
3251                        target_start + start,
3252                        target_start + end,
3253                    ));
3254                }
3255                return Some(CpythonDiagnostic::new(
3256                    "invalid syntax. Maybe you meant '==' or ':=' instead of '='?".to_owned(),
3257                    target_start,
3258                    equal + 1,
3259                ));
3260            }
3261            _ => index += 1,
3262        }
3263    }
3264    None
3265}
3266
3267fn condition_plain_assignment(bytes: &[u8], start: usize, end: usize) -> Option<usize> {
3268    let mut index = start;
3269    let mut nest = Vec::new();
3270    while index < end {
3271        match bytes[index] {
3272            b'#' => {
3273                while index < end && bytes[index] != b'\n' {
3274                    index += 1;
3275                }
3276            }
3277            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3278            b'(' => {
3279                nest.push(if is_call_open(bytes, start, index) {
3280                    b'c'
3281                } else {
3282                    b'g'
3283                });
3284                index += 1;
3285            }
3286            b'[' => {
3287                nest.push(b'[');
3288                index += 1;
3289            }
3290            b'{' => {
3291                nest.push(b'{');
3292                index += 1;
3293            }
3294            b')' | b']' | b'}' => {
3295                nest.pop();
3296                index += 1;
3297            }
3298            b':' if nest.last() == Some(&b'l') => {
3299                nest.pop();
3300                index += 1;
3301            }
3302            b'=' if is_plain_assignment_operator(bytes, index)
3303                && !nest.contains(&b'c')
3304                && !nest.contains(&b'l') =>
3305            {
3306                return Some(index);
3307            }
3308            _ if starts_identifier(bytes, index, b"lambda") => {
3309                nest.push(b'l');
3310                index += 6;
3311            }
3312            _ => index += 1,
3313        }
3314    }
3315    None
3316}
3317
3318fn is_call_open(bytes: &[u8], start: usize, open: usize) -> bool {
3319    let mut index = open;
3320    while index > start {
3321        index -= 1;
3322        match bytes[index] {
3323            b' ' | b'\t' | b'\n' | b'\r' | b'\x0c' => {}
3324            b')' | b']' => return true,
3325            byte if byte >= 0x80 || is_ascii_identifier_char(byte) => {
3326                let mut ident_start = index;
3327                while ident_start > start
3328                    && bytes
3329                        .get(ident_start - 1)
3330                        .is_some_and(|b| *b >= 0x80 || is_ascii_identifier_char(*b))
3331                {
3332                    ident_start -= 1;
3333                }
3334                return !is_condition_keyword(&bytes[ident_start..=index]);
3335            }
3336            _ => return false,
3337        }
3338    }
3339    false
3340}
3341
3342fn is_condition_keyword(word: &[u8]) -> bool {
3343    matches!(
3344        word,
3345        b"not" | b"and" | b"or" | b"in" | b"is" | b"if" | b"elif" | b"while" | b"await" | b"lambda"
3346    )
3347}
3348
3349fn invalid_assignment_target_error(source: &str) -> Option<CpythonDiagnostic> {
3350    let bytes = source.as_bytes();
3351    let offsets = top_level_plain_assignment_offsets(bytes);
3352    if offsets.is_empty() {
3353        return None;
3354    }
3355    let mut start = 0usize;
3356    for offset in offsets {
3357        if let Some(error) = assignment_target_error_for_slice(source, start, offset) {
3358            return Some(error);
3359        }
3360        start = offset + 1;
3361    }
3362    None
3363}
3364
3365fn top_level_augassign_offset(bytes: &[u8]) -> Option<(usize, usize)> {
3366    let mut index = 0usize;
3367    let mut level = 0usize;
3368    while index < bytes.len() {
3369        match bytes[index] {
3370            b'#' if level == 0 => {
3371                while index < bytes.len() && bytes[index] != b'\n' {
3372                    index += 1;
3373                }
3374            }
3375            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3376            b'(' | b'[' | b'{' => {
3377                level += 1;
3378                index += 1;
3379            }
3380            b')' | b']' | b'}' => {
3381                level = level.saturating_sub(1);
3382                index += 1;
3383            }
3384            b'+' | b'-' | b'*' | b'@' | b'/' | b'%' | b'&' | b'|' | b'^'
3385                if level == 0 && bytes.get(index + 1) == Some(&b'=') =>
3386            {
3387                return Some((index, 2));
3388            }
3389            b'<' | b'>'
3390                if level == 0
3391                    && bytes.get(index + 1) == Some(&bytes[index])
3392                    && bytes.get(index + 2) == Some(&b'=') =>
3393            {
3394                return Some((index, 3));
3395            }
3396            b'*' if level == 0
3397                && bytes.get(index + 1) == Some(&b'*')
3398                && bytes.get(index + 2) == Some(&b'=') =>
3399            {
3400                return Some((index, 3));
3401            }
3402            b'/' if level == 0
3403                && bytes.get(index + 1) == Some(&b'/')
3404                && bytes.get(index + 2) == Some(&b'=') =>
3405            {
3406                return Some((index, 3));
3407            }
3408            _ => index += 1,
3409        }
3410    }
3411    None
3412}
3413
3414fn invalid_augassign_target_error(source: &str) -> Option<CpythonDiagnostic> {
3415    let bytes = source.as_bytes();
3416    let (operator, _) = top_level_augassign_offset(bytes)?;
3417    let (target_start, target_end) = assignment_target_expr_range(source, 0, operator)?;
3418    let target_text = &source[target_start..target_end];
3419    let Ok(parsed) = parser::parse(target_text, parser::Mode::Expression.into()) else {
3420        return None;
3421    };
3422    let ast::Mod::Expression(expression) = parsed.into_syntax() else {
3423        return None;
3424    };
3425    let name = delete_target_expr_name(&expression.body);
3426    Some(CpythonDiagnostic::new(
3427        format!("'{name}' is an illegal expression for augmented assignment"),
3428        target_start,
3429        target_end,
3430    ))
3431}
3432
3433fn find_for_target_delimiter(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
3434    let mut level = 0usize;
3435    while index < end {
3436        match bytes[index] {
3437            b'#' if level == 0 => return None,
3438            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3439            b'(' | b'[' | b'{' => {
3440                level += 1;
3441                index += 1;
3442            }
3443            b')' | b']' | b'}' => {
3444                level = level.saturating_sub(1);
3445                index += 1;
3446            }
3447            _ if level == 0 && starts_identifier(bytes, index, b"in") => return Some(index),
3448            b':' if level == 0 => return Some(index),
3449            _ => index += 1,
3450        }
3451    }
3452    None
3453}
3454
3455fn invalid_for_target_error(source: &str) -> Option<CpythonDiagnostic> {
3456    let bytes = source.as_bytes();
3457    let mut index = 0usize;
3458    while index < bytes.len() {
3459        match bytes[index] {
3460            b'#' => {
3461                while index < bytes.len() && bytes[index] != b'\n' {
3462                    index += 1;
3463                }
3464            }
3465            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3466            _ if starts_identifier(bytes, index, b"for") => {
3467                let target_start = skip_horizontal_whitespace(bytes, index + 3);
3468                let line_end = source[index..]
3469                    .find('\n')
3470                    .map_or(bytes.len(), |newline| index + newline);
3471                if let Some(target_end) = find_for_target_delimiter(bytes, target_start, line_end) {
3472                    if let Some(error) =
3473                        for_target_error_for_slice(source, target_start, target_end)
3474                    {
3475                        return Some(error);
3476                    }
3477                    if let Some(error) =
3478                        non_in_compare_for_target_error(source, target_start, target_end)
3479                    {
3480                        return Some(error);
3481                    }
3482                }
3483                index = target_start.max(index + 3);
3484            }
3485            _ => index += 1,
3486        }
3487    }
3488    None
3489}
3490
3491fn find_with_target_delimiter(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
3492    let mut level = 0usize;
3493    while index < end {
3494        match bytes[index] {
3495            b'#' if level == 0 => return None,
3496            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3497            b'(' | b'[' | b'{' => {
3498                level += 1;
3499                index += 1;
3500            }
3501            b')' if level == 0 => return Some(index),
3502            b')' | b']' | b'}' => {
3503                level = level.saturating_sub(1);
3504                index += 1;
3505            }
3506            b',' | b':' if level == 0 => return Some(index),
3507            _ => index += 1,
3508        }
3509    }
3510    None
3511}
3512
3513fn invalid_with_target_error(source: &str) -> Option<CpythonDiagnostic> {
3514    let bytes = source.as_bytes();
3515    let mut line_start = 0usize;
3516    for line in source.split_inclusive('\n') {
3517        let line_end = line_start + line.len();
3518        let mut column = skip_horizontal_whitespace(bytes, line_start);
3519        if starts_identifier(bytes, column, b"async") {
3520            column = skip_horizontal_whitespace(bytes, column + 5);
3521        }
3522        if !starts_identifier(bytes, column, b"with") {
3523            line_start = line_end;
3524            continue;
3525        }
3526        let mut index = column + 4;
3527        while let Some(as_index) = find_keyword_at_level(bytes, index, line_end, b"as") {
3528            let target_start = skip_horizontal_whitespace(bytes, as_index + 2);
3529            if let Some(target_end) = find_with_target_delimiter(bytes, target_start, line_end) {
3530                if let Some(error) = star_target_error_for_slice(source, target_start, target_end) {
3531                    return Some(error);
3532                }
3533                index = target_end.saturating_add(1);
3534            } else {
3535                break;
3536            }
3537        }
3538        line_start = line_end;
3539    }
3540    None
3541}
3542
3543fn find_missing_in_if_keyword(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
3544    let mut level = 0usize;
3545    while index < end {
3546        match bytes[index] {
3547            b'#' if level == 0 => return None,
3548            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3549            b'(' | b'[' | b'{' => {
3550                level += 1;
3551                index += 1;
3552            }
3553            b')' | b']' | b'}' if level == 0 => return None,
3554            b')' | b']' | b'}' => {
3555                level = level.saturating_sub(1);
3556                index += 1;
3557            }
3558            _ if level == 0 && starts_identifier(bytes, index, b"in") => return None,
3559            _ if level == 0 && starts_identifier(bytes, index, b"if") => return Some(index),
3560            _ => index += 1,
3561        }
3562    }
3563    None
3564}
3565
3566fn invalid_for_if_clause_error(source: &str) -> Option<CpythonDiagnostic> {
3567    let bytes = source.as_bytes();
3568    let mut index = 0usize;
3569    let mut level = 0usize;
3570    while index < bytes.len() {
3571        match bytes[index] {
3572            b'#' if level == 0 => {
3573                while index < bytes.len() && bytes[index] != b'\n' {
3574                    index += 1;
3575                }
3576            }
3577            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3578            b'(' | b'[' | b'{' => {
3579                level += 1;
3580                index += 1;
3581            }
3582            b')' | b']' | b'}' => {
3583                level = level.saturating_sub(1);
3584                index += 1;
3585            }
3586            _ if level > 0 && starts_identifier(bytes, index, b"for") => {
3587                let target_start = skip_horizontal_whitespace(bytes, index + 3);
3588                let line_end = source[index..]
3589                    .find('\n')
3590                    .map_or(bytes.len(), |newline| index + newline);
3591                if let Some(if_index) = find_missing_in_if_keyword(bytes, target_start, line_end) {
3592                    return Some(CpythonDiagnostic::new(
3593                        "'in' expected after for-loop variables".to_owned(),
3594                        if_index,
3595                        (if_index + 2).min(line_end),
3596                    ));
3597                }
3598                index = target_start.max(index + 3);
3599            }
3600            _ => index += 1,
3601        }
3602    }
3603    None
3604}
3605
3606fn invalid_delete_target_error(source: &str) -> Option<CpythonDiagnostic> {
3607    let bytes = source.as_bytes();
3608    let mut index = 0;
3609    while index < bytes.len() {
3610        match bytes[index] {
3611            b'#' => {
3612                while index < bytes.len() && bytes[index] != b'\n' {
3613                    index += 1;
3614                }
3615            }
3616            b'\'' | b'"' => {
3617                index = skip_quoted_string(bytes, index);
3618            }
3619            b'd' if starts_identifier(bytes, index, b"del") => {
3620                let mut target_start = index + 3;
3621                if !matches!(bytes.get(target_start), Some(b' ' | b'\t' | b'\x0c')) {
3622                    index += 3;
3623                    continue;
3624                }
3625                while matches!(bytes.get(target_start), Some(b' ' | b'\t' | b'\x0c')) {
3626                    target_start += 1;
3627                }
3628                let mut target_end = statement_target_end(bytes, target_start);
3629                while target_end > target_start
3630                    && matches!(bytes.get(target_end - 1), Some(b' ' | b'\t' | b'\x0c'))
3631                {
3632                    target_end -= 1;
3633                }
3634                if target_start >= target_end {
3635                    index = target_end.max(index + 3);
3636                    continue;
3637                }
3638                if parenthesized_single_starred_delete_target(bytes, target_start, target_end) {
3639                    return Some(CpythonDiagnostic::new(
3640                        "cannot use starred expression here".to_owned(),
3641                        target_start,
3642                        target_end,
3643                    ));
3644                }
3645                if bytes.get(target_start) == Some(&b'*') {
3646                    return Some(CpythonDiagnostic::new(
3647                        "cannot delete starred".to_owned(),
3648                        target_start,
3649                        (target_start + 1).min(target_end),
3650                    ));
3651                }
3652                let target_text = &source[target_start..target_end];
3653                let Ok(parsed) = parser::parse(target_text, parser::Mode::Expression.into()) else {
3654                    index = target_end;
3655                    continue;
3656                };
3657                let ast::Mod::Expression(expression) = parsed.into_syntax() else {
3658                    index = target_end;
3659                    continue;
3660                };
3661                let Some(invalid_target) = invalid_delete_target(&expression.body) else {
3662                    index = target_end;
3663                    continue;
3664                };
3665                let start = target_start + invalid_target.range().start().to_usize();
3666                let end = target_start + invalid_target.range().end().to_usize();
3667                if matches!(invalid_target, ast::Expr::FString(_)) {
3668                    return Some(CpythonDiagnostic::new(
3669                        "invalid syntax".to_owned(),
3670                        start,
3671                        end,
3672                    ));
3673                }
3674                let name = delete_target_expr_name(invalid_target);
3675                return Some(CpythonDiagnostic::new(
3676                    format!("cannot delete {name}"),
3677                    start,
3678                    end,
3679                ));
3680            }
3681            _ => index += 1,
3682        }
3683    }
3684    None
3685}
3686
3687fn skip_horizontal_whitespace(bytes: &[u8], mut index: usize) -> usize {
3688    while matches!(bytes.get(index), Some(b' ' | b'\t' | b'\x0c')) {
3689        index += 1;
3690    }
3691    index
3692}
3693
3694fn find_keyword_at_level(
3695    bytes: &[u8],
3696    mut index: usize,
3697    end: usize,
3698    keyword: &[u8],
3699) -> Option<usize> {
3700    let mut level = 0usize;
3701    while index < end {
3702        match bytes[index] {
3703            b'#' if level == 0 => return None,
3704            b'\'' | b'"' => {
3705                index = skip_quoted_string(bytes, index);
3706            }
3707            b'(' | b'[' | b'{' => {
3708                level += 1;
3709                index += 1;
3710            }
3711            b')' | b']' | b'}' => {
3712                level = level.saturating_sub(1);
3713                index += 1;
3714            }
3715            _ if level == 0 && starts_identifier(bytes, index, keyword) => return Some(index),
3716            _ => index += 1,
3717        }
3718    }
3719    None
3720}
3721
3722fn find_byte_at_level(bytes: &[u8], mut index: usize, end: usize, needle: u8) -> Option<usize> {
3723    let mut level = 0usize;
3724    while index < end {
3725        match bytes[index] {
3726            b'#' if level == 0 => return None,
3727            b'\'' | b'"' => {
3728                index = skip_quoted_string(bytes, index);
3729            }
3730            b'(' | b'[' | b'{' => {
3731                level += 1;
3732                index += 1;
3733            }
3734            b')' | b']' | b'}' => {
3735                level = level.saturating_sub(1);
3736                index += 1;
3737            }
3738            byte if level == 0 && byte == needle => return Some(index),
3739            _ => index += 1,
3740        }
3741    }
3742    None
3743}
3744
3745fn expression_name_and_range(source: &str) -> Option<(&'static str, usize, usize, bool)> {
3746    let parsed = parser::parse(source, parser::Mode::Expression.into()).ok()?;
3747    let ast::Mod::Expression(expression) = parsed.into_syntax() else {
3748        return None;
3749    };
3750    let is_name = matches!(expression.body.as_ref(), ast::Expr::Name(_));
3751    Some((
3752        delete_target_expr_name(&expression.body),
3753        expression.body.range().start().to_usize(),
3754        expression.body.range().end().to_usize(),
3755        is_name,
3756    ))
3757}
3758
3759fn invalid_standalone_except_error(source: &str) -> Option<CpythonDiagnostic> {
3760    let bytes = source.as_bytes();
3761    let mut line_start = 0usize;
3762    let mut seen_try = false;
3763    for line in source.split_inclusive('\n') {
3764        let line_end = line_start + line.len();
3765        let column = skip_horizontal_whitespace(bytes, line_start);
3766        if column >= line_end {
3767            line_start = line_end;
3768            continue;
3769        }
3770        if starts_identifier(bytes, column, b"try") {
3771            seen_try = true;
3772        } else if (bytes.get(column..column + 7) == Some(b"except*")
3773            || starts_identifier(bytes, column, b"except"))
3774            && !seen_try
3775        {
3776            let end = if bytes.get(column..column + 7) == Some(b"except*") {
3777                column + 7
3778            } else {
3779                column + 6
3780            };
3781            return Some(CpythonDiagnostic::new(
3782                "invalid syntax".to_owned(),
3783                column,
3784                end,
3785            ));
3786        }
3787        line_start = line_end;
3788    }
3789    None
3790}
3791
3792fn invalid_import_statement_error(source: &str) -> Option<CpythonDiagnostic> {
3793    let bytes = source.as_bytes();
3794    let mut line_start = 0usize;
3795    for line in source.split_inclusive('\n') {
3796        let line_end = line_start + line.len();
3797        let column = skip_horizontal_whitespace(bytes, line_start);
3798        if column < line_end
3799            && starts_identifier(bytes, column, b"import")
3800            && find_keyword_at_level(bytes, column + 6, line_end, b"from").is_some()
3801        {
3802            return Some(CpythonDiagnostic::new(
3803                "Did you mean to use 'from ... import ...' instead?".to_owned(),
3804                column,
3805                column + 6,
3806            ));
3807        }
3808        line_start = line_end;
3809    }
3810    None
3811}
3812
3813fn import_as_target_end(bytes: &[u8], mut index: usize) -> usize {
3814    let mut level = 0usize;
3815    while index < bytes.len() {
3816        match bytes[index] {
3817            b'#' if level == 0 => return index,
3818            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3819            b'(' | b'[' | b'{' => {
3820                level += 1;
3821                index += 1;
3822            }
3823            b')' if level == 0 => return index,
3824            b')' | b']' | b'}' => {
3825                level = level.saturating_sub(1);
3826                index += 1;
3827            }
3828            b',' | b';' | b'\n' if level == 0 => return index,
3829            _ => index += 1,
3830        }
3831    }
3832    index
3833}
3834
3835fn valid_import_alias_name(bytes: &[u8], mut start: usize, end: usize) -> bool {
3836    start = skip_horizontal_whitespace(bytes, start);
3837    let Some(&first) = bytes.get(start) else {
3838        return false;
3839    };
3840    if !(first == b'_' || first.is_ascii_alphabetic() || first >= 0x80) {
3841        return false;
3842    }
3843    let mut index = start + 1;
3844    while index < end {
3845        match bytes[index] {
3846            b' ' | b'\t' | b'\x0c' => break,
3847            byte if byte >= 0x80 || is_ascii_identifier_char(byte) => index += 1,
3848            _ => return false,
3849        }
3850    }
3851    let index = skip_horizontal_whitespace(bytes, index);
3852    matches!(
3853        bytes.get(index),
3854        None | Some(b',' | b')' | b';' | b'\n' | b'\r')
3855    )
3856}
3857
3858fn import_target_error_for_slice(
3859    source: &str,
3860    start: usize,
3861    end: usize,
3862) -> Option<CpythonDiagnostic> {
3863    let bytes = source.as_bytes();
3864    let (target_start, target_end) = trim_target_range(bytes, start, end);
3865    if target_start >= target_end || valid_import_alias_name(bytes, target_start, target_end) {
3866        return None;
3867    }
3868    let parsed = parser::parse(
3869        &source[target_start..target_end],
3870        parser::Mode::Expression.into(),
3871    )
3872    .ok()?;
3873    let ast::Mod::Expression(expression) = parsed.into_syntax() else {
3874        return None;
3875    };
3876    let name = delete_target_expr_name(&expression.body);
3877    let start = target_start + expression.body.range().start().to_usize();
3878    let end = target_start + expression.body.range().end().to_usize();
3879    Some(CpythonDiagnostic::new(
3880        format!("cannot use {name} as import target"),
3881        start,
3882        end,
3883    ))
3884}
3885
3886fn statement_starts_import(bytes: &[u8], line_start: usize, line_end: usize) -> bool {
3887    let column = skip_horizontal_whitespace(bytes, line_start);
3888    if starts_identifier(bytes, column, b"import") {
3889        return true;
3890    }
3891    starts_identifier(bytes, column, b"from")
3892        && find_keyword_at_level(bytes, column + 4, line_end, b"import").is_some()
3893}
3894
3895fn invalid_import_target_error(source: &str) -> Option<CpythonDiagnostic> {
3896    let bytes = source.as_bytes();
3897    let mut line_start = 0usize;
3898    let mut in_parenthesized_from_import = false;
3899    for line in source.split_inclusive('\n') {
3900        let line_end = line_start + line.len();
3901        let starts_import = statement_starts_import(bytes, line_start, line_end);
3902        if starts_import && bytes[line_start..line_end].contains(&b'(') {
3903            in_parenthesized_from_import = true;
3904        }
3905        if starts_import || in_parenthesized_from_import {
3906            let mut index = line_start;
3907            while index < line_end {
3908                if starts_identifier(bytes, index, b"as") {
3909                    let target_start = skip_horizontal_whitespace(bytes, index + 2);
3910                    let target_end = import_as_target_end(bytes, target_start);
3911                    if let Some(error) =
3912                        import_target_error_for_slice(source, target_start, target_end)
3913                    {
3914                        return Some(error);
3915                    }
3916                    index = target_end.max(index + 2);
3917                } else {
3918                    index += 1;
3919                }
3920            }
3921        }
3922        if in_parenthesized_from_import && bytes[line_start..line_end].contains(&b')') {
3923            in_parenthesized_from_import = false;
3924        }
3925        line_start = line_end;
3926    }
3927    None
3928}
3929
3930fn invalid_except_as_target_error(source: &str) -> Option<CpythonDiagnostic> {
3931    let bytes = source.as_bytes();
3932    let mut line_start = 0usize;
3933    let mut seen_try = false;
3934    for line in source.split_inclusive('\n') {
3935        let line_end = line_start + line.len();
3936        let mut column = skip_horizontal_whitespace(bytes, line_start);
3937        if column >= line_end {
3938            line_start = line_end;
3939            continue;
3940        }
3941        if starts_identifier(bytes, column, b"try") {
3942            seen_try = true;
3943            line_start = line_end;
3944            continue;
3945        }
3946        let (keyword_len, starred) = if bytes.get(column..column + 7) == Some(b"except*") {
3947            (7, true)
3948        } else if starts_identifier(bytes, column, b"except") {
3949            (6, false)
3950        } else {
3951            line_start = line_end;
3952            continue;
3953        };
3954        if !seen_try {
3955            line_start = line_end;
3956            continue;
3957        }
3958        column += keyword_len;
3959        let Some(as_index) = find_keyword_at_level(bytes, column, line_end, b"as") else {
3960            line_start = line_end;
3961            continue;
3962        };
3963        let target_start = skip_horizontal_whitespace(bytes, as_index + 2);
3964        let Some(delimiter) = find_byte_at_level(bytes, target_start, line_end, b':')
3965            .into_iter()
3966            .chain(find_byte_at_level(bytes, target_start, line_end, b','))
3967            .min()
3968        else {
3969            line_start = line_end;
3970            continue;
3971        };
3972        let mut target_end = delimiter;
3973        while target_end > target_start
3974            && matches!(bytes.get(target_end - 1), Some(b' ' | b'\t' | b'\x0c'))
3975        {
3976            target_end -= 1;
3977        }
3978        let Some((expr_name, start, end, is_name)) =
3979            expression_name_and_range(&source[target_start..target_end])
3980        else {
3981            line_start = line_end;
3982            continue;
3983        };
3984        if !is_name {
3985            let statement = if starred { "except*" } else { "except" };
3986            return Some(CpythonDiagnostic::new(
3987                format!("cannot use {statement} statement with {expr_name}"),
3988                target_start + start,
3989                target_start + end,
3990            ));
3991        }
3992        line_start = line_end;
3993    }
3994    None
3995}
3996
3997fn invalid_match_as_target_error(source: &str) -> Option<CpythonDiagnostic> {
3998    let bytes = source.as_bytes();
3999    let quoted_ranges = quoted_string_ranges(bytes);
4000    let mut quoted_range = 0usize;
4001    let mut line_start = 0usize;
4002    for line in source.split_inclusive('\n') {
4003        let line_end = line_start + line.len();
4004        let mut column = skip_horizontal_whitespace(bytes, line_start);
4005        if column >= line_end
4006            || offset_in_ranges(&quoted_ranges, &mut quoted_range, column)
4007            || !starts_identifier(bytes, column, b"case")
4008        {
4009            line_start = line_end;
4010            continue;
4011        }
4012        column += 4;
4013        let Some(as_index) = find_keyword_at_level(bytes, column, line_end, b"as") else {
4014            line_start = line_end;
4015            continue;
4016        };
4017        let target_start = skip_horizontal_whitespace(bytes, as_index + 2);
4018        let Some(delimiter) = find_byte_at_level(bytes, target_start, line_end, b':')
4019            .into_iter()
4020            .chain(find_byte_at_level(bytes, target_start, line_end, b','))
4021            .min()
4022        else {
4023            line_start = line_end;
4024            continue;
4025        };
4026        let mut target_end = delimiter;
4027        while target_end > target_start
4028            && matches!(bytes.get(target_end - 1), Some(b' ' | b'\t' | b'\x0c'))
4029        {
4030            target_end -= 1;
4031        }
4032        if source[target_start..target_end].trim() == "_" {
4033            return Some(CpythonDiagnostic::new(
4034                "cannot use '_' as a target".to_owned(),
4035                target_start,
4036                target_end,
4037            ));
4038        }
4039        let Some((expr_name, start, end, is_name)) =
4040            expression_name_and_range(&source[target_start..target_end])
4041        else {
4042            line_start = line_end;
4043            continue;
4044        };
4045        if !is_name {
4046            if matches!(expr_name, "expression" | "subscript") {
4047                line_start = line_end;
4048                continue;
4049            }
4050            return Some(CpythonDiagnostic::new(
4051                format!("cannot use {expr_name} as pattern target"),
4052                target_start + start,
4053                target_start + end,
4054            ));
4055        }
4056        line_start = line_end;
4057    }
4058    None
4059}
4060
4061fn quoted_string_ranges(bytes: &[u8]) -> Vec<(usize, usize)> {
4062    let mut ranges = Vec::new();
4063    let mut index = 0usize;
4064    while index < bytes.len() {
4065        match bytes[index] {
4066            b'#' => {
4067                while index < bytes.len() && bytes[index] != b'\n' {
4068                    index += 1;
4069                }
4070            }
4071            b'\'' | b'"' => {
4072                let end = skip_quoted_string(bytes, index);
4073                ranges.push((index, end));
4074                index = end;
4075            }
4076            _ => index += 1,
4077        }
4078    }
4079    ranges
4080}
4081
4082fn offset_in_ranges(ranges: &[(usize, usize)], range_index: &mut usize, offset: usize) -> bool {
4083    while ranges
4084        .get(*range_index)
4085        .is_some_and(|(_, end)| *end <= offset)
4086    {
4087        *range_index += 1;
4088    }
4089    ranges
4090        .get(*range_index)
4091        .is_some_and(|(start, end)| *start <= offset && offset < *end)
4092}
4093
4094fn invalid_match_mapping_rest_wildcard_error(source: &str) -> Option<CpythonDiagnostic> {
4095    let bytes = source.as_bytes();
4096    let next_line_end = |line_start: usize| {
4097        line_start
4098            + bytes[line_start..]
4099                .iter()
4100                .position(|byte| *byte == b'\n')
4101                .unwrap_or(bytes.len() - line_start)
4102    };
4103    let mut index = 0usize;
4104    let mut line_start = 0usize;
4105    let mut line_end = next_line_end(line_start);
4106    while index < bytes.len() {
4107        match bytes[index] {
4108            b'#' => {
4109                while index < bytes.len() && bytes[index] != b'\n' {
4110                    index += 1;
4111                }
4112            }
4113            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
4114            b'\n' => {
4115                index += 1;
4116                line_start = index;
4117                line_end = next_line_end(line_start);
4118            }
4119            _ => {
4120                let column = skip_horizontal_whitespace(bytes, line_start);
4121                if index != column
4122                    || column >= line_end
4123                    || !starts_identifier(bytes, column, b"case")
4124                {
4125                    index += 1;
4126                    continue;
4127                }
4128                let mut cursor = column + 4;
4129                while cursor < line_end {
4130                    match bytes[cursor] {
4131                        b'#' => break,
4132                        b'\'' | b'"' => cursor = skip_quoted_string(bytes, cursor),
4133                        b'{' => {
4134                            let rest = next_non_horizontal_whitespace(bytes, cursor + 1);
4135                            if bytes.get(rest..rest + 2) == Some(b"**") {
4136                                let name_start = next_non_horizontal_whitespace(bytes, rest + 2);
4137                                let name_end = identifier_end(bytes, name_start, line_end);
4138                                if source.get(name_start..name_end) == Some("_") {
4139                                    return Some(CpythonDiagnostic::new(
4140                                        "invalid syntax".to_owned(),
4141                                        name_start,
4142                                        name_end,
4143                                    ));
4144                                }
4145                            }
4146                            cursor += 1;
4147                        }
4148                        _ => cursor += 1,
4149                    }
4150                }
4151                index = line_end;
4152            }
4153        }
4154    }
4155    None
4156}
4157
4158fn invalid_if_expression_statement_error(source: &str) -> Option<CpythonDiagnostic> {
4159    let bytes = source.as_bytes();
4160    let mut line_start = 0usize;
4161    for line in source.split_inclusive('\n') {
4162        let line_end = line_start + line.len();
4163        if let Some(if_index) = find_keyword_at_level(bytes, line_start, line_end, b"if")
4164            && let Some((start, end)) = statement_before_if_expression(bytes, line_start, if_index)
4165            && find_keyword_at_level(bytes, if_index + 2, line_end, b"else").is_some()
4166        {
4167            return Some(CpythonDiagnostic::new(
4168                "expected expression before 'if', but statement is given".to_owned(),
4169                start,
4170                end,
4171            ));
4172        }
4173        if let Some(else_index) = find_keyword_at_level(bytes, line_start, line_end, b"else")
4174            && find_keyword_at_level(bytes, line_start, else_index, b"if").is_some()
4175            && let Some((start, end)) =
4176                statement_after_else_expression(bytes, else_index + 4, line_end)
4177        {
4178            return Some(CpythonDiagnostic::new(
4179                "expected expression after 'else', but statement is given".to_owned(),
4180                start,
4181                end,
4182            ));
4183        }
4184        line_start = line_end;
4185    }
4186    None
4187}
4188
4189fn statement_before_if_expression(
4190    bytes: &[u8],
4191    line_start: usize,
4192    if_index: usize,
4193) -> Option<(usize, usize)> {
4194    let mut start = if_index;
4195    while start > line_start && matches!(bytes.get(start - 1), Some(b' ' | b'\t' | b'\x0c')) {
4196        start -= 1;
4197    }
4198    while start > line_start
4199        && !matches!(
4200            bytes.get(start - 1),
4201            Some(b'=' | b':' | b',' | b'(' | b'[' | b'{')
4202        )
4203    {
4204        start -= 1;
4205    }
4206    start = skip_horizontal_whitespace(bytes, start);
4207    for keyword in [b"pass".as_slice(), b"break", b"continue"] {
4208        if starts_identifier(bytes, start, keyword) {
4209            return Some((start, start + keyword.len()));
4210        }
4211    }
4212    None
4213}
4214
4215fn statement_after_else_expression(
4216    bytes: &[u8],
4217    else_end: usize,
4218    line_end: usize,
4219) -> Option<(usize, usize)> {
4220    let start = skip_horizontal_whitespace(bytes, else_end);
4221    for keyword in [
4222        b"pass".as_slice(),
4223        b"return",
4224        b"raise",
4225        b"del",
4226        b"yield",
4227        b"assert",
4228        b"break",
4229        b"continue",
4230        b"import",
4231        b"from",
4232    ] {
4233        if starts_identifier(bytes, start, keyword) {
4234            let end = statement_target_end(bytes, start).min(line_end);
4235            return Some((start, end.max(start + keyword.len())));
4236        }
4237    }
4238    None
4239}
4240
4241fn invalid_else_elif_error(source: &str) -> Option<CpythonDiagnostic> {
4242    let bytes = source.as_bytes();
4243    let mut line_start = 0usize;
4244    let mut else_indents: Vec<usize> = Vec::new();
4245    for line in source.split_inclusive('\n') {
4246        let line_end = line_start + line.len();
4247        let column = skip_horizontal_whitespace(bytes, line_start);
4248        let line_column = column.saturating_sub(line_start);
4249        if column >= line_end {
4250            line_start = line_end;
4251            continue;
4252        }
4253        while else_indents
4254            .last()
4255            .is_some_and(|indent| line_column < *indent)
4256        {
4257            else_indents.pop();
4258        }
4259        if starts_identifier(bytes, column, b"else")
4260            && find_byte_at_level(bytes, column + 4, line_end, b':').is_some()
4261        {
4262            else_indents.push(line_column);
4263        } else if starts_identifier(bytes, column, b"elif") && else_indents.contains(&line_column) {
4264            return Some(CpythonDiagnostic::new(
4265                "'elif' block follows an 'else' block".to_owned(),
4266                column,
4267                column + 4,
4268            ));
4269        }
4270        line_start = line_end;
4271    }
4272    None
4273}
4274
4275fn mixed_except_handlers_error(source: &str) -> Option<CpythonDiagnostic> {
4276    let message = "cannot have both 'except' and 'except*' on the same 'try'".to_owned();
4277    let mut seen_except = false;
4278    let mut seen_except_star = false;
4279    let mut line_start = 0usize;
4280    for line in source.split_inclusive('\n') {
4281        let bytes = line.as_bytes();
4282        let mut column = 0usize;
4283        while matches!(bytes.get(column), Some(b' ' | b'\t' | b'\x0c')) {
4284            column += 1;
4285        }
4286        let token_start = line_start + column;
4287        if bytes.get(column..column + 7) == Some(b"except*") {
4288            if seen_except {
4289                return Some(CpythonDiagnostic::new(
4290                    message,
4291                    token_start,
4292                    token_start + 7,
4293                ));
4294            }
4295            seen_except_star = true;
4296        } else if starts_identifier(bytes, column, b"except") {
4297            if seen_except_star {
4298                return Some(CpythonDiagnostic::new(
4299                    message,
4300                    token_start,
4301                    token_start + 6,
4302                ));
4303            }
4304            seen_except = true;
4305        }
4306        line_start += line.len();
4307    }
4308    None
4309}
4310
4311fn non_printable_character_error(source: &str) -> Option<CpythonDiagnostic> {
4312    let bytes = source.as_bytes();
4313    let mut index = 0;
4314    while index < bytes.len() {
4315        match bytes[index] {
4316            b'#' => {
4317                while index < bytes.len() && bytes[index] != b'\n' {
4318                    index += 1;
4319                }
4320            }
4321            b'\'' | b'"' => {
4322                index = skip_quoted_string(bytes, index);
4323            }
4324            byte if byte.is_ascii_control() && !matches!(byte, b'\t' | b'\n' | b'\r' | b'\x0c') => {
4325                return Some(CpythonDiagnostic::new(
4326                    format!("invalid non-printable character U+{byte:04X}"),
4327                    index,
4328                    index + 1,
4329                ));
4330            }
4331            byte if byte >= 0x80 => {
4332                let ch = source[index..].chars().next()?;
4333                if ch.is_control() {
4334                    return Some(CpythonDiagnostic::new(
4335                        format!("invalid non-printable character U+{:04X}", ch as u32),
4336                        index,
4337                        index + ch.len_utf8(),
4338                    ));
4339                }
4340                index += ch.len_utf8();
4341            }
4342            _ => index += 1,
4343        }
4344    }
4345    None
4346}
4347
4348fn unterminated_string_error(source: &str, mode: Mode) -> Option<CpythonDiagnostic> {
4349    let bytes = source.as_bytes();
4350    let mut index = 0;
4351    let mut line = 1usize;
4352    while index < bytes.len() {
4353        match bytes[index] {
4354            b'#' => {
4355                while index < bytes.len() && bytes[index] != b'\n' {
4356                    index += 1;
4357                }
4358            }
4359            b'\n' => {
4360                line += 1;
4361                index += 1;
4362            }
4363            quote @ (b'\'' | b'"') => {
4364                let start = index;
4365                let start_line = line;
4366                let quote_size = if bytes.get(index + 1) == Some(&quote)
4367                    && bytes.get(index + 2) == Some(&quote)
4368                {
4369                    3
4370                } else {
4371                    1
4372                };
4373                index += quote_size;
4374                let mut has_escaped_quote = false;
4375                let mut ended_with_escape = false;
4376                let mut closed = false;
4377                while index < bytes.len() {
4378                    let c = bytes[index];
4379                    if c == b'\n' {
4380                        if quote_size == 1 {
4381                            // A single-quoted literal cannot span a line, so this is the same
4382                            // "the literal never ended" case as running out of source; CPython
4383                            // spells the two as one condition in lexer.c too.
4384                            break;
4385                        }
4386                        line += 1;
4387                        index += 1;
4388                    } else if c == quote {
4389                        if quote_size == 3 {
4390                            if bytes.get(index + 1) == Some(&quote)
4391                                && bytes.get(index + 2) == Some(&quote)
4392                            {
4393                                index += 3;
4394                                closed = true;
4395                                break;
4396                            }
4397                            index += 1;
4398                        } else {
4399                            index += 1;
4400                            closed = true;
4401                            break;
4402                        }
4403                    } else if c == b'\\' {
4404                        if bytes.get(index + 1) == Some(&quote) {
4405                            has_escaped_quote = true;
4406                        }
4407                        ended_with_escape = index + 1 >= bytes.len();
4408                        index = (index + 2).min(bytes.len());
4409                    } else {
4410                        index += 1;
4411                    }
4412                }
4413                if !closed {
4414                    if let Some(error) =
4415                        unclosed_replacement_field_error(bytes, start, start + quote_size, index)
4416                    {
4417                        return Some(error);
4418                    }
4419                    let detected_line = if quote_size == 3 { line } else { start_line };
4420                    let interpolated = interpolated_string_prefix(bytes, start);
4421                    let diagnostic = CpythonDiagnostic::new(
4422                        unterminated_string_message(
4423                            detected_line,
4424                            quote_size == 3,
4425                            has_escaped_quote,
4426                            interpolated,
4427                        ),
4428                        start,
4429                        start,
4430                    );
4431                    // E_EOLS is only for a plain single-quoted string at EOF.
4432                    // Single-quoted f/t-strings never set it. exec input appends a
4433                    // newline, so a single-quoted literal becomes the newline case
4434                    // unless a final `\` consumes that newline.
4435                    let exec_implicit_newline =
4436                        matches!(mode, Mode::Exec) && quote_size == 1 && !ended_with_escape;
4437                    let eval_assignment =
4438                        matches!(mode, Mode::Eval) && eval_has_assignment_before(bytes, start);
4439                    let continuable = index >= bytes.len()
4440                        && !(interpolated.is_some() && quote_size == 1)
4441                        && !exec_implicit_newline
4442                        && !eval_assignment;
4443                    return Some(if continuable {
4444                        diagnostic.with_unclosed_string()
4445                    } else {
4446                        diagnostic
4447                    });
4448                }
4449            }
4450            _ => index += 1,
4451        }
4452    }
4453    None
4454}
4455
4456fn eval_has_assignment_before(bytes: &[u8], end: usize) -> bool {
4457    let mut index = 0;
4458    let mut level = 0usize;
4459    let mut in_lambda_params = false;
4460    while index < end {
4461        match bytes[index] {
4462            b'#' => {
4463                while index < end && bytes[index] != b'\n' {
4464                    index += 1;
4465                }
4466            }
4467            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
4468            b'(' | b'[' | b'{' => {
4469                level += 1;
4470                index += 1;
4471            }
4472            b')' | b']' | b'}' => {
4473                level = level.saturating_sub(1);
4474                index += 1;
4475            }
4476            b':' if level == 0 => {
4477                in_lambda_params = false;
4478                index += 1;
4479            }
4480            b'=' if level == 0
4481                && !in_lambda_params
4482                && bytes.get(index + 1) != Some(&b'=')
4483                && !matches!(
4484                    bytes.get(index.saturating_sub(1)),
4485                    Some(b':' | b'!' | b'<' | b'>')
4486                ) =>
4487            {
4488                return true;
4489            }
4490            _ if level == 0 && starts_identifier(bytes, index, b"lambda") => {
4491                in_lambda_params = true;
4492                index += b"lambda".len();
4493            }
4494            _ => index += 1,
4495        }
4496    }
4497    false
4498}
4499
4500fn invalid_interpolated_string_error(source: &str) -> Option<CpythonDiagnostic> {
4501    let bytes = source.as_bytes();
4502    let mut index = 0;
4503    while index < bytes.len() {
4504        match bytes[index] {
4505            b'#' => {
4506                while index < bytes.len() && bytes[index] != b'\n' {
4507                    index += 1;
4508                }
4509            }
4510            quote @ (b'\'' | b'"') => {
4511                let Some(prefix) = interpolated_string_prefix(bytes, index) else {
4512                    index = skip_quoted_string(bytes, index);
4513                    continue;
4514                };
4515                if let Some(error) =
4516                    single_quoted_format_spec_newline_error(bytes, index, quote, prefix)
4517                {
4518                    return Some(error);
4519                }
4520                let Some((content_start, content_end)) =
4521                    quoted_string_content_range(bytes, index, quote)
4522                else {
4523                    index = skip_quoted_string(bytes, index);
4524                    continue;
4525                };
4526                if let Some(error) =
4527                    invalid_replacement_field_error(bytes, content_start, content_end, prefix)
4528                {
4529                    return Some(error);
4530                }
4531                index = skip_quoted_string(bytes, index);
4532            }
4533            _ => index += 1,
4534        }
4535    }
4536    None
4537}
4538
4539/// Whether the literal opened at `quote` carries a `b` in its prefix.
4540fn is_bytes_literal(bytes: &[u8], quote: usize) -> bool {
4541    let mut start = quote;
4542    while quote - start < 2
4543        && start > 0
4544        && matches!(
4545            bytes[start - 1].to_ascii_lowercase(),
4546            b'r' | b'b' | b'u' | b'f' | b't'
4547        )
4548    {
4549        start -= 1;
4550    }
4551    if start > 0 && is_ascii_identifier_char(bytes[start - 1]) {
4552        return false;
4553    }
4554    bytes[start..quote]
4555        .iter()
4556        .any(|byte| byte.eq_ignore_ascii_case(&b'b'))
4557}
4558
4559/// Which of CPython's two messages a mixed literal concatenation gets. Ruff reports only the
4560/// bytes/non-bytes mix, so `t"x" b"y"` arrives here as a bytes error where CPython names the
4561/// t-string, and this scanner owns the choice between them.
4562///
4563/// CPython's `invalid_string_tstring_concat` is an `invalid_` rule, reached only on the error
4564/// pass, and the `strings` rule tries `(fstring|string)+` first. That alternative consumes the
4565/// concatenation's leading run of non-t-string literals, so a bytes/non-bytes mix among *those*
4566/// raises from `_PyPegen_concatenate_strings` on the first pass and the t-string rule never runs.
4567/// The bytes message therefore keeps precedence for `"a" b"b" t"c"` but not for `t"x" b"y"`.
4568fn mixed_tstring_literal_error(
4569    error: &parser::ParseError,
4570    source: &str,
4571) -> Option<CpythonDiagnostic> {
4572    let parser::ParseErrorType::OtherError(message) = &error.error else {
4573        return None;
4574    };
4575    if !message.eq_ignore_ascii_case("bytes literal cannot be mixed with non-bytes literals") {
4576        return None;
4577    }
4578
4579    let start = error.location.start().to_usize();
4580    let end = error.location.end().to_usize();
4581    let bytes = source.as_bytes();
4582    if end > bytes.len() {
4583        return None;
4584    }
4585
4586    let mut index = start;
4587    let (mut bytes_seen, mut nonbytes_seen) = (false, false);
4588    while index < end {
4589        match bytes[index] {
4590            b'\'' | b'"' => {
4591                if interpolated_string_prefix(bytes, index) == Some("t-string") {
4592                    let message = if bytes_seen && nonbytes_seen {
4593                        // Reported here rather than left to ruff's error so that the scanners
4594                        // further down the chain cannot claim the concatenation instead.
4595                        "cannot mix bytes and nonbytes literals"
4596                    } else {
4597                        "cannot mix t-string literals with string or bytes literals"
4598                    };
4599                    return Some(CpythonDiagnostic::new(message.to_owned(), start, end));
4600                }
4601                if is_bytes_literal(bytes, index) {
4602                    bytes_seen = true;
4603                } else {
4604                    nonbytes_seen = true;
4605                }
4606                index = skip_quoted_string(bytes, index);
4607            }
4608            _ => index += 1,
4609        }
4610    }
4611    None
4612}
4613
4614fn single_quoted_format_spec_newline_error(
4615    bytes: &[u8],
4616    quote_index: usize,
4617    quote: u8,
4618    prefix: &str,
4619) -> Option<CpythonDiagnostic> {
4620    if bytes.get(quote_index + 1) == Some(&quote) && bytes.get(quote_index + 2) == Some(&quote) {
4621        return None;
4622    }
4623
4624    let (content_start, content_end) = quoted_string_content_range(bytes, quote_index, quote)?;
4625    let mut index = content_start;
4626    while index < content_end {
4627        match bytes[index] {
4628            b'{' if bytes.get(index + 1) == Some(&b'{') => index += 2,
4629            b'}' if bytes.get(index + 1) == Some(&b'}') => index += 2,
4630            b'{' => {
4631                let expr_start = skip_ascii_whitespace(bytes, index + 1, content_end);
4632                if let Some(separator) = replacement_field_separator(bytes, expr_start, content_end)
4633                    && bytes[separator] == b':'
4634                {
4635                    let format_end =
4636                        replacement_field_closing_brace(bytes, separator + 1, content_end)
4637                            .unwrap_or(content_end);
4638                    if bytes[separator + 1..format_end].contains(&b'\n') {
4639                        return Some(CpythonDiagnostic::new(
4640                            format!(
4641                                "{prefix}: newlines are not allowed in format specifiers for single quoted {prefix}s"
4642                            ),
4643                            quote_index,
4644                            quote_index + 1,
4645                        ));
4646                    }
4647                }
4648                index += 1;
4649            }
4650            _ => index += 1,
4651        }
4652    }
4653    None
4654}
4655
4656fn interpolated_string_prefix(bytes: &[u8], quote: usize) -> Option<&'static str> {
4657    let prev = quote.checked_sub(1).and_then(|index| bytes.get(index))?;
4658    let lower_prev = prev.to_ascii_lowercase();
4659    let (prefix_start, marker) = if matches!(lower_prev, b'f' | b't') {
4660        if quote >= 2 && bytes[quote - 2].eq_ignore_ascii_case(&b'r') {
4661            (quote - 2, lower_prev)
4662        } else {
4663            (quote - 1, lower_prev)
4664        }
4665    } else if lower_prev == b'r'
4666        && quote >= 2
4667        && matches!(bytes[quote - 2].to_ascii_lowercase(), b'f' | b't')
4668    {
4669        (quote - 2, bytes[quote - 2].to_ascii_lowercase())
4670    } else {
4671        return None;
4672    };
4673
4674    if prefix_start > 0 && identifier_continue_before(bytes, prefix_start) {
4675        return None;
4676    }
4677
4678    Some(if marker == b'f' {
4679        "f-string"
4680    } else {
4681        "t-string"
4682    })
4683}
4684
4685const MAXFSTRINGLEVEL: usize = 150;
4686
4687fn skip_string_token(bytes: &[u8], quote_index: usize) -> usize {
4688    if interpolated_string_prefix(bytes, quote_index).is_some() {
4689        interpolated_string_end(bytes, quote_index).unwrap_or(bytes.len())
4690    } else {
4691        skip_quoted_string(bytes, quote_index)
4692    }
4693}
4694
4695fn interpolated_string_end(bytes: &[u8], quote_index: usize) -> Option<usize> {
4696    let quote = bytes[quote_index];
4697    let triple =
4698        bytes.get(quote_index + 1) == Some(&quote) && bytes.get(quote_index + 2) == Some(&quote);
4699    let quote_len = if triple { 3 } else { 1 };
4700    let mut index = quote_index + quote_len;
4701    let mut brace_depth = 0usize;
4702    while index < bytes.len() {
4703        if brace_depth == 0 {
4704            if bytes[index] == b'\\' {
4705                index = (index + 2).min(bytes.len());
4706                continue;
4707            }
4708            if triple
4709                && bytes.get(index) == Some(&quote)
4710                && bytes.get(index + 1) == Some(&quote)
4711                && bytes.get(index + 2) == Some(&quote)
4712            {
4713                return Some(index + 3);
4714            }
4715            if !triple && bytes[index] == quote {
4716                return Some(index + 1);
4717            }
4718            if bytes[index] == b'{' {
4719                if bytes.get(index + 1) == Some(&b'{') {
4720                    index += 2;
4721                } else {
4722                    brace_depth = 1;
4723                    index += 1;
4724                }
4725            } else if bytes[index] == b'}' && bytes.get(index + 1) == Some(&b'}') {
4726                index += 2;
4727            } else {
4728                index += 1;
4729            }
4730        } else {
4731            match bytes[index] {
4732                quote_ch @ (b'\'' | b'"') => {
4733                    if quoted_string_is_closed(bytes, index) {
4734                        index = skip_string_token(bytes, index).max(index + 1);
4735                    } else if !triple && quote_ch == quote {
4736                        return Some(index + 1);
4737                    } else {
4738                        index = skip_string_token(bytes, index).max(index + 1);
4739                    }
4740                }
4741                b'#' => {
4742                    while index < bytes.len() && bytes[index] != b'\n' {
4743                        index += 1;
4744                    }
4745                }
4746                b'{' => {
4747                    brace_depth += 1;
4748                    index += 1;
4749                }
4750                b'}' => {
4751                    brace_depth -= 1;
4752                    index += 1;
4753                }
4754                b'\\' => index = (index + 2).min(bytes.len()),
4755                _ => index += 1,
4756            }
4757        }
4758    }
4759    None
4760}
4761
4762fn quote_len_at(bytes: &[u8], quote_index: usize) -> usize {
4763    let quote = bytes[quote_index];
4764    if bytes.get(quote_index + 1) == Some(&quote) && bytes.get(quote_index + 2) == Some(&quote) {
4765        3
4766    } else {
4767        1
4768    }
4769}
4770
4771fn interpolated_string_content_range(bytes: &[u8], quote_index: usize) -> Option<(usize, usize)> {
4772    let end = interpolated_string_end(bytes, quote_index)?;
4773    let quote_len = quote_len_at(bytes, quote_index);
4774    Some((quote_index + quote_len, end - quote_len))
4775}
4776
4777fn quoted_string_content_range(
4778    bytes: &[u8],
4779    quote_index: usize,
4780    quote: u8,
4781) -> Option<(usize, usize)> {
4782    let triple =
4783        bytes.get(quote_index + 1) == Some(&quote) && bytes.get(quote_index + 2) == Some(&quote);
4784    let quote_len = if triple { 3 } else { 1 };
4785    let content_start = quote_index + quote_len;
4786    let mut index = content_start;
4787    while index < bytes.len() {
4788        if bytes[index] == b'\\' {
4789            index = (index + 2).min(bytes.len());
4790        } else if (triple
4791            && bytes.get(index) == Some(&quote)
4792            && bytes.get(index + 1) == Some(&quote)
4793            && bytes.get(index + 2) == Some(&quote))
4794            || (!triple && bytes[index] == quote)
4795        {
4796            return Some((content_start, index));
4797        } else {
4798            index += 1;
4799        }
4800    }
4801    None
4802}
4803
4804fn invalid_replacement_field_error(
4805    bytes: &[u8],
4806    start: usize,
4807    end: usize,
4808    prefix: &str,
4809) -> Option<CpythonDiagnostic> {
4810    let mut index = start;
4811    while index < end {
4812        match bytes[index] {
4813            b'{' if bytes.get(index + 1) == Some(&b'{') => index += 2,
4814            b'}' if bytes.get(index + 1) == Some(&b'}') => index += 2,
4815            b'{' => {
4816                if let Some(error) = replacement_field_error(bytes, index, end, prefix, 0) {
4817                    return Some(error);
4818                }
4819                // Resume in the literal text. Inside the field a `#` starts a comment and a `{`
4820                // opens no new field, so walking on byte by byte would misread both.
4821                index = replacement_field_end(bytes, index, end).unwrap_or(end);
4822            }
4823            _ => index += 1,
4824        }
4825    }
4826    None
4827}
4828
4829/// Position just past a replacement field's closing brace. The expression part is code, where a
4830/// comment runs to the end of the line; past the separator only nested fields nest.
4831fn replacement_field_end(bytes: &[u8], open: usize, end: usize) -> Option<usize> {
4832    let expr_start = skip_replacement_field_trivia(bytes, open + 1, end);
4833    let separator = replacement_field_separator(bytes, expr_start, end)?;
4834    if bytes[separator] == b'}' {
4835        return Some(separator + 1);
4836    }
4837    replacement_field_closing_brace(bytes, separator + 1, end).map(|brace| brace + 1)
4838}
4839
4840/// The one-based line holding `offset`. Counted on demand: these scanners only run once the
4841/// source has failed to parse, and tracking a line through the spans they skip would have to
4842/// re-scan them anyway.
4843fn line_of_offset(bytes: &[u8], offset: usize) -> usize {
4844    1 + bytes[..offset]
4845        .iter()
4846        .filter(|byte| **byte == b'\n')
4847        .count()
4848}
4849
4850/// Brackets opened in a replacement field's expression must close before the field does. CPython
4851/// words these exactly as it does for brackets anywhere else, except that a closer with nothing
4852/// open is attributed to the literal. `start..end` must cover only the expression: a format spec
4853/// is plain text, so a bracket there opens nothing.
4854fn replacement_field_bracket_error(
4855    bytes: &[u8],
4856    start: usize,
4857    end: usize,
4858    prefix: &str,
4859) -> Option<CpythonDiagnostic> {
4860    // Each entry carries where its bracket opened, so a mismatch can name that line.
4861    let mut stack: Vec<(u8, usize)> = Vec::new();
4862    let mut index = start;
4863    while index < end {
4864        let byte = bytes[index];
4865        match byte {
4866            b'\'' | b'"' => {
4867                index = skip_quoted_string(bytes, index);
4868                continue;
4869            }
4870            b'#' => {
4871                index = skip_replacement_field_comment(bytes, index, end);
4872                continue;
4873            }
4874            b'(' | b'[' | b'{' => stack.push((byte, index)),
4875            b')' | b']' | b'}' => {
4876                let expected = expected_opening_bracket(byte as char) as u8;
4877                match stack.last() {
4878                    // The field's own closing brace: everything inside it balanced.
4879                    None if byte == b'}' => return None,
4880                    None => {
4881                        return Some(CpythonDiagnostic::new(
4882                            format!("{prefix}: unmatched '{}'", byte as char),
4883                            index,
4884                            index + 1,
4885                        ));
4886                    }
4887                    Some(&(opening, _)) if opening == expected => {
4888                        stack.pop();
4889                    }
4890                    Some(&(opening, opened_at)) => {
4891                        // CPython names the opening line only when it is not the closing one,
4892                        // comparing `parenlinenostack[level]` against the current `lineno`.
4893                        let opening_line = line_of_offset(bytes, opened_at);
4894                        let suffix = if opening_line == line_of_offset(bytes, index) {
4895                            String::new()
4896                        } else {
4897                            format!(" on line {opening_line}")
4898                        };
4899                        return Some(CpythonDiagnostic::new(
4900                            format!(
4901                                "closing parenthesis '{}' does not match opening parenthesis '{}'{suffix}",
4902                                byte as char, opening as char
4903                            ),
4904                            index,
4905                            index + 1,
4906                        ));
4907                    }
4908                }
4909            }
4910            _ => {}
4911        }
4912        index += 1;
4913    }
4914    None
4915}
4916
4917/// A `#` in a replacement field's expression starts a comment. When no newline follows it before
4918/// the end of the literal, the comment swallows the closing brace and the field can never close.
4919/// `start..end` must cover only the expression: `#` is an ordinary character in a format spec,
4920/// where it selects the alternate form.
4921fn replacement_field_comment_error(
4922    bytes: &[u8],
4923    open: usize,
4924    start: usize,
4925    end: usize,
4926    literal_end: usize,
4927) -> Option<CpythonDiagnostic> {
4928    let mut index = start;
4929    while index < end {
4930        match bytes[index] {
4931            b'\'' | b'"' => {
4932                index = skip_quoted_string(bytes, index);
4933                continue;
4934            }
4935            b'#' => {
4936                if !bytes[index..literal_end].contains(&b'\n') {
4937                    return Some(
4938                        CpythonDiagnostic::new("'{' was never closed".to_owned(), open, open + 1)
4939                            .with_unclosed_bracket(),
4940                    );
4941                }
4942                index = skip_replacement_field_comment(bytes, index, end);
4943                continue;
4944            }
4945            _ => index += 1,
4946        }
4947    }
4948    None
4949}
4950
4951/// Advances past a comment, stopping on the newline that ends it.
4952fn skip_replacement_field_comment(bytes: &[u8], mut index: usize, end: usize) -> usize {
4953    while index < end && bytes[index] != b'\n' {
4954        index += 1;
4955    }
4956    index
4957}
4958
4959/// Whitespace and comments may both precede a replacement field's expression.
4960fn skip_replacement_field_trivia(bytes: &[u8], mut index: usize, end: usize) -> usize {
4961    loop {
4962        index = skip_ascii_whitespace(bytes, index, end);
4963        if index < end && bytes[index] == b'#' {
4964            index = skip_replacement_field_comment(bytes, index, end);
4965            continue;
4966        }
4967        return index;
4968    }
4969}
4970
4971/// Bounds the mutual recursion between a replacement field and the format spec it contains,
4972/// so that pathologically nested input cannot exhaust the stack.
4973const MAX_REPLACEMENT_FIELD_DEPTH: usize = 32;
4974
4975fn replacement_field_error(
4976    bytes: &[u8],
4977    open: usize,
4978    end: usize,
4979    prefix: &str,
4980    depth: usize,
4981) -> Option<CpythonDiagnostic> {
4982    let expr_start = skip_replacement_field_trivia(bytes, open + 1, end);
4983
4984    // The expression runs up to the field's first top-level `=`, `!`, `:` or `}`. Everything past
4985    // it is a conversion or a format spec, which are text rather than code, so the checks that
4986    // treat the field as code have to stop here.
4987    let separator = replacement_field_separator(bytes, expr_start, end);
4988    let expr_end = separator.unwrap_or(end);
4989
4990    if let Some(backslash) = replacement_field_line_continuation(bytes, expr_start, expr_end) {
4991        return Some(CpythonDiagnostic::new(
4992            "unexpected character after line continuation character".to_owned(),
4993            backslash + 1,
4994            (backslash + 2).min(end),
4995        ));
4996    }
4997    if let Some(quote) = unterminated_string_in_replacement_field(bytes, expr_start, expr_end) {
4998        return Some(CpythonDiagnostic::new(
4999            unterminated_string_message(1, false, false, None),
5000            quote,
5001            quote + 1,
5002        ));
5003    }
5004    // Brackets are reported ahead of everything else in the field, and a comment that swallows
5005    // the closing brace ahead of the expression itself. The comment scan starts at the brace
5006    // because a comment may also sit between it and the expression.
5007    if let Some(error) = replacement_field_bracket_error(bytes, expr_start, expr_end, prefix) {
5008        return Some(error);
5009    }
5010    if let Some(error) = replacement_field_comment_error(bytes, open, open + 1, expr_end, end) {
5011        return Some(error);
5012    }
5013
5014    // Nothing follows `{` at all, so the field simply never closed. CPython asks for the
5015    // brace here; an expression that merely fails to start is reported further down.
5016    if expr_start >= end {
5017        return Some(CpythonDiagnostic::new(
5018            format!("{prefix}: expecting '}}'"),
5019            open,
5020            open + 1,
5021        ));
5022    }
5023
5024    if let marker @ (b'=' | b'!' | b':' | b'}') = bytes[expr_start]
5025        && is_replacement_field_marker(bytes, expr_start)
5026    {
5027        return Some(CpythonDiagnostic::new(
5028            format!(
5029                "{prefix}: valid expression required before '{}'",
5030                marker as char
5031            ),
5032            expr_start,
5033            expr_start + 1,
5034        ));
5035    }
5036
5037    // A `{` display is a valid expression, but only if something can follow it. CPython
5038    // reports the inner brace when it cannot, as in `f'{3:{{>10}'`.
5039    if bytes[expr_start] == b'{' {
5040        let inner = skip_ascii_whitespace(bytes, expr_start + 1, end);
5041        if inner < end
5042            && bytes[inner] != b'}'
5043            && invalid_replacement_expression_start(bytes, inner, end)
5044        {
5045            return Some(CpythonDiagnostic::new(
5046                format!("{prefix}: expecting a valid expression after '{{'"),
5047                expr_start,
5048                expr_start + 1,
5049            ));
5050        }
5051    }
5052
5053    if starts_identifier(bytes, expr_start, b"lambda") {
5054        return Some(CpythonDiagnostic::new(
5055            format!("{prefix}: lambda expressions are not allowed without parentheses"),
5056            expr_start,
5057            expr_start + b"lambda".len(),
5058        ));
5059    }
5060
5061    if invalid_replacement_expression_start(bytes, expr_start, end) {
5062        return Some(CpythonDiagnostic::new(
5063            format!("{prefix}: expecting a valid expression after '{{'"),
5064            open,
5065            open + 1,
5066        ));
5067    }
5068
5069    // Adjacent atoms in a replacement field are a missing comma, not an
5070    // f-string "expecting '}'" at the opening brace. Prefix `yield`/`await`/
5071    // `not` and continuation keywords (`and`, `for`, ...) are not a second atom.
5072    if !starts_identifier(bytes, expr_start, b"yield")
5073        && !starts_identifier(bytes, expr_start, b"await")
5074        && !starts_identifier(bytes, expr_start, b"not")
5075        && let Some(atom_end) = adjacent_atom_end(bytes, expr_start)
5076    {
5077        let next = skip_ascii_whitespace(bytes, atom_end, expr_end);
5078        if next > atom_end
5079            && expression_atom_start(bytes, next)
5080            && !expression_continuation_keyword(bytes, next)
5081        {
5082            let second_end = adjacent_atom_end(bytes, next).unwrap_or(next + 1);
5083            return Some(CpythonDiagnostic::new(
5084                "invalid syntax. Perhaps you forgot a comma?".to_owned(),
5085                expr_start,
5086                second_end,
5087            ));
5088        }
5089    }
5090
5091    // The expression started fine but ran into a character that cannot continue it. CPython
5092    // points at that character and lists the separators it wanted instead.
5093    if let Some(stray) = replacement_expression_stray_character(bytes, expr_start, expr_end) {
5094        return Some(CpythonDiagnostic::new(
5095            format!("{prefix}: expecting '=', or '!', or ':', or '}}'"),
5096            stray,
5097            stray + 1,
5098        ));
5099    }
5100
5101    let Some(separator) = separator else {
5102        return Some(CpythonDiagnostic::new(
5103            format!("{prefix}: expecting '}}'"),
5104            open,
5105            open + 1,
5106        ));
5107    };
5108
5109    // Only now that the field is known to close. A stray character above is a finished token
5110    // the parser rejects on lookahead, but a dangling operator makes it ask for one more, and
5111    // producing that token runs into the literal's closing quote: lexer.c then answers from
5112    // `INSIDE_FSTRING(tok)` with "%c-string: expecting '}'" before any `invalid_` rule is tried.
5113    if let Some(operator) = dangling_operator(bytes, expr_start, expr_end) {
5114        return Some(CpythonDiagnostic::new(
5115            format!("{prefix}: expecting '=', or '!', or ':', or '}}'"),
5116            operator,
5117            operator + 1,
5118        ));
5119    }
5120
5121    if let Some(lambda_at) = top_level_lambda(bytes, expr_start, expr_end)
5122        && lambda_allowed_at_expression_position(bytes, expr_start, lambda_at)
5123    {
5124        return Some(CpythonDiagnostic::new(
5125            format!("{prefix}: lambda expressions are not allowed without parentheses"),
5126            lambda_at,
5127            lambda_at + b"lambda".len(),
5128        ));
5129    }
5130
5131    if bytes[separator] == b':'
5132        && replacement_expression_has_parse_error(bytes, expr_start, separator)
5133    {
5134        return Some(CpythonDiagnostic::new(
5135            "invalid syntax".to_owned(),
5136            expr_start,
5137            separator,
5138        ));
5139    }
5140
5141    match bytes[separator] {
5142        b'=' => invalid_debug_expression_error(bytes, separator, end, prefix, depth),
5143        b'!' => invalid_conversion_error(bytes, separator, end, prefix, depth),
5144        b':' => invalid_format_spec_error(bytes, separator, end, prefix, depth),
5145        b'}' => None,
5146        _ => unreachable!(),
5147    }
5148}
5149
5150fn replacement_field_line_continuation(
5151    bytes: &[u8],
5152    mut index: usize,
5153    end: usize,
5154) -> Option<usize> {
5155    let mut level = 0usize;
5156    while index < end {
5157        match bytes[index] {
5158            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
5159            b'#' => index = skip_replacement_field_comment(bytes, index, end),
5160            b'\\' => return Some(index),
5161            b'(' | b'[' | b'{' => {
5162                level += 1;
5163                index += 1;
5164            }
5165            b')' | b']' | b'}' if level > 0 => {
5166                level -= 1;
5167                index += 1;
5168            }
5169            b'=' | b'!' | b':' | b'}'
5170                if level == 0 && is_replacement_field_marker(bytes, index) =>
5171            {
5172                return None;
5173            }
5174            _ => index += 1,
5175        }
5176    }
5177    None
5178}
5179
5180fn unterminated_string_in_replacement_field(
5181    bytes: &[u8],
5182    mut index: usize,
5183    end: usize,
5184) -> Option<usize> {
5185    while index < end {
5186        match bytes[index] {
5187            quote @ (b'\'' | b'"') => {
5188                let string_end = skip_quoted_string(bytes, index);
5189                if string_end >= end && !bytes[index + 1..end].contains(&quote) {
5190                    return Some(index);
5191                }
5192                index = string_end;
5193            }
5194            b'#' => index = skip_replacement_field_comment(bytes, index, end),
5195            _ => index += 1,
5196        }
5197    }
5198    None
5199}
5200
5201fn top_level_lambda(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
5202    let mut level = 0usize;
5203    while index < end {
5204        match bytes[index] {
5205            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
5206            b'#' => index = skip_replacement_field_comment(bytes, index, end),
5207            b'(' | b'[' | b'{' => {
5208                level += 1;
5209                index += 1;
5210            }
5211            b')' | b']' | b'}' if level > 0 => {
5212                level -= 1;
5213                index += 1;
5214            }
5215            _ if level == 0 && starts_identifier(bytes, index, b"lambda") => {
5216                return Some(index);
5217            }
5218            _ => index += 1,
5219        }
5220    }
5221    None
5222}
5223
5224/// Unparenthesized `lambda` is allowed in the same places a `lambdef` can appear:
5225/// at the start of the field or after a top-level comma.
5226fn lambda_allowed_at_expression_position(bytes: &[u8], mut index: usize, lambda_at: usize) -> bool {
5227    let mut after_comma = true;
5228    let mut level = 0usize;
5229    while index < lambda_at {
5230        match bytes[index] {
5231            b'\'' | b'"' => {
5232                after_comma = false;
5233                index = skip_quoted_string(bytes, index);
5234            }
5235            b'#' => index = skip_replacement_field_comment(bytes, index, lambda_at),
5236            b'(' | b'[' | b'{' => {
5237                level += 1;
5238                after_comma = false;
5239                index += 1;
5240            }
5241            b')' | b']' | b'}' if level > 0 => {
5242                level -= 1;
5243                after_comma = false;
5244                index += 1;
5245            }
5246            b',' if level == 0 => {
5247                after_comma = true;
5248                index += 1;
5249            }
5250            b' ' | b'\t' | b'\n' | b'\r' | b'\x0c' => index += 1,
5251            _ => {
5252                after_comma = false;
5253                index += 1;
5254            }
5255        }
5256    }
5257    after_comma
5258}
5259
5260fn replacement_expression_has_parse_error(bytes: &[u8], start: usize, end: usize) -> bool {
5261    let Ok(expression) = ::core::str::from_utf8(&bytes[start..end]) else {
5262        return false;
5263    };
5264    parser::parse_expression(expression).is_err()
5265}
5266
5267fn invalid_replacement_expression_start(bytes: &[u8], index: usize, end: usize) -> bool {
5268    if index >= end {
5269        return true;
5270    }
5271
5272    // A leading `.` is Ellipsis or a float without an integer part, both of which start a
5273    // perfectly good expression.
5274    if bytes[index] == b'.'
5275        && (bytes[index..end].starts_with(b"...")
5276            || bytes
5277                .get(index + 1)
5278                .is_some_and(|byte| byte.is_ascii_digit()))
5279    {
5280        return false;
5281    }
5282
5283    if matches!(
5284        bytes[index],
5285        b'.' | b',' | b'*' | b'/' | b'%' | b'&' | b'|' | b'^' | b'<' | b'>' | b'@' | b'=' | b'!'
5286    ) || is_stray_in_replacement_expression(bytes[index])
5287    {
5288        return true;
5289    }
5290
5291    if matches!(bytes[index], b'+' | b'-' | b'~') {
5292        let operand = skip_ascii_whitespace(bytes, index + 1, end);
5293        if starts_identifier(bytes, operand, b"lambda") {
5294            return true;
5295        }
5296        return !bytes.get(operand).is_some_and(|byte| {
5297            *byte >= 0x80
5298                || *byte == b'_'
5299                || byte.is_ascii_alphabetic()
5300                || byte.is_ascii_digit()
5301                || matches!(*byte, b'\'' | b'"' | b'(' | b'[' | b'{')
5302                // `-.5` is a signed float.
5303                || (*byte == b'.'
5304                    && bytes
5305                        .get(operand + 1)
5306                        .is_some_and(|next| next.is_ascii_digit()))
5307        });
5308    }
5309
5310    [
5311        b"and".as_slice(),
5312        b"as".as_slice(),
5313        b"else".as_slice(),
5314        b"for".as_slice(),
5315        b"if".as_slice(),
5316        b"in".as_slice(),
5317        b"is".as_slice(),
5318        b"or".as_slice(),
5319    ]
5320    .iter()
5321    .any(|keyword| starts_identifier(bytes, index, keyword))
5322}
5323
5324/// A top-level `=` or `!` marks a debug specifier or a conversion only when it is not part of a
5325/// longer operator. CPython's tokenizer consumes `==`, `!=`, `<=`, `+=` and the rest as single
5326/// tokens before it ever considers the debug marker, so those must not end the expression here.
5327fn is_replacement_field_marker(bytes: &[u8], index: usize) -> bool {
5328    match bytes[index] {
5329        b'=' => {
5330            if bytes.get(index + 1) == Some(&b'=') {
5331                return false;
5332            }
5333            !index
5334                .checked_sub(1)
5335                .and_then(|previous| bytes.get(previous).copied())
5336                .is_some_and(precedes_equals_in_one_operator)
5337        }
5338        b'!' => bytes.get(index + 1) != Some(&b'='),
5339        _ => true,
5340    }
5341}
5342
5343fn replacement_field_separator(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
5344    let mut level = 0usize;
5345    while index < end {
5346        match bytes[index] {
5347            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
5348            b'#' => index = skip_replacement_field_comment(bytes, index, end),
5349            b'(' | b'[' | b'{' => {
5350                level += 1;
5351                index += 1;
5352            }
5353            b')' | b']' | b'}' if level > 0 => {
5354                level -= 1;
5355                index += 1;
5356            }
5357            b'=' | b'!' | b':' | b'}'
5358                if level == 0 && is_replacement_field_marker(bytes, index) =>
5359            {
5360                return Some(index);
5361            }
5362            _ => index += 1,
5363        }
5364    }
5365    None
5366}
5367
5368/// Characters that can neither start nor continue a replacement field expression. CPython's
5369/// interpolated-string tokenizer stops at them and asks for a separator instead.
5370const fn is_stray_in_replacement_expression(byte: u8) -> bool {
5371    matches!(byte, b';' | b'$' | b'?' | b'`')
5372}
5373
5374fn replacement_expression_stray_character(
5375    bytes: &[u8],
5376    mut index: usize,
5377    end: usize,
5378) -> Option<usize> {
5379    while index < end {
5380        match bytes[index] {
5381            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
5382            b'#' => index = skip_replacement_field_comment(bytes, index, end),
5383            byte if is_stray_in_replacement_expression(byte) => return Some(index),
5384            _ => index += 1,
5385        }
5386    }
5387    None
5388}
5389
5390/// A byte that pairs with a following `=` to make one operator token, so that the `=` belongs
5391/// to it rather than opening a debug specifier: `==`, `!=`, `<=`, `>=`, the augmented
5392/// assignments, and the `:=` walrus. Against `is_dangling_operator_byte` below, this has `:`
5393/// and lacks `.` and `~`, because neither `.=` nor `~=` is a Python operator.
5394const fn precedes_equals_in_one_operator(byte: u8) -> bool {
5395    matches!(
5396        byte,
5397        b'!' | b'%'
5398            | b'&'
5399            | b'*'
5400            | b'+'
5401            | b'-'
5402            | b'/'
5403            | b':'
5404            | b'<'
5405            | b'='
5406            | b'>'
5407            | b'@'
5408            | b'^'
5409            | b'|'
5410    )
5411}
5412
5413/// Characters that make up an operator which cannot end an expression. Against
5414/// `precedes_equals_in_one_operator` above, this has `.` and `~` for `a.b.` and `~a`, and lacks
5415/// `:`, which separates a replacement field's parts rather than operating on anything.
5416const fn is_dangling_operator_byte(byte: u8) -> bool {
5417    matches!(
5418        byte,
5419        b'!' | b'%'
5420            | b'&'
5421            | b'*'
5422            | b'+'
5423            | b'-'
5424            | b'.'
5425            | b'/'
5426            | b'<'
5427            | b'='
5428            | b'>'
5429            | b'@'
5430            | b'^'
5431            | b'|'
5432            | b'~'
5433    )
5434}
5435
5436/// The identifier that ends just before `word`, skipping the whitespace between them.
5437fn preceding_keyword(bytes: &[u8], start: usize, word: usize) -> Option<(usize, &[u8])> {
5438    let mut cursor = word;
5439    while cursor > start && matches!(bytes[cursor - 1], b' ' | b'\t' | b'\r' | b'\n' | 0x0c) {
5440        cursor -= 1;
5441    }
5442    let stop = cursor;
5443    while cursor > start && is_ascii_identifier_char(bytes[cursor - 1]) {
5444        cursor -= 1;
5445    }
5446    (cursor < stop).then(|| (cursor, &bytes[cursor..stop]))
5447}
5448
5449/// A replacement field expression cannot end on an operator that still wants an operand.
5450/// CPython's tokenizer then reaches the field brace with the expression unfinished and asks for
5451/// a separator, pointing at the operator that left it that way. Returns that operator's start.
5452fn dangling_operator(bytes: &[u8], start: usize, end: usize) -> Option<usize> {
5453    // The region cannot be read backwards: the newline that ends a comment is whitespace, so
5454    // trimming from `end` would step into the comment body. Walk forward instead and keep the
5455    // position just past the last byte that is really part of the expression.
5456    let mut index = start;
5457    let mut tail = start;
5458    while index < end {
5459        match bytes[index] {
5460            b'\'' | b'"' => {
5461                index = skip_quoted_string(bytes, index).min(end);
5462                tail = index;
5463            }
5464            b'#' => index = skip_replacement_field_comment(bytes, index, end),
5465            b' ' | b'\t' | b'\r' | b'\n' | 0x0c => index += 1,
5466            _ => {
5467                index += 1;
5468                tail = index;
5469            }
5470        }
5471    }
5472    if tail == start {
5473        return None;
5474    }
5475
5476    // A trailing keyword operator reads back as an identifier.
5477    if is_ascii_identifier_char(bytes[tail - 1]) {
5478        let mut word = tail;
5479        while word > start && is_ascii_identifier_char(bytes[word - 1]) {
5480            word -= 1;
5481        }
5482        if ![
5483            b"and".as_slice(),
5484            b"or".as_slice(),
5485            b"not".as_slice(),
5486            b"in".as_slice(),
5487            b"is".as_slice(),
5488            b"if".as_slice(),
5489            b"for".as_slice(),
5490        ]
5491        .iter()
5492        .any(|keyword| &bytes[word..tail] == *keyword)
5493        {
5494            return None;
5495        }
5496        // `is not` and `not in` are single operators, and CPython points at their first word.
5497        let previous = preceding_keyword(bytes, start, word);
5498        return Some(match (previous, &bytes[word..tail]) {
5499            (Some((at, b"is")), b"not") | (Some((at, b"not")), b"in") => at,
5500            _ => word,
5501        });
5502    }
5503
5504    if !is_dangling_operator_byte(bytes[tail - 1]) {
5505        return None;
5506    }
5507    let mut token = tail;
5508    while token > start && is_dangling_operator_byte(bytes[token - 1]) {
5509        token -= 1;
5510    }
5511
5512    // `1.` is a float: that dot finishes the literal rather than dangling.
5513    if bytes[token..tail] == *b"." && token > start && bytes[token - 1].is_ascii_digit() {
5514        return None;
5515    }
5516    // `...` is Ellipsis, so consume whole triples: a leftover dot is what dangles.
5517    if bytes[token..tail].iter().all(|byte| *byte == b'.') {
5518        let leftover = (tail - token) % 3;
5519        return (leftover != 0).then(|| tail - leftover);
5520    }
5521    Some(token)
5522}
5523
5524fn invalid_debug_expression_error(
5525    bytes: &[u8],
5526    equals: usize,
5527    end: usize,
5528    prefix: &str,
5529    depth: usize,
5530) -> Option<CpythonDiagnostic> {
5531    let next = equals + 1;
5532    if next >= end {
5533        return None;
5534    }
5535    // A conversion or format spec may follow `=`, and is validated exactly as it would be
5536    // when following the expression directly.
5537    match bytes[next] {
5538        b'!' => return invalid_conversion_error(bytes, next, end, prefix, depth),
5539        b':' => return invalid_format_spec_error(bytes, next, end, prefix, depth),
5540        b'}' => return None,
5541        _ => {}
5542    }
5543    Some(CpythonDiagnostic::new(
5544        format!("{prefix}: expecting '!', or ':', or '}}'"),
5545        next,
5546        next.saturating_add(1).min(end),
5547    ))
5548}
5549
5550fn invalid_conversion_error(
5551    bytes: &[u8],
5552    bang: usize,
5553    end: usize,
5554    prefix: &str,
5555    depth: usize,
5556) -> Option<CpythonDiagnostic> {
5557    let next = bang + 1;
5558    if next >= end {
5559        return Some(CpythonDiagnostic::new(
5560            format!("{prefix}: expecting '}}'"),
5561            bang,
5562            bang + 1,
5563        ));
5564    }
5565
5566    if bytes[next].is_ascii_whitespace() {
5567        let following = skip_ascii_whitespace(bytes, next, end);
5568        let message = if bytes
5569            .get(following)
5570            .is_some_and(|byte| byte.is_ascii_alphabetic() || *byte == b'_')
5571        {
5572            "conversion type must come right after the exclamation mark"
5573        } else {
5574            "missing conversion character"
5575        };
5576        return Some(CpythonDiagnostic::new(
5577            format!("{prefix}: {message}"),
5578            next,
5579            next + 1,
5580        ));
5581    }
5582
5583    if matches!(bytes[next], b':' | b'}') {
5584        return Some(CpythonDiagnostic::new(
5585            format!("{prefix}: missing conversion character"),
5586            next,
5587            next + 1,
5588        ));
5589    }
5590
5591    // A non-ASCII character is still a conversion character as far as CPython is concerned,
5592    // so it gets named in the message like any other invalid one.
5593    if bytes[next] >= 0x80
5594        && let Some(character) = ::core::str::from_utf8(&bytes[next..end])
5595            .ok()
5596            .and_then(|text| text.chars().next())
5597    {
5598        return Some(CpythonDiagnostic::new(
5599            format!(
5600                "{prefix}: invalid conversion character '{character}': expected 's', 'r', or 'a'"
5601            ),
5602            next,
5603            next + character.len_utf8(),
5604        ));
5605    }
5606
5607    if !bytes[next].is_ascii_alphabetic() && bytes[next] != b'_' {
5608        return Some(CpythonDiagnostic::new(
5609            format!("{prefix}: invalid conversion character"),
5610            next,
5611            next + 1,
5612        ));
5613    }
5614
5615    let conversion_end = identifier_end(bytes, next, end);
5616    let conversion = &bytes[next..conversion_end];
5617    if !matches!(conversion, b"s" | b"r" | b"a") {
5618        let conversion = ::core::str::from_utf8(conversion).unwrap_or("");
5619        return Some(CpythonDiagnostic::new(
5620            format!(
5621                "{prefix}: invalid conversion character '{conversion}': expected 's', 'r', or 'a'"
5622            ),
5623            next,
5624            conversion_end,
5625        ));
5626    }
5627
5628    if conversion_end >= end {
5629        return None;
5630    }
5631
5632    match bytes[conversion_end] {
5633        b':' => return invalid_format_spec_error(bytes, conversion_end, end, prefix, depth),
5634        b'}' => return None,
5635        _ => {}
5636    }
5637
5638    Some(CpythonDiagnostic::new(
5639        format!("{prefix}: expecting ':' or '}}'"),
5640        conversion_end,
5641        conversion_end + 1,
5642    ))
5643}
5644
5645fn invalid_format_spec_error(
5646    bytes: &[u8],
5647    colon: usize,
5648    end: usize,
5649    prefix: &str,
5650    depth: usize,
5651) -> Option<CpythonDiagnostic> {
5652    // Unlike literal text, a format spec has no `{{` escape: every `{` opens a nested
5653    // replacement field, so the outer scanner cannot validate these for us.
5654    let mut index = colon + 1;
5655    while index < end {
5656        match bytes[index] {
5657            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
5658            b'{' => {
5659                if depth < MAX_REPLACEMENT_FIELD_DEPTH
5660                    && let Some(error) =
5661                        replacement_field_error(bytes, index, end, prefix, depth + 1)
5662                {
5663                    return Some(error);
5664                }
5665                // Resume after the nested field. Walking into it again from every enclosing
5666                // spec would re-scan the same bytes once per level, which costs O(2^depth).
5667                let Some(after) = replacement_field_end(bytes, index, end) else {
5668                    break;
5669                };
5670                index = after;
5671            }
5672            b'}' => return None,
5673            _ => index += 1,
5674        }
5675    }
5676    Some(CpythonDiagnostic::new(
5677        format!("{prefix}: expecting '}}', or format specs"),
5678        colon,
5679        colon + 1,
5680    ))
5681}
5682
5683fn replacement_field_closing_brace(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
5684    let mut level = 0usize;
5685    while index < end {
5686        match bytes[index] {
5687            b'\'' | b'"' => index = skip_quoted_string(bytes, index),
5688            b'{' => {
5689                level += 1;
5690                index += 1;
5691            }
5692            b'}' if level > 0 => {
5693                level -= 1;
5694                index += 1;
5695            }
5696            b'}' => return Some(index),
5697            _ => index += 1,
5698        }
5699    }
5700    None
5701}
5702
5703fn skip_ascii_whitespace(bytes: &[u8], mut index: usize, end: usize) -> usize {
5704    while index < end && matches!(bytes[index], b' ' | b'\t' | b'\r' | b'\n' | 0x0c) {
5705        index += 1;
5706    }
5707    index
5708}
5709
5710fn string_literal_end_at(bytes: &[u8], index: usize) -> Option<usize> {
5711    match bytes.get(index).copied()? {
5712        b'\'' | b'"' => Some(skip_quoted_string(bytes, index)),
5713        first if first.is_ascii_alphabetic() => {
5714            if matches!(bytes.get(index + 1), Some(b'\'' | b'"')) {
5715                return string_literal_prefix(bytes, index, index + 1)
5716                    .then(|| skip_quoted_string(bytes, index + 1));
5717            }
5718            if matches!(bytes.get(index + 2), Some(b'\'' | b'"')) {
5719                return string_literal_prefix(bytes, index, index + 2)
5720                    .then(|| skip_quoted_string(bytes, index + 2));
5721            }
5722            None
5723        }
5724        _ => None,
5725    }
5726}
5727
5728fn string_literal_prefix(bytes: &[u8], start: usize, quote: usize) -> bool {
5729    let prefix = &bytes[start..quote];
5730    let valid = matches!(
5731        prefix,
5732        b"b" | b"B"
5733            | b"r"
5734            | b"R"
5735            | b"u"
5736            | b"U"
5737            | b"f"
5738            | b"F"
5739            | b"t"
5740            | b"T"
5741            | b"br"
5742            | b"bR"
5743            | b"Br"
5744            | b"BR"
5745            | b"rb"
5746            | b"rB"
5747            | b"Rb"
5748            | b"RB"
5749            | b"fr"
5750            | b"fR"
5751            | b"Fr"
5752            | b"FR"
5753            | b"rf"
5754            | b"rF"
5755            | b"Rf"
5756            | b"RF"
5757            | b"tr"
5758            | b"tR"
5759            | b"Tr"
5760            | b"TR"
5761            | b"rt"
5762            | b"rT"
5763            | b"Rt"
5764            | b"RT"
5765    );
5766    valid && (start == 0 || !is_ascii_identifier_char(bytes[start - 1]))
5767}
5768
5769fn invalid_expression_error(source: &str) -> Option<CpythonDiagnostic> {
5770    invalid_string_expression_error(source).or_else(|| missing_comma_expression_error(source))
5771}
5772
5773fn invalid_string_expression_error(source: &str) -> Option<CpythonDiagnostic> {
5774    let bytes = source.as_bytes();
5775    let mut index = 0;
5776    while index < bytes.len() {
5777        if let Some(first_string_end) = string_literal_end_at(bytes, index) {
5778            let expr_start = skip_ascii_whitespace(bytes, first_string_end, bytes.len());
5779            if expression_atom_start(bytes, expr_start)
5780                && let Some(expr_end) = adjacent_atom_end(bytes, expr_start)
5781            {
5782                let next = skip_ascii_whitespace(bytes, expr_end, bytes.len());
5783                if string_literal_end_at(bytes, next).is_some() {
5784                    return Some(CpythonDiagnostic::new(
5785                        "invalid syntax. Is this intended to be part of the string?".to_owned(),
5786                        expr_start,
5787                        expr_end,
5788                    ));
5789                }
5790            }
5791            index = first_string_end;
5792        } else {
5793            index += 1;
5794        }
5795    }
5796    None
5797}
5798
5799fn missing_comma_expression_error(source: &str) -> Option<CpythonDiagnostic> {
5800    let bytes = source.as_bytes();
5801    let mut stack: Vec<u8> = Vec::new();
5802    let mut index = 0;
5803    while index < bytes.len() {
5804        if bytes[index] == b'#' {
5805            while index < bytes.len() && bytes[index] != b'\n' {
5806                index += 1;
5807            }
5808        } else if let Some(string_end) = string_literal_end_at(bytes, index) {
5809            index = string_end;
5810        } else {
5811            match bytes[index] {
5812                b'(' | b'[' | b'{' => {
5813                    if bytes[index] == b'[' && opening_bracket_is_class_type_params(bytes, index) {
5814                        let Some(close) = matching_delimiter(bytes, index, b']') else {
5815                            index += 1;
5816                            continue;
5817                        };
5818                        index = close + 1;
5819                        continue;
5820                    }
5821                    stack.push(bytes[index]);
5822                    index += 1;
5823                }
5824                b')' | b']' | b'}' => {
5825                    stack.pop();
5826                    index += 1;
5827                }
5828                _ if !stack.is_empty() && expression_continuation_keyword(bytes, index) => {
5829                    index = identifier_end(bytes, index, bytes.len());
5830                }
5831                byte if !stack.is_empty() && expression_atom_start_byte(byte) => {
5832                    // `yield x` / `await x` are prefix expressions, not two
5833                    // adjacent atoms missing a comma.
5834                    if starts_identifier(bytes, index, b"yield") {
5835                        index = identifier_end(bytes, index, bytes.len());
5836                        let after_yield = skip_ascii_whitespace(bytes, index, bytes.len());
5837                        if starts_identifier(bytes, after_yield, b"from") {
5838                            index = identifier_end(bytes, after_yield, bytes.len());
5839                        }
5840                        continue;
5841                    }
5842                    if starts_identifier(bytes, index, b"await") {
5843                        index = identifier_end(bytes, index, bytes.len());
5844                        continue;
5845                    }
5846                    let atom_end = adjacent_atom_end(bytes, index).unwrap_or(index + 1);
5847                    let next = skip_ascii_whitespace(bytes, atom_end, bytes.len());
5848                    if next > atom_end
5849                        && expression_atom_start(bytes, next)
5850                        && !expression_continuation_keyword(bytes, next)
5851                    {
5852                        // The diagnostic covers both atoms, so it ends past
5853                        // the whole second one.  `next + 1` ended past its
5854                        // first byte instead, which is only the same thing
5855                        // when that atom is one ASCII character.
5856                        let second_end = adjacent_atom_end(bytes, next).unwrap_or(next + 1);
5857                        return Some(CpythonDiagnostic::new(
5858                            "invalid syntax. Perhaps you forgot a comma?".to_owned(),
5859                            index,
5860                            second_end,
5861                        ));
5862                    }
5863                    index = atom_end;
5864                }
5865                _ => index += 1,
5866            }
5867        }
5868    }
5869    None
5870}
5871
5872fn opening_bracket_is_class_type_params(bytes: &[u8], bracket: usize) -> bool {
5873    let mut cursor = bracket;
5874    while cursor > 0 && matches!(bytes[cursor - 1], b' ' | b'\t' | b'\x0c') {
5875        cursor -= 1;
5876    }
5877    while cursor > 0
5878        && bytes
5879            .get(cursor - 1)
5880            .is_some_and(|byte| *byte >= 0x80 || is_ascii_identifier_char(*byte))
5881    {
5882        cursor -= 1;
5883    }
5884    while cursor > 0 && matches!(bytes[cursor - 1], b' ' | b'\t' | b'\x0c') {
5885        cursor -= 1;
5886    }
5887    cursor >= 5 && starts_identifier(bytes, cursor - 5, b"class")
5888}
5889
5890fn expression_continuation_keyword(bytes: &[u8], index: usize) -> bool {
5891    [
5892        b"and".as_slice(),
5893        b"else".as_slice(),
5894        b"for".as_slice(),
5895        b"if".as_slice(),
5896        b"in".as_slice(),
5897        b"is".as_slice(),
5898        b"not".as_slice(),
5899        b"or".as_slice(),
5900    ]
5901    .iter()
5902    .any(|keyword| starts_identifier(bytes, index, keyword))
5903}
5904
5905fn expression_atom_start(bytes: &[u8], index: usize) -> bool {
5906    bytes
5907        .get(index)
5908        .is_some_and(|byte| expression_atom_start_byte(*byte))
5909        || string_literal_end_at(bytes, index).is_some()
5910}
5911
5912fn expression_atom_start_byte(byte: u8) -> bool {
5913    byte >= 0x80
5914        || byte == b'_'
5915        || byte.is_ascii_alphabetic()
5916        || byte.is_ascii_digit()
5917        || matches!(byte, b'\'' | b'"' | b'(' | b'[' | b'{')
5918}
5919
5920fn adjacent_atom_end(bytes: &[u8], index: usize) -> Option<usize> {
5921    if let Some(string_end) = string_literal_end_at(bytes, index) {
5922        return Some(string_end);
5923    }
5924    match bytes.get(index).copied()? {
5925        byte if byte >= 0x80 || byte == b'_' || byte.is_ascii_alphabetic() => {
5926            Some(identifier_end(bytes, index, bytes.len()))
5927        }
5928        byte if byte.is_ascii_digit() => {
5929            let mut end = index + 1;
5930            while end < bytes.len() && (bytes[end].is_ascii_alphanumeric() || bytes[end] == b'_') {
5931                end += 1;
5932            }
5933            Some(end)
5934        }
5935        b'(' | b'[' | b'{' => Some(index + 1),
5936        _ => None,
5937    }
5938}
5939
5940struct OpenDelimiter {
5941    position: usize,
5942    opener: u8,
5943    in_format_spec: bool,
5944}
5945
5946/// An f-string or t-string that runs off the end with a replacement field still open reports
5947/// the brace rather than the missing quote, matching CPython — but only while the tokenizer is
5948/// still reading the field's *expression*. Past the field's own `:` it is emitting FSTRING_MIDDLE
5949/// again, so running out of input there is an unterminated literal like any other, and
5950/// `f'{a:>5` gets the quote message where `f'{a` gets the brace.
5951fn unclosed_replacement_field_error(
5952    bytes: &[u8],
5953    quote_start: usize,
5954    content_start: usize,
5955    content_end: usize,
5956) -> Option<CpythonDiagnostic> {
5957    interpolated_string_prefix(bytes, quote_start)?;
5958
5959    // Brackets and quotes are delimiters only once a field is open; in the literal's text they
5960    // are ordinary characters. And past a field's own format spec the tokenizer emits literal
5961    // text again, so a field still open at end of input is an unclosed brace only before that.
5962    let mut open: Vec<OpenDelimiter> = Vec::new();
5963    let mut index = content_start;
5964    while index < content_end {
5965        let inside_expression = matches!(
5966            open.last(),
5967            Some(OpenDelimiter {
5968                opener: b'{',
5969                in_format_spec: false,
5970                ..
5971            })
5972        );
5973        match bytes[index] {
5974            // `{{` and `}}` escape only in literal text, which is where a brace at depth zero is.
5975            b'{' if open.is_empty() && bytes.get(index + 1) == Some(&b'{') => index += 2,
5976            b'}' if open.is_empty() && bytes.get(index + 1) == Some(&b'}') => index += 2,
5977            b'{' => {
5978                open.push(OpenDelimiter {
5979                    position: index,
5980                    opener: b'{',
5981                    in_format_spec: false,
5982                });
5983                index += 1;
5984            }
5985            b'}' => {
5986                open.pop();
5987                index += 1;
5988            }
5989            byte @ (b'(' | b'[') if !open.is_empty() => {
5990                open.push(OpenDelimiter {
5991                    position: index,
5992                    opener: byte,
5993                    in_format_spec: false,
5994                });
5995                index += 1;
5996            }
5997            b')' | b']' if !open.is_empty() => {
5998                open.pop();
5999                index += 1;
6000            }
6001            // Only the field's own `:` opens a format spec — one in a slice, a display or a
6002            // lambda belongs to whatever bracket encloses it.
6003            b':' if inside_expression => {
6004                open.last_mut().expect("a brace is open").in_format_spec = true;
6005                index += 1;
6006            }
6007            b'\'' | b'"' if !open.is_empty() => {
6008                index = skip_quoted_string(bytes, index).min(content_end);
6009            }
6010            b'#' if inside_expression => {
6011                index = skip_replacement_field_comment(bytes, index, content_end);
6012            }
6013            _ => index += 1,
6014        }
6015    }
6016
6017    // CPython names the innermost unclosed delimiter, so a bracket opened inside the expression
6018    // takes the message from the field that encloses it.
6019    let &OpenDelimiter {
6020        position,
6021        opener,
6022        in_format_spec,
6023    } = open.last()?;
6024    (!in_format_spec).then(|| {
6025        CpythonDiagnostic::new(
6026            format!("'{}' was never closed", opener as char),
6027            position,
6028            position + 1,
6029        )
6030        .with_unclosed_bracket()
6031    })
6032}
6033
6034fn unterminated_string_message(
6035    detected_line: usize,
6036    triple: bool,
6037    has_escaped_quote: bool,
6038    prefix: Option<&str>,
6039) -> String {
6040    // CPython names the literal kind, e.g. "unterminated f-string literal".
6041    let kind = prefix.unwrap_or("string");
6042    if triple {
6043        format!("unterminated triple-quoted {kind} literal (detected at line {detected_line})")
6044    // The escaped-quote hint belongs to the plain-string branch of CPython's tokenizer alone.
6045    // Its interpolated branch has only the two forms above and below, so pairing the hint with
6046    // a prefix would invent a message CPython never emits.
6047    } else if has_escaped_quote && prefix.is_none() {
6048        format!(
6049            "unterminated {kind} literal (detected at line {detected_line}); perhaps you escaped the end quote?"
6050        )
6051    } else {
6052        format!("unterminated {kind} literal (detected at line {detected_line})")
6053    }
6054}
6055
6056fn expected_opening_bracket(closing: char) -> char {
6057    match closing {
6058        ')' => '(',
6059        ']' => '[',
6060        '}' => '{',
6061        _ => unreachable!(),
6062    }
6063}
6064
6065/// A bracket diagnostic, and whether it is an opener that was never closed. The caller needs
6066/// that apart from the message because ruff reports the unclosed case as an EOF error.
6067#[derive(Clone)]
6068struct BracketError {
6069    diagnostic: CpythonDiagnostic,
6070    unclosed: bool,
6071}
6072
6073fn bracket_syntax_error(source: &str) -> Option<BracketError> {
6074    let mut stack: Vec<(char, usize, usize)> = Vec::new();
6075    let mut in_string = false;
6076    let mut string_quote = '\0';
6077    let mut triple_quote = false;
6078    let mut escape_next = false;
6079    let mut is_raw_string = false;
6080    let mut line = 1usize;
6081
6082    let chars: Vec<(usize, char)> = source.char_indices().collect();
6083    let mut index = 0;
6084    while index < chars.len() {
6085        let (byte_offset, ch) = chars[index];
6086
6087        if ch == '\n' {
6088            line += 1;
6089        }
6090
6091        if escape_next {
6092            escape_next = false;
6093            index += 1;
6094            continue;
6095        }
6096
6097        if in_string {
6098            if ch == '\\' && !is_raw_string {
6099                escape_next = true;
6100            } else if triple_quote {
6101                if ch == string_quote
6102                    && index + 2 < chars.len()
6103                    && chars[index + 1].1 == string_quote
6104                    && chars[index + 2].1 == string_quote
6105                {
6106                    in_string = false;
6107                    index += 3;
6108                    continue;
6109                }
6110            } else if ch == string_quote {
6111                in_string = false;
6112            }
6113            index += 1;
6114            continue;
6115        }
6116
6117        if ch == '#' {
6118            while index < chars.len() && chars[index].1 != '\n' {
6119                index += 1;
6120            }
6121            continue;
6122        }
6123
6124        if ch == '\\' {
6125            match chars.get(index + 1).map(|(_, next)| *next) {
6126                Some('\n' | '\r') => {
6127                    escape_next = true;
6128                    index += 1;
6129                    continue;
6130                }
6131                Some(_) => {
6132                    index += 2;
6133                    continue;
6134                }
6135                None => {
6136                    index += 1;
6137                    continue;
6138                }
6139            }
6140        }
6141
6142        if ch == '\'' || ch == '"' {
6143            is_raw_string = false;
6144            for look_back in 1..=2.min(index) {
6145                let prev = chars[index - look_back].1;
6146                if matches!(prev, 'r' | 'R') {
6147                    is_raw_string = true;
6148                    break;
6149                }
6150                if !matches!(prev, 'b' | 'B' | 'f' | 'F' | 'u' | 'U') {
6151                    break;
6152                }
6153            }
6154            string_quote = ch;
6155            if index + 2 < chars.len() && chars[index + 1].1 == ch && chars[index + 2].1 == ch {
6156                triple_quote = true;
6157                in_string = true;
6158                index += 3;
6159                continue;
6160            }
6161            triple_quote = false;
6162            in_string = true;
6163            index += 1;
6164            continue;
6165        }
6166
6167        match ch {
6168            '(' | '[' | '{' => stack.push((ch, byte_offset, line)),
6169            ')' | ']' | '}' => {
6170                let expected = expected_opening_bracket(ch);
6171                let Some(&(opening, _, opening_line)) = stack.last() else {
6172                    return Some(BracketError {
6173                        diagnostic: CpythonDiagnostic::new(
6174                            format!("unmatched '{ch}'"),
6175                            byte_offset,
6176                            byte_offset,
6177                        ),
6178                        unclosed: false,
6179                    });
6180                };
6181                if opening == expected {
6182                    stack.pop();
6183                } else {
6184                    let suffix = if opening_line != line {
6185                        format!(" on line {opening_line}")
6186                    } else {
6187                        String::new()
6188                    };
6189                    return Some(BracketError {
6190                        diagnostic: CpythonDiagnostic::new(
6191                            format!(
6192                                "closing parenthesis '{ch}' does not match opening parenthesis '{opening}'{suffix}"
6193                            ),
6194                            byte_offset,
6195                            byte_offset,
6196                        ),
6197                        unclosed: false,
6198                    });
6199                }
6200            }
6201            _ => {}
6202        }
6203
6204        index += 1;
6205    }
6206
6207    stack.last().map(|(opening, byte_offset, _)| BracketError {
6208        diagnostic: CpythonDiagnostic::new(
6209            format!("'{opening}' was never closed"),
6210            *byte_offset,
6211            *byte_offset,
6212        ),
6213        unclosed: true,
6214    })
6215}
6216
6217fn is_legacy_statement_expression_start(byte: u8) -> bool {
6218    byte >= 0x80
6219        || byte == b'_'
6220        || byte.is_ascii_alphabetic()
6221        || byte.is_ascii_digit()
6222        || matches!(byte, b'\'' | b'"' | b'{' | b'[')
6223}
6224
6225fn legacy_statement_container_has_invalid_attribute(bytes: &[u8], start: usize) -> bool {
6226    let Some(&opening) = bytes.get(start) else {
6227        return false;
6228    };
6229    if !matches!(opening, b'{' | b'[') {
6230        return false;
6231    }
6232
6233    let mut index = start;
6234    let mut level = 0usize;
6235    while index < bytes.len() {
6236        match bytes[index] {
6237            b'#' => {
6238                while index < bytes.len() && bytes[index] != b'\n' {
6239                    index += 1;
6240                }
6241            }
6242            b'\n' | b';' if level == 0 => return false,
6243            b'\'' | b'"' => {
6244                index = skip_quoted_string(bytes, index);
6245            }
6246            b'(' | b'[' | b'{' => {
6247                level += 1;
6248                index += 1;
6249            }
6250            b')' | b']' | b'}' => {
6251                level = level.saturating_sub(1);
6252                index += 1;
6253                if level == 0 {
6254                    return false;
6255                }
6256            }
6257            b'.' => {
6258                let mut cursor = index + 1;
6259                while matches!(bytes.get(cursor), Some(b' ' | b'\t' | b'\x0c')) {
6260                    cursor += 1;
6261                }
6262                if matches!(bytes.get(cursor), Some(b')' | b']' | b'}')) {
6263                    return true;
6264                }
6265                index += 1;
6266            }
6267            _ => index += 1,
6268        }
6269    }
6270    false
6271}
6272
6273fn invalid_legacy_statement_error(source: &str) -> Option<CpythonDiagnostic> {
6274    let bytes = source.as_bytes();
6275    let mut index = 0;
6276    while index < bytes.len() {
6277        match bytes[index] {
6278            b'#' => {
6279                while index < bytes.len() && bytes[index] != b'\n' {
6280                    index += 1;
6281                }
6282            }
6283            b'\'' | b'"' => {
6284                index = skip_quoted_string(bytes, index);
6285            }
6286            b'p' | b'e' => {
6287                let keyword = if starts_identifier(bytes, index, b"print") {
6288                    Some("print")
6289                } else if starts_identifier(bytes, index, b"exec") {
6290                    Some("exec")
6291                } else {
6292                    None
6293                };
6294                let Some(keyword) = keyword else {
6295                    index += 1;
6296                    continue;
6297                };
6298                let after_keyword = index + keyword.len();
6299                if !matches!(bytes.get(after_keyword), Some(b' ' | b'\t' | b'\x0c')) {
6300                    index = after_keyword;
6301                    continue;
6302                }
6303                let mut cursor = after_keyword;
6304                while matches!(bytes.get(cursor), Some(b' ' | b'\t' | b'\x0c')) {
6305                    cursor += 1;
6306                }
6307                if legacy_statement_container_has_invalid_attribute(bytes, cursor) {
6308                    index = after_keyword;
6309                    continue;
6310                }
6311                if bytes.get(cursor).is_some_and(|byte| {
6312                    *byte != b'(' && is_legacy_statement_expression_start(*byte)
6313                }) {
6314                    return Some(CpythonDiagnostic::new(
6315                        format!(
6316                            "Missing parentheses in call to '{keyword}'. Did you mean {keyword}(...)?"
6317                        ),
6318                        index,
6319                        after_keyword,
6320                    ));
6321                }
6322                index = after_keyword;
6323            }
6324            _ => index += 1,
6325        }
6326    }
6327    None
6328}
6329
6330/// Return the syntax error for a decimal integer literal exceeding the configured limit.
6331///
6332/// The parser has already distinguished integer literals from strings, comments, floats, and
6333/// complex numbers. Inspecting its tokens keeps the limit consistent for every source parsing
6334/// entry point without reimplementing Python's lexer here.
6335#[must_use]
6336pub fn long_decimal_integer_literal_error(
6337    source_file: &SourceFile,
6338    tokens: &Tokens,
6339    max_str_digits: usize,
6340) -> Option<CompileError> {
6341    if max_str_digits == 0 {
6342        return None;
6343    }
6344    tokens.iter().find_map(|token| {
6345        if token.kind() != TokenKind::Int {
6346            return None;
6347        }
6348        let literal = source_file.source_text().slice(token.range());
6349        if literal
6350            .as_bytes()
6351            .get(..2)
6352            .is_some_and(|prefix| matches!(prefix, b"0x" | b"0X" | b"0o" | b"0O" | b"0b" | b"0B"))
6353        {
6354            return None;
6355        }
6356        let digits = literal.bytes().filter(u8::is_ascii_digit).count();
6357        (digits > max_str_digits).then(|| {
6358            let start = token.range().start().to_usize();
6359            CompileError::from_source_error(source_file, CpythonDiagnostic::new(format!(
6360                    "Exceeds the limit ({max_str_digits} digits) for integer string conversion: value has {digits} digits; use sys.set_int_max_str_digits() to increase the limit - Consider hexadecimal for huge integer literals to avoid decimal conversion limits."
6361                ), start, start))
6362        })
6363    })
6364}
6365
6366fn invalid_parenthesized_import_star_error(source: &str) -> Option<CpythonDiagnostic> {
6367    let bytes = source.as_bytes();
6368    let mut index = 0;
6369    while index < bytes.len() {
6370        match bytes[index] {
6371            b'#' => {
6372                while index < bytes.len() && bytes[index] != b'\n' {
6373                    index += 1;
6374                }
6375            }
6376            b'\'' | b'"' => {
6377                index = skip_quoted_string(bytes, index);
6378            }
6379            b'f' if starts_identifier(bytes, index, b"from") => {
6380                let mut cursor = index + 4;
6381                while cursor < bytes.len() && !matches!(bytes[cursor], b'\n' | b';') {
6382                    if starts_identifier(bytes, cursor, b"import") {
6383                        cursor += 6;
6384                        while matches!(bytes.get(cursor), Some(b' ' | b'\t' | b'\r')) {
6385                            cursor += 1;
6386                        }
6387                        if bytes.get(cursor) == Some(&b'(') {
6388                            cursor += 1;
6389                            while cursor < bytes.len()
6390                                && !matches!(bytes[cursor], b')' | b'\n' | b';')
6391                            {
6392                                if bytes[cursor] == b'*' {
6393                                    return Some(CpythonDiagnostic::new(
6394                                        "invalid syntax".to_owned(),
6395                                        cursor,
6396                                        cursor + 1,
6397                                    ));
6398                                }
6399                                cursor += 1;
6400                            }
6401                        }
6402                        break;
6403                    }
6404                    cursor += 1;
6405                }
6406                index = cursor;
6407            }
6408            _ => index += 1,
6409        }
6410    }
6411    None
6412}
6413
6414fn prefix_letters_before_quote(bytes: &[u8], quote_index: usize) -> Option<(usize, Vec<u8>)> {
6415    let mut letters = Vec::new();
6416    let mut index = quote_index;
6417    while index > 0 {
6418        let byte = bytes[index - 1];
6419        if !matches!(
6420            byte,
6421            b'r' | b'R' | b'b' | b'B' | b'u' | b'U' | b'f' | b'F' | b't' | b'T'
6422        ) {
6423            break;
6424        }
6425        letters.push(byte.to_ascii_lowercase());
6426        index -= 1;
6427    }
6428    if letters.is_empty() {
6429        return None;
6430    }
6431    if index > 0 && (bytes[index - 1] == b'_' || bytes[index - 1].is_ascii_alphabetic()) {
6432        return None;
6433    }
6434    letters.reverse();
6435    Some((index, letters))
6436}
6437
6438fn incompatible_string_prefix_error(source: &str) -> Option<CpythonDiagnostic> {
6439    let bytes = source.as_bytes();
6440    let mut index = 0;
6441    while index < bytes.len() {
6442        match bytes[index] {
6443            b'#' => {
6444                while index < bytes.len() && bytes[index] != b'\n' {
6445                    index += 1;
6446                }
6447            }
6448            b'\'' | b'"' => {
6449                if let Some((start, letters)) = prefix_letters_before_quote(bytes, index)
6450                    && let Some(message) = incompatible_prefix_message(&letters)
6451                {
6452                    return Some(CpythonDiagnostic::new(message, start, start + 1));
6453                }
6454                index = skip_string_token(bytes, index);
6455            }
6456            _ => index += 1,
6457        }
6458    }
6459    None
6460}
6461
6462fn incompatible_prefix_message(letters: &[u8]) -> Option<String> {
6463    if letters.len() < 2 {
6464        return None;
6465    }
6466    let mut seen_r = false;
6467    let mut seen_b = false;
6468    let mut seen_u = false;
6469    let mut seen_f = false;
6470    let mut seen_t = false;
6471    for &letter in letters {
6472        match letter {
6473            b'r' if seen_r => return None,
6474            b'b' if seen_b => return None,
6475            b'u' if seen_u => return None,
6476            b'f' if seen_f => return None,
6477            b't' if seen_t => return None,
6478            b'r' => seen_r = true,
6479            b'b' => seen_b = true,
6480            b'u' => seen_u = true,
6481            b'f' => seen_f = true,
6482            b't' => seen_t = true,
6483            _ => {}
6484        }
6485    }
6486    let pair = if seen_u && seen_b {
6487        ("u", "b")
6488    } else if seen_u && seen_r {
6489        ("u", "r")
6490    } else if seen_u && seen_f {
6491        ("u", "f")
6492    } else if seen_u && seen_t {
6493        ("u", "t")
6494    } else if seen_b && seen_f {
6495        ("b", "f")
6496    } else if seen_b && seen_t {
6497        ("b", "t")
6498    } else if seen_f && seen_t {
6499        ("f", "t")
6500    } else {
6501        return None;
6502    };
6503    Some(format!(
6504        "'{}' and '{}' prefixes are incompatible",
6505        pair.0, pair.1
6506    ))
6507}
6508
6509fn malformed_unicode_n_escape_error(source: &str) -> Option<CpythonDiagnostic> {
6510    let bytes = source.as_bytes();
6511    let mut index = 0;
6512    while index < bytes.len() {
6513        match bytes[index] {
6514            b'#' => {
6515                while index < bytes.len() && bytes[index] != b'\n' {
6516                    index += 1;
6517                }
6518            }
6519            b'\'' | b'"' => {
6520                let quote_index = index;
6521                let interpolated = interpolated_string_prefix(bytes, quote_index).is_some();
6522                let Some((content_start, content_end)) =
6523                    quoted_string_content_range(bytes, quote_index, bytes[quote_index])
6524                else {
6525                    index = skip_quoted_string(bytes, quote_index);
6526                    continue;
6527                };
6528                if let Some(error) = malformed_unicode_n_in_content(
6529                    bytes,
6530                    content_start,
6531                    content_end,
6532                    quote_index,
6533                    interpolated,
6534                ) {
6535                    return Some(error);
6536                }
6537                index = skip_string_token(bytes, quote_index);
6538            }
6539            _ => index += 1,
6540        }
6541    }
6542    None
6543}
6544
6545fn malformed_unicode_n_in_content(
6546    bytes: &[u8],
6547    start: usize,
6548    end: usize,
6549    quote_index: usize,
6550    interpolated: bool,
6551) -> Option<CpythonDiagnostic> {
6552    let mut index = start;
6553    let mut brace_depth = 0usize;
6554    while index + 1 < end {
6555        if interpolated && brace_depth == 0 && bytes[index] == b'{' {
6556            if bytes.get(index + 1) == Some(&b'{') {
6557                index += 2;
6558                continue;
6559            }
6560            brace_depth = 1;
6561            index += 1;
6562            continue;
6563        }
6564        if interpolated && brace_depth > 0 {
6565            match bytes[index] {
6566                b'\'' | b'"' => index = skip_string_token(bytes, index).max(index + 1),
6567                b'{' => {
6568                    brace_depth += 1;
6569                    index += 1;
6570                }
6571                b'}' => {
6572                    brace_depth -= 1;
6573                    index += 1;
6574                }
6575                _ => index += 1,
6576            }
6577            continue;
6578        }
6579        if bytes[index] == b'\\' && bytes[index + 1] == b'N' {
6580            let escape_start = index;
6581            if bytes.get(index + 2) == Some(&b'{') {
6582                let mut look = index + 3;
6583                while look < end && bytes[look] != b'}' {
6584                    look += 1;
6585                }
6586                if look >= end {
6587                    return Some(unicode_n_diagnostic(escape_start, end, start, quote_index));
6588                }
6589                index = look + 1;
6590                continue;
6591            }
6592            return Some(unicode_n_diagnostic(
6593                escape_start,
6594                escape_start + 2,
6595                start,
6596                quote_index,
6597            ));
6598        }
6599        if interpolated && bytes[index] == b'}' && bytes.get(index + 1) == Some(&b'}') {
6600            index += 2;
6601            continue;
6602        }
6603        index += 1;
6604    }
6605    None
6606}
6607
6608fn unicode_n_diagnostic(
6609    escape_start: usize,
6610    escape_end: usize,
6611    content_start: usize,
6612    quote_index: usize,
6613) -> CpythonDiagnostic {
6614    let start = escape_start - content_start;
6615    let end = escape_end.saturating_sub(content_start).saturating_sub(1);
6616    CpythonDiagnostic::new(
6617        format!(
6618            "(unicode error) 'unicodeescape' codec can't decode bytes in position {start}-{end}: malformed \\N character escape"
6619        ),
6620        quote_index,
6621        quote_index + 1,
6622    )
6623}
6624
6625fn too_many_nested_interpolated_strings(source: &str) -> Option<CpythonDiagnostic> {
6626    too_many_nested_interpolated_strings_in(source.as_bytes(), 0, source.len(), 0)
6627}
6628
6629fn too_many_nested_interpolated_strings_in(
6630    bytes: &[u8],
6631    mut index: usize,
6632    end: usize,
6633    depth: usize,
6634) -> Option<CpythonDiagnostic> {
6635    while index < end {
6636        match bytes[index] {
6637            b'#' => {
6638                while index < end && bytes[index] != b'\n' {
6639                    index += 1;
6640                }
6641            }
6642            b'\'' | b'"' => {
6643                if interpolated_string_prefix(bytes, index).is_some() {
6644                    if depth + 1 >= MAXFSTRINGLEVEL {
6645                        return Some(CpythonDiagnostic::new(
6646                            "too many nested f-strings or t-strings".to_owned(),
6647                            index,
6648                            index + 1,
6649                        ));
6650                    }
6651                    if let Some((content_start, content_end)) =
6652                        interpolated_string_content_range(bytes, index)
6653                        && let Some(error) = too_many_nested_interpolated_strings_in(
6654                            bytes,
6655                            content_start,
6656                            content_end,
6657                            depth + 1,
6658                        )
6659                    {
6660                        return Some(error);
6661                    }
6662                }
6663                index = skip_string_token(bytes, index).max(index + 1);
6664            }
6665            _ => index += 1,
6666        }
6667    }
6668    None
6669}
6670
6671fn too_many_nested_parentheses_error(source: &str) -> Option<CpythonDiagnostic> {
6672    const MAXLEVEL: usize = 200;
6673
6674    let bytes = source.as_bytes();
6675    let mut index = 0;
6676    let mut level = 0usize;
6677    while index < bytes.len() {
6678        match bytes[index] {
6679            b'#' => {
6680                while index < bytes.len() && bytes[index] != b'\n' {
6681                    index += 1;
6682                }
6683            }
6684            b'\'' | b'"' => {
6685                index = skip_quoted_string(bytes, index);
6686            }
6687            b'(' | b'[' | b'{' => {
6688                if level >= MAXLEVEL {
6689                    return Some(CpythonDiagnostic::new(
6690                        "too many nested parentheses".to_owned(),
6691                        index,
6692                        index + 1,
6693                    ));
6694                }
6695                level += 1;
6696                index += 1;
6697            }
6698            b')' | b']' | b'}' => {
6699                level = level.saturating_sub(1);
6700                index += 1;
6701            }
6702            _ => index += 1,
6703        }
6704    }
6705    None
6706}
6707
6708fn invalid_unparenthesized_yield_after_comma_error(source: &str) -> Option<CpythonDiagnostic> {
6709    let bytes = source.as_bytes();
6710    let mut index = 0;
6711    while index < bytes.len() {
6712        match bytes[index] {
6713            b'#' => {
6714                while index < bytes.len() && bytes[index] != b'\n' {
6715                    index += 1;
6716                }
6717            }
6718            b'\'' | b'"' => {
6719                index = skip_quoted_string(bytes, index);
6720            }
6721            b',' => {
6722                let mut cursor = index + 1;
6723                while matches!(bytes.get(cursor), Some(b' ' | b'\t' | b'\x0c')) {
6724                    cursor += 1;
6725                }
6726                if starts_identifier(bytes, cursor, b"yield") {
6727                    return Some(CpythonDiagnostic::new(
6728                        "invalid syntax".to_owned(),
6729                        cursor,
6730                        cursor + 5,
6731                    ));
6732                }
6733                index += 1;
6734            }
6735            _ => index += 1,
6736        }
6737    }
6738    None
6739}
6740
6741/// Horizontal and vertical space the tokenizer skips, not `str::trim()`.
6742const fn is_ascii_tokenizer_whitespace(c: char) -> bool {
6743    matches!(c, ' ' | '\t' | '\n' | '\r' | '\x0c')
6744}
6745
6746/// True when every line is empty or a comment after tokenizer whitespace.
6747#[doc(hidden)]
6748#[must_use]
6749pub fn is_blank_python_source(source: &str) -> bool {
6750    source.lines().all(|line| {
6751        let trimmed = line.trim_matches(is_ascii_tokenizer_whitespace);
6752        trimmed.is_empty() || trimmed.starts_with('#')
6753    })
6754}
6755
6756fn single_mode_blank_source_error(source_file: &SourceFile) -> Option<CompileError> {
6757    let source = source_file.source_text();
6758    if !is_blank_python_source(source) {
6759        return None;
6760    }
6761    let has_indent_only_line = source.lines().any(|line| {
6762        line.trim_matches(is_ascii_tokenizer_whitespace).is_empty()
6763            && line.chars().any(|c| matches!(c, ' ' | '\t'))
6764    });
6765    if has_indent_only_line {
6766        let (location, end_location) =
6767            source_locations(source_file, TextSize::new(0), TextSize::new(0));
6768        return Some(CompileError::Parse(ParseError {
6769            error: parser::ParseErrorType::UnexpectedIndentation,
6770            raw_location: ruff_text_size::TextRange::new(TextSize::new(0), TextSize::new(0)),
6771            location,
6772            end_location,
6773            source_path: source_file.name().to_owned(),
6774            is_unclosed_bracket: false,
6775            is_unclosed_string: false,
6776        }));
6777    }
6778    Some(CompileError::from_source_error(
6779        source_file,
6780        CpythonDiagnostic::new("invalid syntax".to_owned(), 0, 0),
6781    ))
6782}
6783
6784/// The byte-order mark is only stripped while decoding source bytes, so one
6785/// that survives into the text is just a non-printable character. The tokenizer
6786/// rejects it everywhere except at the very start of the text, which is where
6787/// this covers.
6788#[doc(hidden)]
6789#[must_use]
6790pub fn leading_byte_order_mark_error(source_file: &SourceFile) -> Option<CompileError> {
6791    source_file.source_text().starts_with('\u{feff}').then(|| {
6792        CompileError::from_source_error(
6793            source_file,
6794            CpythonDiagnostic::new("invalid non-printable character U+FEFF".to_owned(), 0, 0),
6795        )
6796    })
6797}
6798
6799/// Source-level errors raised before parsing, as CPython's tokenizer does.
6800///
6801/// Checking after the parse would build the tree first, and one nested that
6802/// deep exhausts the native stack when it is dropped.
6803pub fn pre_parse_source_error(source_file: &SourceFile) -> Result<(), CompileError> {
6804    match too_many_nested_parentheses_error(source_file.source_text()) {
6805        Some(error) => Err(CompileError::from_source_error(source_file, error)),
6806        None => Ok(()),
6807    }
6808}
6809
6810fn post_parse_source_error(
6811    source_file: &SourceFile,
6812    tokens: &Tokens,
6813    opts: &CompileOpts,
6814) -> Option<CompileError> {
6815    if let Some(error) = leading_byte_order_mark_error(source_file) {
6816        return Some(error);
6817    }
6818    if let Some(error) = too_many_nested_interpolated_strings(source_file.source_text()) {
6819        return Some(CompileError::from_source_error(source_file, error));
6820    }
6821    if let Some(error) =
6822        long_decimal_integer_literal_error(source_file, tokens, opts.int_max_str_digits)
6823    {
6824        return Some(error);
6825    }
6826    invalid_call_argument_error(source_file.source_text())
6827        .or_else(|| invalid_match_mapping_rest_wildcard_error(source_file.source_text()))
6828        .or_else(|| invalid_match_as_target_error(source_file.source_text()))
6829        .or_else(|| invalid_unparenthesized_yield_after_comma_error(source_file.source_text()))
6830        .or_else(|| invalid_parenthesized_import_star_error(source_file.source_text()))
6831        .map(|error| CompileError::from_source_error(source_file, error))
6832}
6833
6834fn is_compound_stmt(stmt: &ast::Stmt) -> bool {
6835    matches!(
6836        stmt,
6837        ast::Stmt::FunctionDef(_)
6838            | ast::Stmt::ClassDef(_)
6839            | ast::Stmt::If(_)
6840            | ast::Stmt::For(_)
6841            | ast::Stmt::While(_)
6842            | ast::Stmt::With(_)
6843            | ast::Stmt::Try(_)
6844            | ast::Stmt::Match(_)
6845    )
6846}
6847
6848/// Rejects an AST nested deeper than `limit`.
6849///
6850/// The parser grows its own stack, so it returns trees deeper than the passes
6851/// after it can walk, and each of those recurses without a limit of its own.
6852/// The symbol table applies the same limit, but `compile(..., PyCF_ONLY_AST)`
6853/// returns before reaching it.
6854pub fn too_deeply_nested_error(
6855    ast: &ast::Mod,
6856    source_file: &SourceFile,
6857    limit: usize,
6858) -> Result<(), CompileError> {
6859    use ast::visitor::Visitor;
6860
6861    struct DepthChecker {
6862        depth: usize,
6863        limit: usize,
6864        too_deep: Option<ruff_text_size::TextRange>,
6865    }
6866
6867    impl DepthChecker {
6868        fn descend(&mut self, range: ruff_text_size::TextRange, walk: impl FnOnce(&mut Self)) {
6869            if self.too_deep.is_some() {
6870                return;
6871            }
6872            if self.depth >= self.limit {
6873                self.too_deep = Some(range);
6874                return;
6875            }
6876            self.depth += 1;
6877            walk(self);
6878            self.depth -= 1;
6879        }
6880    }
6881
6882    impl<'a> Visitor<'a> for DepthChecker {
6883        fn visit_stmt(&mut self, stmt: &'a ast::Stmt) {
6884            self.descend(stmt.range(), |checker| {
6885                ast::visitor::walk_stmt(checker, stmt);
6886            });
6887        }
6888
6889        fn visit_expr(&mut self, expr: &'a ast::Expr) {
6890            self.descend(expr.range(), |checker| {
6891                ast::visitor::walk_expr(checker, expr);
6892            });
6893        }
6894
6895        fn visit_pattern(&mut self, pattern: &'a ast::Pattern) {
6896            self.descend(pattern.range(), |checker| {
6897                ast::visitor::walk_pattern(checker, pattern);
6898            });
6899        }
6900    }
6901
6902    let mut checker = DepthChecker {
6903        depth: 0,
6904        limit,
6905        too_deep: None,
6906    };
6907    match ast {
6908        ast::Mod::Module(module) => checker.visit_body(&module.body),
6909        ast::Mod::Expression(expression) => checker.visit_expr(&expression.body),
6910    }
6911
6912    let Some(range) = checker.too_deep else {
6913        return Ok(());
6914    };
6915    let (location, end_location) = source_locations(source_file, range.start(), range.end());
6916    Err(CompileError::Codegen(codegen::error::CodegenError {
6917        location: Some(location),
6918        end_location: Some(end_location),
6919        error: codegen::error::CodegenErrorType::RecursionError,
6920        source_path: source_file.name().to_owned(),
6921    }))
6922}
6923
6924/// Syntax the reference grammar has no rule for, but that this parser accepts.
6925///
6926/// A bare generator expression is one: `f(x for x in y)` is the `primary
6927/// genexp` alternative of a call, and a class header only takes `arguments`,
6928/// which has no such alternative. A format spec nested more than two deep is
6929/// the other; the tokenizer runs out of nesting levels for it.
6930#[doc(hidden)]
6931#[must_use]
6932pub fn unsupported_grammar_error(ast: &ast::Mod, source_file: &SourceFile) -> Option<CompileError> {
6933    use ast::visitor::Visitor;
6934
6935    /// The deepest chain of format specs one string literal may hold.
6936    const MAX_FORMAT_SPEC_DEPTH: usize = 2;
6937
6938    struct Checker<'a> {
6939        source_file: &'a SourceFile,
6940        error: Option<CompileError>,
6941    }
6942
6943    impl Checker<'_> {
6944        fn fail(&mut self, message: &str, range: ruff_text_size::TextRange) {
6945            self.error = Some(CompileError::from_source_error(
6946                self.source_file,
6947                CpythonDiagnostic::new(
6948                    message.to_owned(),
6949                    range.start().to_usize(),
6950                    range.end().to_usize(),
6951                ),
6952            ));
6953        }
6954
6955        fn check_format_specs(
6956            &mut self,
6957            kind: &str,
6958            elements: &ast::InterpolatedStringElements,
6959            depth: usize,
6960        ) {
6961            for element in elements.interpolations() {
6962                let Some(format_spec) = &element.format_spec else {
6963                    continue;
6964                };
6965                if depth == MAX_FORMAT_SPEC_DEPTH {
6966                    self.fail(
6967                        &alloc::format!("{kind}: expressions nested too deeply"),
6968                        format_spec.range,
6969                    );
6970                    return;
6971                }
6972                self.check_format_specs(kind, &format_spec.elements, depth + 1);
6973                if self.error.is_some() {
6974                    return;
6975                }
6976            }
6977        }
6978    }
6979
6980    impl<'a> Visitor<'a> for Checker<'_> {
6981        fn visit_stmt(&mut self, stmt: &'a ast::Stmt) {
6982            if self.error.is_some() {
6983                return;
6984            }
6985            if let ast::Stmt::ClassDef(class_def) = stmt
6986                && let Some(arguments) = &class_def.arguments
6987                && let [ast::Expr::Generator(generator)] = &arguments.args[..]
6988                && !generator.parenthesized
6989            {
6990                let range = generator
6991                    .generators
6992                    .first()
6993                    .map_or(generator.range, |comprehension| comprehension.range);
6994                self.fail("invalid syntax", range);
6995                return;
6996            }
6997            ast::visitor::walk_stmt(self, stmt);
6998        }
6999
7000        fn visit_expr(&mut self, expr: &'a ast::Expr) {
7001            if self.error.is_some() {
7002                return;
7003            }
7004            // Each literal counts its own nesting: one written inside a format
7005            // spec starts over.
7006            match expr {
7007                ast::Expr::FString(fstring) => {
7008                    for part in &fstring.value {
7009                        if let ast::FStringPart::FString(part) = part {
7010                            self.check_format_specs("f-string", &part.elements, 0);
7011                        }
7012                    }
7013                }
7014                ast::Expr::TString(tstring) => {
7015                    for part in &tstring.value {
7016                        self.check_format_specs("t-string", &part.elements, 0);
7017                    }
7018                }
7019                _ => {}
7020            }
7021            if self.error.is_some() {
7022                return;
7023            }
7024            ast::visitor::walk_expr(self, expr);
7025        }
7026    }
7027
7028    let mut checker = Checker {
7029        source_file,
7030        error: None,
7031    };
7032    match ast {
7033        ast::Mod::Module(module) => checker.visit_body(&module.body),
7034        ast::Mod::Expression(expression) => checker.visit_expr(&expression.body),
7035    }
7036    checker.error
7037}
7038
7039fn single_mode_body_error(body: &[ast::Stmt], source_file: &SourceFile) -> Option<CompileError> {
7040    let first = body.first()?;
7041    let source_code = source_file.to_source_code();
7042    let first_start = source_code.source_location(first.range().start(), PositionEncoding::Utf8);
7043    let first_end = source_code.source_location(first.range().end(), PositionEncoding::Utf8);
7044
7045    if body.iter().skip(1).any(|stmt| {
7046        source_code
7047            .source_location(stmt.range().start(), PositionEncoding::Utf8)
7048            .line
7049            > first_start.line
7050    }) {
7051        return Some(CompileError::from_source_error(
7052            source_file,
7053            CpythonDiagnostic::new(
7054                "multiple statements found while compiling a single statement".to_owned(),
7055                first.range().end().to_usize(),
7056                first.range().end().to_usize(),
7057            ),
7058        ));
7059    }
7060
7061    if is_compound_stmt(first)
7062        && first_start.line == first_end.line
7063        && !ends_with_line_break(source_file.source_text())
7064    {
7065        return Some(CompileError::from_source_error(
7066            source_file,
7067            CpythonDiagnostic::new(
7068                "invalid syntax".to_owned(),
7069                first.range().start().to_usize(),
7070                first.range().start().to_usize(),
7071            ),
7072        ));
7073    }
7074    None
7075}
7076
7077fn single_mode_source_error(ast: &ast::Mod, source_file: &SourceFile) -> Option<CompileError> {
7078    let ast::Mod::Module(module) = ast else {
7079        return None;
7080    };
7081    single_mode_body_error(&module.body, source_file)
7082}
7083
7084fn ends_with_line_break(source: &str) -> bool {
7085    source.ends_with('\n') || source.ends_with('\r')
7086}
7087
7088fn ends_with_implied_dedent(source: &str) -> bool {
7089    let mut lexer = parser::lexer::lex(source, parser::Mode::Module);
7090    let mut last_kind = TokenKind::EndOfFile;
7091    loop {
7092        let kind = lexer.next_token();
7093        if kind.is_eof() {
7094            break;
7095        }
7096        last_kind = kind;
7097    }
7098    matches!(last_kind, TokenKind::Dedent)
7099}
7100
7101/// Detect input that only parses because Ruff's lexer closes indentation at EOF.
7102///
7103/// `PyCF_DONT_IMPLY_DEDENT` is used by `codeop` and interactive compile
7104/// paths to keep an indented block incomplete until a terminating newline is seen.
7105#[must_use]
7106pub fn dont_imply_dedent_source_error(source_file: &SourceFile) -> Option<CompileError> {
7107    let source = source_file.source_text();
7108    if ends_with_line_break(source) || !ends_with_implied_dedent(source) {
7109        return None;
7110    }
7111    let eof = source.len();
7112    Some(CompileError::from_source_error(
7113        source_file,
7114        CpythonDiagnostic::new("incomplete input".to_owned(), eof, eof),
7115    ))
7116}
7117
7118/// Find the last unclosed opening bracket in source code.
7119/// Returns the bracket character and its byte offset, or None if all brackets are balanced.
7120fn find_unclosed_bracket(source: &str) -> Option<(char, usize)> {
7121    let mut stack: Vec<(char, usize)> = Vec::new();
7122    let mut in_string = false;
7123    let mut string_quote = '\0';
7124    let mut triple_quote = false;
7125    let mut escape_next = false;
7126    let mut is_raw_string = false;
7127
7128    let chars: Vec<(usize, char)> = source.char_indices().collect();
7129    let mut i = 0;
7130
7131    while i < chars.len() {
7132        let (byte_offset, ch) = chars[i];
7133
7134        if escape_next {
7135            escape_next = false;
7136            i += 1;
7137            continue;
7138        }
7139
7140        if in_string {
7141            if ch == '\\' && !is_raw_string {
7142                escape_next = true;
7143            } else if triple_quote {
7144                if ch == string_quote
7145                    && i + 2 < chars.len()
7146                    && chars[i + 1].1 == string_quote
7147                    && chars[i + 2].1 == string_quote
7148                {
7149                    in_string = false;
7150                    i += 3;
7151                    continue;
7152                }
7153            } else if ch == string_quote {
7154                in_string = false;
7155            }
7156            i += 1;
7157            continue;
7158        }
7159
7160        // Check for comments
7161        if ch == '#' {
7162            // Skip to end of line
7163            while i < chars.len() && chars[i].1 != '\n' {
7164                i += 1;
7165            }
7166            continue;
7167        }
7168
7169        // Check for string start (with optional prefix like r, b, f, u, rb, br, etc.)
7170        if ch == '\'' || ch == '"' {
7171            // Check up to 2 characters before the quote for string prefix
7172            is_raw_string = false;
7173            for look_back in 1..=2.min(i) {
7174                let prev = chars[i - look_back].1;
7175                if matches!(prev, 'r' | 'R') {
7176                    is_raw_string = true;
7177                    break;
7178                }
7179                if !matches!(prev, 'b' | 'B' | 'f' | 'F' | 'u' | 'U') {
7180                    break;
7181                }
7182            }
7183            string_quote = ch;
7184            if i + 2 < chars.len() && chars[i + 1].1 == ch && chars[i + 2].1 == ch {
7185                triple_quote = true;
7186                in_string = true;
7187                i += 3;
7188                continue;
7189            }
7190            triple_quote = false;
7191            in_string = true;
7192            i += 1;
7193            continue;
7194        }
7195
7196        match ch {
7197            '(' | '[' | '{' => stack.push((ch, byte_offset)),
7198            ')' | ']' | '}' => {
7199                let expected = match ch {
7200                    ')' => '(',
7201                    ']' => '[',
7202                    '}' => '{',
7203                    _ => unreachable!(),
7204                };
7205                if stack.last().is_some_and(|&(open, _)| open == expected) {
7206                    stack.pop();
7207                }
7208            }
7209            _ => {}
7210        }
7211
7212        i += 1;
7213    }
7214
7215    stack.last().copied()
7216}
7217
7218/// Compile a given source code into a bytecode object.
7219pub fn compile(
7220    source: &str,
7221    mode: Mode,
7222    source_path: &str,
7223    opts: CompileOpts,
7224) -> Result<CodeObject, CompileError> {
7225    // TODO: do this less hacky; ruff's parser should translate a CRLF line
7226    //       break in a multiline string into just an LF in the parsed value
7227    #[cfg(windows)]
7228    let source = source.replace("\r\n", "\n");
7229    #[cfg(windows)]
7230    let source = source.as_str();
7231
7232    let source_file = SourceFileBuilder::new(source_path, source).finish();
7233    _compile(source_file, mode, opts)
7234    // let index = LineIndex::from_source_text(source);
7235    // let source_code = SourceCode::new(source, &index);
7236    // let mut locator = LinearLocator::new(source);
7237    // let mut ast = match parser::parse(source, mode.into(), &source_path) {
7238    //     Ok(x) => x,
7239    //     Err(e) => return Err(locator.locate_error(e)),
7240    // };
7241
7242    // TODO:
7243    // if opts.optimize > 0 {
7244    //     ast = ConstantOptimizer::new()
7245    //         .fold_mod(ast)
7246    //         .unwrap_or_else(|e| match e {});
7247    // }
7248    // let ast = locator.fold_mod(ast).unwrap_or_else(|e| match e {});
7249}
7250
7251fn _compile(
7252    source_file: SourceFile,
7253    mode: Mode,
7254    opts: CompileOpts,
7255) -> Result<CodeObject, CompileError> {
7256    _compile_with_syntax_warning_handler(source_file, mode, opts, None)
7257}
7258
7259fn _compile_with_syntax_warning_handler<'a>(
7260    source_file: SourceFile,
7261    mode: Mode,
7262    opts: CompileOpts,
7263    syntax_warning_handler: Option<&'a mut compile::SyntaxWarningHandler<'a>>,
7264) -> Result<CodeObject, CompileError> {
7265    let parser_mode = match mode {
7266        Mode::Exec => parser::Mode::Module,
7267        Mode::Eval => parser::Mode::Expression,
7268        // ruff does not have an interactive mode, which is fine,
7269        // since these are only different in terms of compilation
7270        Mode::Single | Mode::BlockExpr => parser::Mode::Module,
7271    };
7272    let parser_options = parser::ParseOptions::from(parser_mode);
7273    let barry_source = prepare_barry_as_flufl_source(
7274        source_file.source_text(),
7275        parser_options.clone(),
7276        opts.future_features
7277            .contains(core::bytecode::CodeFlags::FUTURE_BARRY_AS_BDFL),
7278    );
7279    pre_parse_source_error(&source_file)?;
7280    let parsed = parser::parse(barry_source.source(), parser_options);
7281    if let Some(error) = barry_source.diagnostic(parsed.as_ref().err(), &source_file) {
7282        return Err(error);
7283    }
7284    let parsed =
7285        parsed.map_err(|err| CompileError::from_ruff_parse_error(err, &source_file, mode))?;
7286    if matches!(mode, Mode::Single)
7287        && let Some(error) = single_mode_blank_source_error(&source_file)
7288    {
7289        return Err(error);
7290    }
7291    if opts.dont_imply_dedent
7292        && matches!(mode, Mode::Single)
7293        && let Some(error) = dont_imply_dedent_source_error(&source_file)
7294    {
7295        return Err(error);
7296    }
7297    if let Some(error) = post_parse_source_error(&source_file, parsed.tokens(), &opts) {
7298        return Err(error);
7299    }
7300    let ast = parsed.into_syntax();
7301    too_deeply_nested_error(&ast, &source_file, opts.recursion_limit)?;
7302    if let Some(error) = unsupported_grammar_error(&ast, &source_file) {
7303        return Err(error);
7304    }
7305    let single_mode_error = matches!(mode, Mode::Single)
7306        .then(|| single_mode_source_error(&ast, &source_file))
7307        .flatten();
7308    let code = compile::compile_top_with_syntax_warning_handler(
7309        ast,
7310        source_file,
7311        mode,
7312        opts,
7313        syntax_warning_handler,
7314    )
7315    .map_err(CompileError::from)?;
7316    if let Some(error) = single_mode_error {
7317        return Err(error);
7318    }
7319    Ok(code)
7320}
7321
7322#[doc(hidden)]
7323pub struct BarrySource<'a> {
7324    source: Cow<'a, str>,
7325    not_equal: Option<ruff_text_size::TextRange>,
7326    legacy_not_equal: Vec<ruff_text_size::TextRange>,
7327}
7328
7329impl BarrySource<'_> {
7330    #[must_use]
7331    pub fn source(&self) -> &str {
7332        &self.source
7333    }
7334
7335    #[must_use]
7336    pub fn not_equal_before(
7337        &self,
7338        parse_error: Option<&parser::ParseError>,
7339    ) -> Option<ruff_text_size::TextRange> {
7340        self.not_equal.filter(|range| {
7341            parse_error.is_none_or(|error| {
7342                let diagnostic_start = if matches!(
7343                    &error.error,
7344                    parser::ParseErrorType::Lexical(parser::LexicalErrorType::Eof)
7345                ) {
7346                    find_unclosed_bracket(&self.source).map_or_else(
7347                        || error.location.start(),
7348                        |(_, offset)| TextSize::new(offset as u32),
7349                    )
7350                } else {
7351                    error.location.start()
7352                };
7353                range.start() <= diagnostic_start
7354            })
7355        })
7356    }
7357
7358    /// The obsolete `<>` operator the parse error points at, if any. In Barry
7359    /// mode the operator was rewritten to `!=`, so the error lands on its
7360    /// start; outside Barry mode ruff lexes `<` and then an unexpected `>`, so
7361    /// the error lands one character in. Either way the whole operator is one
7362    /// token to the tokenizer, so report it as one.
7363    #[must_use]
7364    pub fn invalid_legacy_operator(
7365        &self,
7366        parse_error: &parser::ParseError,
7367    ) -> Option<ruff_text_size::TextRange> {
7368        let location = parse_error.location.start();
7369        self.legacy_not_equal
7370            .iter()
7371            .copied()
7372            .find(|range| range.contains(location) || range.start() == location)
7373            .filter(|range| self.outranks_unclosed_bracket(*range))
7374    }
7375
7376    /// Whether `range` outranks an unclosed bracket. The bracket is reported
7377    /// at itself, so it wins over anything that starts after it.
7378    fn outranks_unclosed_bracket(&self, range: ruff_text_size::TextRange) -> bool {
7379        find_unclosed_bracket(&self.source)
7380            .is_none_or(|(_, offset)| range.start() <= TextSize::new(offset as u32))
7381    }
7382
7383    /// The diagnostic for this source, if any: an obsolete `<>` the parse
7384    /// error points at, or -- in Barry mode only -- the first `!=`. A `<>`
7385    /// takes precedence over a `!=` reported later in the source.
7386    #[must_use]
7387    pub fn diagnostic(
7388        &self,
7389        parse_error: Option<&parser::ParseError>,
7390        source_file: &SourceFile,
7391    ) -> Option<CompileError> {
7392        if let Some(range) = parse_error.and_then(|error| self.invalid_legacy_operator(error)) {
7393            return Some(barry_as_flufl_invalid_legacy_operator_error(
7394                source_file,
7395                range,
7396            ));
7397        }
7398        self.not_equal_before(parse_error)
7399            .map(|range| barry_as_flufl_not_equal_error(source_file, range))
7400    }
7401}
7402
7403#[doc(hidden)]
7404#[must_use]
7405pub fn barry_as_flufl_not_equal_error(
7406    source_file: &SourceFile,
7407    range: ruff_text_size::TextRange,
7408) -> CompileError {
7409    CompileError::from_source_error(
7410        source_file,
7411        CpythonDiagnostic::new(
7412            "with Barry as BDFL, use '<>' instead of '!='".to_owned(),
7413            range.start().to_usize(),
7414            range.end().to_usize(),
7415        ),
7416    )
7417}
7418
7419#[doc(hidden)]
7420#[must_use]
7421pub fn barry_as_flufl_invalid_legacy_operator_error(
7422    source_file: &SourceFile,
7423    range: ruff_text_size::TextRange,
7424) -> CompileError {
7425    CompileError::from_source_error(
7426        source_file,
7427        CpythonDiagnostic::new(
7428            "invalid syntax".to_owned(),
7429            range.start().to_usize(),
7430            range.end().to_usize(),
7431        ),
7432    )
7433}
7434
7435/// Every `<>` in `source`, located by plain text search. Only used where the
7436/// operator is not rewritten, so an occurrence inside a string or a comment
7437/// costs nothing: it can never coincide with the location of a parse error.
7438fn textual_legacy_not_equal(source: &str) -> Vec<ruff_text_size::TextRange> {
7439    source
7440        .match_indices("<>")
7441        .map(|(offset, matched)| {
7442            ruff_text_size::TextRange::at(
7443                TextSize::new(offset as u32),
7444                TextSize::new(matched.len() as u32),
7445            )
7446        })
7447        .collect()
7448}
7449
7450#[doc(hidden)]
7451pub fn prepare_barry_as_flufl_source(
7452    source: &str,
7453    parser_options: parser::ParseOptions,
7454    inherited: bool,
7455) -> BarrySource<'_> {
7456    let scanned = (inherited || source.contains("barry_as_FLUFL"))
7457        .then(|| parser::parse_unchecked(source, parser_options));
7458    let enabled = scanned.as_ref().is_some_and(|scanned| {
7459        inherited
7460            || codegen::preprocess::future_features(scanned.syntax())
7461                .contains(core::bytecode::CodeFlags::FUTURE_BARRY_AS_BDFL)
7462    });
7463    let Some(scanned) = scanned.filter(|_| enabled) else {
7464        return BarrySource {
7465            source: Cow::Borrowed(source),
7466            not_equal: None,
7467            legacy_not_equal: textual_legacy_not_equal(source),
7468        };
7469    };
7470
7471    let not_equal = scanned
7472        .tokens()
7473        .iter()
7474        .find(|token| token.kind() == TokenKind::NotEqual)
7475        .map(Ranged::range);
7476    let replacements = scanned
7477        .tokens()
7478        .windows(2)
7479        .filter_map(|tokens| {
7480            let [less, greater] = tokens else {
7481                return None;
7482            };
7483            (less.kind() == TokenKind::Less
7484                && greater.kind() == TokenKind::Greater
7485                && less.end() == greater.start())
7486            .then(|| ruff_text_size::TextRange::new(less.start(), greater.end()))
7487        })
7488        .collect::<Vec<_>>();
7489
7490    let source = if replacements.is_empty() {
7491        Cow::Borrowed(source)
7492    } else {
7493        let mut rewritten = source.to_owned();
7494        for range in replacements.iter().rev() {
7495            rewritten.replace_range(range.start().to_usize()..range.end().to_usize(), "!=");
7496        }
7497        Cow::Owned(rewritten)
7498    };
7499    BarrySource {
7500        source,
7501        not_equal,
7502        legacy_not_equal: replacements,
7503    }
7504}
7505
7506pub fn compile_with_syntax_warning_handler<'a>(
7507    source: &str,
7508    mode: Mode,
7509    source_path: &str,
7510    opts: CompileOpts,
7511    syntax_warning_handler: &'a mut compile::SyntaxWarningHandler<'a>,
7512) -> Result<CodeObject, CompileError> {
7513    let source = source.replace("\r\n", "\n");
7514    #[cfg(windows)]
7515    let source = source.as_str();
7516
7517    let source_file = SourceFileBuilder::new(source_path, source).finish();
7518    _compile_with_syntax_warning_handler(source_file, mode, opts, Some(syntax_warning_handler))
7519}
7520
7521pub fn compile_symtable(
7522    source: &str,
7523    mode: Mode,
7524    source_path: &str,
7525) -> Result<symboltable::SymbolTable, CompileError> {
7526    let source_file = SourceFileBuilder::new(source_path, source).finish();
7527    _compile_symtable(source_file, mode)
7528}
7529
7530/// Bring a module into the shape the symbol table is built from.
7531fn symtable_preprocess_module(
7532    module: &mut ast::ModModule,
7533    source_file: &SourceFile,
7534) -> Result<(), CompileError> {
7535    let future_features = codegen::preprocess::checked_future_features_in_body(&module.body)
7536        .map_err(|error| future_feature_error(error, source_file))?;
7537    let future_annotations =
7538        future_features.contains(core::bytecode::CodeFlags::FUTURE_ANNOTATIONS);
7539    // Constant folding is left out; the symbol table is built from the parsed
7540    // program as written.
7541    codegen::preprocess::preprocess_statements(&mut module.body, 0, future_annotations, true);
7542    Ok(())
7543}
7544
7545fn future_feature_error(
7546    error: codegen::preprocess::FutureFeatureError,
7547    source_file: &SourceFile,
7548) -> CompileError {
7549    let source_code = source_file.to_source_code();
7550    let location = source_code.source_location(error.range.start(), PositionEncoding::Utf8);
7551    let end_location = source_code.source_location(error.range.end(), PositionEncoding::Utf8);
7552    let error = match error.kind {
7553        codegen::preprocess::FutureFeatureErrorKind::InvalidFeature(feature) => {
7554            codegen::error::CodegenErrorType::InvalidFutureFeature(feature)
7555        }
7556        codegen::preprocess::FutureFeatureErrorKind::InvalidBraces => {
7557            codegen::error::CodegenErrorType::InvalidFutureBraces
7558        }
7559    };
7560    codegen::error::CodegenError {
7561        location: Some(location),
7562        end_location: Some(end_location),
7563        error,
7564        source_path: source_file.name().to_owned(),
7565    }
7566    .into()
7567}
7568
7569pub fn _compile_symtable(
7570    source_file: SourceFile,
7571    mode: Mode,
7572) -> Result<symboltable::SymbolTable, CompileError> {
7573    let parser_mode = match mode {
7574        Mode::Exec | Mode::Single | Mode::BlockExpr => parser::Mode::Module,
7575        Mode::Eval => parser::Mode::Expression,
7576    };
7577    let parser_options = parser::ParseOptions::from(parser_mode);
7578    let barry_source =
7579        prepare_barry_as_flufl_source(source_file.source_text(), parser_options.clone(), false);
7580    let res = match mode {
7581        Mode::Exec | Mode::Single | Mode::BlockExpr => {
7582            pre_parse_source_error(&source_file)?;
7583            let parsed = ruff_python_parser::parse(barry_source.source(), parser_options);
7584            if let Some(error) = barry_source.diagnostic(parsed.as_ref().err(), &source_file) {
7585                return Err(error);
7586            }
7587            let ast =
7588                parsed.map_err(|e| CompileError::from_ruff_parse_error(e, &source_file, mode))?;
7589            if let Some(error) =
7590                post_parse_source_error(&source_file, ast.tokens(), &CompileOpts::default())
7591            {
7592                return Err(error);
7593            }
7594            let ast = ast.into_syntax();
7595            too_deeply_nested_error(&ast, &source_file, CompileOpts::default().recursion_limit)?;
7596            if let Some(error) = unsupported_grammar_error(&ast, &source_file) {
7597                return Err(error);
7598            }
7599            let ast = ast.expect_module();
7600            if matches!(mode, Mode::Single)
7601                && let Some(error) = single_mode_body_error(&ast.body, &source_file)
7602            {
7603                return Err(error);
7604            }
7605            let mut ast = ast;
7606            symtable_preprocess_module(&mut ast, &source_file)?;
7607            symboltable::SymbolTable::scan_program(&ast, source_file.clone())
7608        }
7609        Mode::Eval => {
7610            pre_parse_source_error(&source_file)?;
7611            let parsed = ruff_python_parser::parse(barry_source.source(), parser_options);
7612            if let Some(error) = barry_source.diagnostic(parsed.as_ref().err(), &source_file) {
7613                return Err(error);
7614            }
7615            let ast =
7616                parsed.map_err(|e| CompileError::from_ruff_parse_error(e, &source_file, mode))?;
7617            if let Some(error) =
7618                post_parse_source_error(&source_file, ast.tokens(), &CompileOpts::default())
7619            {
7620                return Err(error);
7621            }
7622            let ast = ast.into_syntax();
7623            too_deeply_nested_error(&ast, &source_file, CompileOpts::default().recursion_limit)?;
7624            if let Some(error) = unsupported_grammar_error(&ast, &source_file) {
7625                return Err(error);
7626            }
7627            let mut ast = ast;
7628            codegen::preprocess::preprocess_mod(&mut ast, 0, false, true);
7629            symboltable::SymbolTable::scan_expr(&ast.expect_expression(), source_file.clone())
7630        }
7631    };
7632    res.map_err(|e| e.into_codegen_error(source_file.name().to_owned()).into())
7633}
7634
7635#[cfg(test)]
7636mod tests {
7637    use super::*;
7638
7639    #[test]
7640    fn basic_compile() {
7641        let code = "x = 'abc'";
7642        let compiled = compile(code, Mode::Single, "<>", CompileOpts::default());
7643        dbg!(compiled.expect("compile error"));
7644    }
7645
7646    #[test]
7647    fn empty_parameter_default_is_reported_before_later_params() {
7648        let err = compile(
7649            "def foo(a=1,d=,c):\n    pass\n",
7650            Mode::Exec,
7651            "<params>",
7652            CompileOpts::default(),
7653        )
7654        .expect_err("empty default should fail");
7655        assert!(
7656            err.to_string()
7657                .contains("expected default value expression"),
7658            "got {err}"
7659        );
7660    }
7661
7662    #[test]
7663    fn too_many_nested_fstrings_match_tokenizer_limit() {
7664        fn nested(n: usize) -> String {
7665            if n == 0 {
7666                return "1+1".to_owned();
7667            }
7668            format!("f\"{{{}}}\"", nested(n - 1))
7669        }
7670
7671        compile(&nested(149), Mode::Eval, "<nest>", CompileOpts::default())
7672            .expect("149 nested f-strings should compile");
7673        let err = compile(&nested(150), Mode::Eval, "<nest>", CompileOpts::default())
7674            .expect_err("150 nested f-strings should fail");
7675        assert!(
7676            err.to_string()
7677                .contains("too many nested f-strings or t-strings"),
7678            "got {err}"
7679        );
7680    }
7681
7682    #[test]
7683    fn interpolated_string_diagnostics_match_cpython() {
7684        for (source, expected) in [
7685            ("f'{'", "f-string: expecting '}'"),
7686            ("t'{'", "t-string: expecting '}'"),
7687            (
7688                "f'{1=}{;'",
7689                "f-string: expecting a valid expression after '{'",
7690            ),
7691            (
7692                "t'{x;y}'",
7693                "t-string: expecting '=', or '!', or ':', or '}'",
7694            ),
7695            ("t'{x!s:'", "t-string: expecting '}', or format specs"),
7696            ("t'{x=!}'", "t-string: missing conversion character"),
7697            (
7698                "t'{x:{;}}'",
7699                "t-string: expecting a valid expression after '{'",
7700            ),
7701            ("f'{1#}'", "'{' was never closed"),
7702            ("t'{", "'{' was never closed"),
7703            // A field is only "never closed" while the tokenizer is reading its expression.
7704            // Past the field's own `:` it emits literal text again, so the quote is what is
7705            // missing — and a `:` in a slice, a display or a lambda is not the field's own.
7706            ("f'{a", "'{' was never closed"),
7707            ("f'{a!r", "'{' was never closed"),
7708            ("f'{a=", "'{' was never closed"),
7709            ("f'{ {1:2}", "'{' was never closed"),
7710            ("f'{d[1:2]", "'{' was never closed"),
7711            ("f'{(lambda x: x)", "'{' was never closed"),
7712            ("f'{a:{b:{c", "'{' was never closed"),
7713            ("f'{a:", "unterminated f-string literal"),
7714            ("f'{a:>5", "unterminated f-string literal"),
7715            ("f'{a!r:", "unterminated f-string literal"),
7716            ("f'{a}{b:", "unterminated f-string literal"),
7717            ("f'{a:{b}c", "unterminated f-string literal"),
7718            ("t'{a:>5", "unterminated t-string literal"),
7719            ("f'''{a:>5", "unterminated triple-quoted f-string literal"),
7720            // CPython names the innermost unclosed delimiter, not the field around it.
7721            ("f'{a[", "'[' was never closed"),
7722            ("f'{(a", "'(' was never closed"),
7723            ("f'{)#}'", "f-string: unmatched ')'"),
7724            (
7725                "f'{a[4)}'",
7726                "closing parenthesis ')' does not match opening parenthesis '['",
7727            ),
7728            ("t'", "unterminated t-string literal (detected at line 1)"),
7729            (
7730                "t'''",
7731                "unterminated triple-quoted t-string literal (detected at line 1)",
7732            ),
7733            (
7734                "t\"x\" b\"y\"",
7735                "Cannot mix t-string literals with string or bytes literals",
7736            ),
7737            (
7738                "b\"x\" t\"y\"",
7739                "Cannot mix t-string literals with string or bytes literals",
7740            ),
7741            // The literals ahead of the first t-string already mix, so CPython's first pass
7742            // raises from `_PyPegen_concatenate_strings` before the t-string rule is reached.
7743            (
7744                "\"a\" b\"b\" t\"c\"",
7745                "cannot mix bytes and nonbytes literals",
7746            ),
7747            // A dangling operator ends the expression before a separator could follow.
7748            (
7749                "f'{a==}'",
7750                "f-string: expecting '=', or '!', or ':', or '}'",
7751            ),
7752            (
7753                "f'{a and}'",
7754                "f-string: expecting '=', or '!', or ':', or '}'",
7755            ),
7756            (
7757                "f'{a is not}'",
7758                "f-string: expecting '=', or '!', or ':', or '}'",
7759            ),
7760            (
7761                "f'{a.b.}'",
7762                "f-string: expecting '=', or '!', or ':', or '}'",
7763            ),
7764            ("t'{a~}'", "t-string: expecting '=', or '!', or ':', or '}'"),
7765            // A doubled `=` or `!` opens no debug specifier and no conversion, so the field has
7766            // no expression at all rather than an empty one before a marker.
7767            (
7768                "f'{==a}'",
7769                "f-string: expecting a valid expression after '{'",
7770            ),
7771            (
7772                "f'{!=a}'",
7773                "f-string: expecting a valid expression after '{'",
7774            ),
7775            (
7776                "t'{==a}'",
7777                "t-string: expecting a valid expression after '{'",
7778            ),
7779            ("f'{=a}'", "f-string: valid expression required before '='"),
7780            ("f'{!}'", "f-string: valid expression required before '!'"),
7781            (
7782                "f'{lambda x:x}'",
7783                "f-string: lambda expressions are not allowed without parentheses",
7784            ),
7785            (
7786                "f'{1, lambda:x}'",
7787                "f-string: lambda expressions are not allowed without parentheses",
7788            ),
7789            (
7790                "f'{+ lambda:None}'",
7791                "f-string: expecting a valid expression after '{'",
7792            ),
7793            ("fu''", "'u' and 'f' prefixes are incompatible"),
7794            ("fb''", "'b' and 'f' prefixes are incompatible"),
7795            ("ufr''", "'u' and 'r' prefixes are incompatible"),
7796            (
7797                "(]\nbu'x'",
7798                "closing parenthesis ']' does not match opening parenthesis '('",
7799            ),
7800            ("0x\nbu'x'", "invalid hexadecimal literal"),
7801            ("(0x", "invalid hexadecimal literal"),
7802            ("print x; 0x", "invalid hexadecimal literal"),
7803            ("exec x; 0x", "invalid hexadecimal literal"),
7804            (
7805                "print x; 0x1",
7806                "Missing parentheses in call to 'print'. Did you mean print(...)?",
7807            ),
7808            (
7809                "print x; (",
7810                "Missing parentheses in call to 'print'. Did you mean print(...)?",
7811            ),
7812            ("print x; )", "unmatched ')'"),
7813            (
7814                "( '\\N'",
7815                "(unicode error) 'unicodeescape' codec can't decode bytes in position 0-1: malformed \\N character escape",
7816            ),
7817            ("(print x", "'(' was never closed"),
7818            (
7819                "(print x; '\\N'",
7820                "Missing parentheses in call to 'print'. Did you mean print(...)?",
7821            ),
7822            ("f'{x'; '", "f-string: expecting '}'"),
7823            ("f'{x'; 0x", "f-string: expecting '}'"),
7824            (
7825                "print x; f'{x'; 0x",
7826                "Missing parentheses in call to 'print'. Did you mean print(...)?",
7827            ),
7828            (
7829                concat!("f'{1:", "d\n}'"),
7830                "f-string: newlines are not allowed in format specifiers",
7831            ),
7832            ("f'{\n}'", "f-string: valid expression required before '}'"),
7833            (
7834                "f'''\n{\n# only a comment\n}'''",
7835                "f-string: valid expression required before '}'",
7836            ),
7837            ("{\\'a\\'}", "unexpected character after line continuation"),
7838            ("\"\\\n\"(1 for c in I,\\\n\\", "'(' was never closed"),
7839            (
7840                r"'\N'",
7841                "(unicode error) 'unicodeescape' codec can't decode bytes in position 0-1: malformed \\N character escape",
7842            ),
7843            (
7844                r"f'\N{'",
7845                "(unicode error) 'unicodeescape' codec can't decode bytes in position 0-2: malformed \\N character escape",
7846            ),
7847        ] {
7848            let err = compile(source, Mode::Eval, "<interp>", CompileOpts::default())
7849                .expect_err("should not compile");
7850            assert!(
7851                err.to_string().contains(expected),
7852                "{source:?}: expected {expected:?}, got {err}"
7853            );
7854        }
7855    }
7856
7857    #[test]
7858    fn missing_indent_outranks_print_missing_parentheses() {
7859        for (source, expected) in [
7860            (
7861                "if True:\nprint \"No indent\"",
7862                "expected an indented block after 'if' statement on line 1",
7863            ),
7864            (
7865                "print \"old style\"",
7866                "Missing parentheses in call to 'print'. Did you mean print(...)?",
7867            ),
7868        ] {
7869            let err = compile(source, Mode::Exec, "<fragment>", CompileOpts::default())
7870                .expect_err("should not compile");
7871            assert!(
7872                err.to_string().contains(expected),
7873                "{source:?}: expected {expected:?}, got {err}"
7874            );
7875        }
7876    }
7877
7878    #[test]
7879    fn unclosed_fstring_field_keeps_the_unclosed_bracket_flag() {
7880        let err = compile("f'{", Mode::Eval, "<interp>", CompileOpts::default())
7881            .expect_err("should not compile");
7882        let crate::CompileError::Parse(parse) = err else {
7883            panic!("expected a parse error, got {err}");
7884        };
7885        assert!(
7886            parse.is_unclosed_bracket,
7887            "unclosed f-string field must stay incomplete, got {parse}"
7888        );
7889        assert!(
7890            parse.to_string().contains("'{' was never closed"),
7891            "got {parse}"
7892        );
7893    }
7894
7895    #[test]
7896    fn interpolated_literals_do_not_take_the_escaped_quote_hint() {
7897        // Parser/lexer/lexer.c offers "perhaps you escaped the end quote?" from its plain-string
7898        // branch only; the interpolated branch has just the triple-quoted and plain forms. The
7899        // assertion has to be exact: the hint is a suffix, so `contains` would pass either way.
7900        for (source, expected) in [
7901            (
7902                r"f'\'",
7903                "unterminated f-string literal (detected at line 1)",
7904            ),
7905            (
7906                r"t'\'",
7907                "unterminated t-string literal (detected at line 1)",
7908            ),
7909        ] {
7910            let err = compile(source, Mode::Eval, "<escaped>", CompileOpts::default())
7911                .expect_err("should not compile");
7912            assert_eq!(err.to_string(), expected, "{source:?}");
7913        }
7914        // A literal without a prefix still gets it.
7915        let err = compile(r"'\'", Mode::Eval, "<escaped>", CompileOpts::default())
7916            .expect_err("should not compile");
7917        assert_eq!(
7918            err.to_string(),
7919            "unterminated string literal (detected at line 1); perhaps you escaped the end quote?"
7920        );
7921    }
7922
7923    #[test]
7924    fn a_field_bracket_mismatch_names_the_opening_line() {
7925        // CPython compares `parenlinenostack[level]` against the current `lineno`, so it names
7926        // the opening line only when the two brackets are not on the same one.
7927        for (source, expected) in [
7928            (
7929                "x = f\"\"\"{a[\n4)}\"\"\"\n",
7930                "closing parenthesis ')' does not match opening parenthesis '[' on line 1",
7931            ),
7932            (
7933                "x = f\"\"\"{a(\n\n4]}\"\"\"\n",
7934                "closing parenthesis ']' does not match opening parenthesis '(' on line 1",
7935            ),
7936            (
7937                "x = f\"{a[4)}\"\n",
7938                "closing parenthesis ')' does not match opening parenthesis '['",
7939            ),
7940        ] {
7941            let err = compile(source, Mode::Exec, "<paren>", CompileOpts::default())
7942                .expect_err("should not compile");
7943            assert_eq!(err.to_string(), expected, "{source:?}");
7944        }
7945    }
7946
7947    #[test]
7948    fn a_dangling_operator_needs_the_field_to_close() {
7949        // A stray character is a finished token, so CPython's grammar rejects it on lookahead
7950        // and names the separators even when the field never closes. A dangling operator makes
7951        // the parser ask for one more token, and producing it hits the closing quote, where
7952        // lexer.c answers "expecting '}'" from the tokenizer instead.
7953        for (source, expected) in [
7954            ("f'{a;'", "f-string: expecting '=', or '!', or ':', or '}'"),
7955            ("f'{a$'", "f-string: expecting '=', or '!', or ':', or '}'"),
7956            ("f'{a?'", "f-string: expecting '=', or '!', or ':', or '}'"),
7957            ("f'{a and'", "f-string: expecting '}'"),
7958            ("f'{a+'", "f-string: expecting '}'"),
7959            ("f'{a=='", "f-string: expecting '}'"),
7960            ("f'{a.b.'", "f-string: expecting '}'"),
7961            ("f'{a is not'", "f-string: expecting '}'"),
7962            ("t'{a and'", "t-string: expecting '}'"),
7963            // With the field closed, the operator is what gets named.
7964            (
7965                "f'{a and}'",
7966                "f-string: expecting '=', or '!', or ':', or '}'",
7967            ),
7968            (
7969                "f'{a==}'",
7970                "f-string: expecting '=', or '!', or ':', or '}'",
7971            ),
7972        ] {
7973            let err = compile(source, Mode::Eval, "<dangling>", CompileOpts::default())
7974                .expect_err("should not compile");
7975            assert!(
7976                err.to_string().contains(expected),
7977                "{source:?}: expected {expected:?}, got {err}"
7978            );
7979        }
7980    }
7981
7982    #[test]
7983    fn deeply_nested_format_specs_stay_linear() {
7984        // Each enclosing spec used to re-scan every field nested inside it, which made this
7985        // O(2^depth): a 110-byte source took six seconds. The scan now resumes past a nested
7986        // field once it has checked it, so this has to finish immediately.
7987        for depth in [32usize, 200] {
7988            let source = format!("x = f\"{}{}\"\n$", "{1:".repeat(depth), "}".repeat(depth));
7989            compile(&source, Mode::Exec, "<nested>", CompileOpts::default())
7990                .expect_err("the trailing `$` is a syntax error");
7991        }
7992    }
7993
7994    #[test]
7995    #[expect(
7996        clippy::literal_string_with_formatting_args,
7997        reason = "these are Python format specs, not Rust format args"
7998    )]
7999    fn valid_interpolated_literals_do_not_shadow_a_later_syntax_error() {
8000        // A format spec is text rather than code, and a `#` in a replacement field starts a
8001        // comment. Mistaking either for expression syntax would blame a perfectly good literal
8002        // for the `$` further down, so the reported line must stay on the `$`.
8003        for source in [
8004            "x = f\"{1:(}\"\n$",
8005            "x = f\"{1:[}\"\n$",
8006            "x = f\"{1:#x}\"\n$",
8007            "x = f\"{1!r:#>5}\"\n$",
8008            "x = t\"{1:(}\"\n$",
8009            "x = f\"\"\"{1 # (\n}\"\"\"\n$",
8010            "x = f\"\"\"{1 # {\n}\"\"\"\n$",
8011            "x = f\"\"\"{1 # ]\n}\"\"\"\n$",
8012            "x = f\"\"\"{1 # a\\ b\n}\"\"\"\n$",
8013            // A comment tail is not the end of the expression, whatever it looks like.
8014            "x = f\"\"\"{a # +\n}\"\"\"\n$",
8015            "x = f\"\"\"{a # and\n}\"\"\"\n$",
8016            "x = f\"\"\"{a # ==\n}\"\"\"\n$",
8017            "x = t\"\"\"{a # .\n}\"\"\"\n$",
8018            "x = f\"\"\"{1:{a # +\n}}\"\"\"\n$",
8019            "x = f\"\"\"{a # +\n + b}\"\"\"\n$",
8020            // A signed leading-dot float is an operand.
8021            "x = f\"{-.5}\"\n$",
8022            "x = f\"{+.5}\"\n$",
8023            // Ellipsis and a leading-dot float start real expressions.
8024            "x = f\"{...}\"\n$",
8025            "x = f\"{.5}\"\n$",
8026            "x = t\"{...}\"\n$",
8027            // A dangling operator is reported, but a complete one is not.
8028            "x = f\"{a+b}\"\n$",
8029            "x = f\"{a.b.c}\"\n$",
8030            "x = f\"{1.}\"\n$",
8031            // A comparison operator is not a debug or conversion marker.
8032            "x = f\"{a==b}\"\n$",
8033            "x = f\"{a!=b}\"\n$",
8034            "x = f\"{a<=b}\"\n$",
8035            "x = f\"{a>=b}\"\n$",
8036            "x = t\"{a==b}\"\n$",
8037            "x = f\"{a:{b==c}}\"\n$",
8038        ] {
8039            let err = compile(source, Mode::Exec, "<interp>", CompileOpts::default())
8040                .expect_err("the trailing `$` is a syntax error");
8041            assert_eq!(
8042                err.python_location().0,
8043                source.lines().count(),
8044                "{source:?} reported the wrong line: {err}"
8045            );
8046        }
8047    }
8048
8049    #[test]
8050    fn dont_imply_dedent_requires_terminating_newline() {
8051        let code = "if True:\n    pass";
8052
8053        let opts = CompileOpts {
8054            dont_imply_dedent: true,
8055            ..CompileOpts::default()
8056        };
8057        let err = compile(code, Mode::Single, "<>", opts.clone()).expect_err("compile succeeded");
8058        assert_eq!(err.to_string(), "incomplete input");
8059
8060        compile("if True:\n    pass\n", Mode::Single, "<>", opts).expect("compile error");
8061        compile(code, Mode::Single, "<>", CompileOpts::default()).expect("compile error");
8062    }
8063
8064    #[test]
8065    fn barry_as_flufl_rewrites_legacy_not_equal_after_future_import() {
8066        let code = compile(
8067            "from __future__ import barry_as_FLUFL\nresult = 2 <> 3\n",
8068            Mode::Exec,
8069            "<barry>",
8070            CompileOpts::default(),
8071        )
8072        .expect("Barry comparison should compile");
8073        assert!(
8074            code.flags
8075                .contains(core::bytecode::CodeFlags::FUTURE_BARRY_AS_BDFL)
8076        );
8077    }
8078
8079    #[test]
8080    fn inherited_barry_as_flufl_rewrites_legacy_not_equal() {
8081        let opts = CompileOpts {
8082            future_features: core::bytecode::CodeFlags::FUTURE_BARRY_AS_BDFL,
8083            ..CompileOpts::default()
8084        };
8085        compile("2 <> 3", Mode::Single, "<barry>", opts)
8086            .expect("inherited Barry comparison should compile");
8087    }
8088
8089    #[test]
8090    fn barry_as_flufl_rejects_modern_not_equal() {
8091        let err = compile(
8092            "from __future__ import barry_as_FLUFL\n2 != 3\n",
8093            Mode::Exec,
8094            "<barry>",
8095            CompileOpts::default(),
8096        )
8097        .expect_err("Barry mode should reject !=");
8098        assert_eq!(
8099            err.to_string(),
8100            "with Barry as BDFL, use '<>' instead of '!='"
8101        );
8102        assert_eq!(err.python_location(), (2, 3));
8103    }
8104
8105    #[test]
8106    fn fstring_adjacent_atoms_are_a_missing_comma() {
8107        let err = compile("f'{6 0}'", Mode::Exec, "<fragment>", CompileOpts::default())
8108            .expect_err("adjacent atoms in an f-string field are a syntax error");
8109        assert_eq!(
8110            err.to_string(),
8111            "invalid syntax. Perhaps you forgot a comma?"
8112        );
8113        assert_eq!(err.python_location(), (1, 4));
8114        assert_eq!(err.python_end_location(), Some((1, 7)));
8115
8116        compile(
8117            "f'{not x}'",
8118            Mode::Exec,
8119            "<fragment>",
8120            CompileOpts::default(),
8121        )
8122        .expect("unary not is a prefix, not two atoms");
8123        let err = compile(
8124            "f'{a and}'",
8125            Mode::Exec,
8126            "<fragment>",
8127            CompileOpts::default(),
8128        )
8129        .expect_err("a dangling 'and' is an f-string separator error");
8130        assert_eq!(
8131            err.to_string(),
8132            "f-string: expecting '=', or '!', or ':', or '}'"
8133        );
8134    }
8135
8136    #[test]
8137    fn missing_comma_diagnostic_spans_the_whole_second_atom() {
8138        // `(start, end)` reported as one-based character columns, matching
8139        // `SyntaxError.offset` / `.end_offset`.
8140        let span = |source: &str| {
8141            let err = compile(source, Mode::Eval, "<comma>", CompileOpts::default())
8142                .expect_err("two adjacent atoms are a syntax error");
8143            assert_eq!(
8144                err.to_string(),
8145                "invalid syntax. Perhaps you forgot a comma?"
8146            );
8147            (
8148                err.python_location().1,
8149                err.python_end_location().unwrap().1,
8150            )
8151        };
8152
8153        // A one-character second atom is the case that already worked.
8154        assert_eq!(span("(a b)"), (2, 5));
8155        // A longer one ends where it ends, not one byte in.
8156        assert_eq!(span("(a bb)"), (2, 6));
8157        assert_eq!(span("(a bbb)"), (2, 7));
8158        assert_eq!(span("(1 22)"), (2, 6));
8159        // A non-ASCII atom is one character but several bytes, so counting
8160        // bytes here used to stop inside it and round back off the boundary.
8161        assert_eq!(span("(a \u{3b2})"), (2, 5));
8162        assert_eq!(span("(a \u{3b2}\u{3b2})"), (2, 6));
8163        assert_eq!(span("(\u{3b1}\u{3b1} \u{3b2})"), (2, 6));
8164        // The first atom's width was never the problem; pin it anyway.
8165        assert_eq!(span("(\u{3b1} b)"), (2, 5));
8166        // Other bracket kinds take the same path.
8167        assert_eq!(span("[\u{3b1} \u{3b2}]"), (2, 5));
8168    }
8169
8170    #[test]
8171    fn parenthesized_yield_assignment_uses_invalid_target_message() {
8172        let err = compile(
8173            "def f(): (yield bar) = y\n",
8174            Mode::Exec,
8175            "<yield>",
8176            CompileOpts::default(),
8177        )
8178        .expect_err("parenthesized yield is not an assignment target");
8179        assert_eq!(
8180            err.to_string(),
8181            "cannot assign to yield expression here. Maybe you meant '==' instead of '='?"
8182        );
8183    }
8184
8185    #[test]
8186    fn parenthesized_yield_augassign_uses_illegal_expression_message() {
8187        let err = compile(
8188            "def f(): (yield bar) += y\n",
8189            Mode::Exec,
8190            "<yield>",
8191            CompileOpts::default(),
8192        )
8193        .expect_err("parenthesized yield is not an augmented assignment target");
8194        assert_eq!(
8195            err.to_string(),
8196            "'yield expression' is an illegal expression for augmented assignment"
8197        );
8198    }
8199
8200    #[test]
8201    fn kwarg_unparenthesized_genexp_uses_eq_or_walrus_message() {
8202        let err = compile(
8203            "dict(a = i for i in range(10))\n",
8204            Mode::Exec,
8205            "<kwarg>",
8206            CompileOpts::default(),
8207        )
8208        .expect_err("unparenthesized genexp after '=' is invalid");
8209        assert_eq!(
8210            err.to_string(),
8211            "invalid syntax. Maybe you meant '==' or ':=' instead of '='?"
8212        );
8213    }
8214
8215    #[test]
8216    fn eval_fstring_assignment_keeps_invalid_syntax() {
8217        for source in ["f'' = 3", "f'{0}' = x", "f'{x}' = x"] {
8218            let err = compile(source, Mode::Eval, "<eval>", CompileOpts::default())
8219                .expect_err("assignment is invalid in eval");
8220            assert_eq!(err.to_string(), "invalid syntax", "{source}");
8221        }
8222        let err = compile("f'' = 3", Mode::Exec, "<exec>", CompileOpts::default())
8223            .expect_err("f-string is not an assignment target");
8224        assert_eq!(
8225            err.to_string(),
8226            "cannot assign to f-string expression here. Maybe you meant '==' instead of '='?"
8227        );
8228    }
8229
8230    #[test]
8231    fn if_assignment_uses_eq_or_walrus_message() {
8232        let err = compile(
8233            "if x = 3: pass\n",
8234            Mode::Exec,
8235            "<if>",
8236            CompileOpts::default(),
8237        )
8238        .expect_err("assignment in if condition is invalid");
8239        assert_eq!(
8240            err.to_string(),
8241            "invalid syntax. Maybe you meant '==' or ':=' instead of '='?"
8242        );
8243    }
8244
8245    #[test]
8246    fn parenthesized_if_assignment_uses_eq_or_walrus_message() {
8247        for source in [
8248            "if (x = 3): pass\n",
8249            "if ((x = 3)): pass\n",
8250            "if (x = 3) and y: pass\n",
8251        ] {
8252            let err = compile(source, Mode::Exec, "<if>", CompileOpts::default())
8253                .expect_err("parenthesized assignment in if condition is invalid");
8254            assert_eq!(
8255                err.to_string(),
8256                "invalid syntax. Maybe you meant '==' or ':=' instead of '='?"
8257            );
8258        }
8259    }
8260
8261    #[test]
8262    fn earlier_syntax_error_is_not_replaced_by_later_condition() {
8263        let err = compile(
8264            "@@@\nif x = 3: pass\n",
8265            Mode::Exec,
8266            "<if>",
8267            CompileOpts::default(),
8268        )
8269        .expect_err("the first invalid token is the syntax error");
8270        assert_eq!(err.to_string(), "invalid syntax");
8271        assert_eq!(err.python_location().0, 1);
8272    }
8273
8274    #[test]
8275    fn if_attribute_assignment_uses_invalid_target_hint() {
8276        let err = compile(
8277            "if x.a = 3: pass\n",
8278            Mode::Exec,
8279            "<if>",
8280            CompileOpts::default(),
8281        )
8282        .expect_err("attribute assignment in if condition is invalid");
8283        assert_eq!(
8284            err.to_string(),
8285            "cannot assign to attribute here. Maybe you meant '==' instead of '='?"
8286        );
8287    }
8288
8289    #[test]
8290    fn parenthesized_yield_from_assignment_uses_invalid_target_message() {
8291        let err = compile(
8292            "def f(): (yield from value) = target\n",
8293            Mode::Exec,
8294            "<yield>",
8295            CompileOpts::default(),
8296        )
8297        .expect_err("parenthesized yield from is not an assignment target");
8298        assert_eq!(
8299            err.to_string(),
8300            "cannot assign to yield expression here. Maybe you meant '==' instead of '='?"
8301        );
8302    }
8303
8304    #[test]
8305    fn set_display_assignment_uses_invalid_target_hint() {
8306        let err = compile(
8307            "{1, 2, 3} = 42\n",
8308            Mode::Exec,
8309            "<set>",
8310            CompileOpts::default(),
8311        )
8312        .expect_err("set display is not an assignment target");
8313        assert_eq!(
8314            err.to_string(),
8315            "cannot assign to set display here. Maybe you meant '==' instead of '='?"
8316        );
8317    }
8318
8319    #[test]
8320    fn obsolete_not_equal_diagnostic_spans_the_whole_operator() {
8321        let err = compile("2 <> 3\n", Mode::Exec, "<obsolete>", CompileOpts::default())
8322            .expect_err("'<>' outside Barry mode is a syntax error");
8323        assert_eq!(err.to_string(), "invalid syntax");
8324        assert_eq!(err.python_location(), (1, 3));
8325        assert_eq!(err.python_end_location(), Some((1, 5)));
8326
8327        // Only `<>` spans two characters; any other token that cannot start an
8328        // expression keeps its own location.
8329        let err = compile("2 <;\n", Mode::Exec, "<obsolete>", CompileOpts::default())
8330            .expect_err("'<;' is a syntax error");
8331        assert_eq!(err.to_string(), "invalid syntax");
8332        assert_eq!(err.python_location(), (1, 4));
8333        assert_eq!(err.python_end_location(), Some((1, 5)));
8334
8335        // A `<>` that starts a statement is reported at the `<` too, where the
8336        // parser stops instead of one character in.
8337        let err = compile("<>\n", Mode::Exec, "<obsolete>", CompileOpts::default())
8338            .expect_err("a bare '<>' is a syntax error");
8339        assert_eq!(err.to_string(), "invalid syntax");
8340        assert_eq!(err.python_location(), (1, 1));
8341        assert_eq!(err.python_end_location(), Some((1, 3)));
8342
8343        // A bracket left open earlier in the source outranks the operator.
8344        let err = compile(
8345            "(\n2 <> 3",
8346            Mode::Exec,
8347            "<obsolete>",
8348            CompileOpts::default(),
8349        )
8350        .expect_err("the bracket is never closed");
8351        assert_eq!(err.to_string(), "'(' was never closed");
8352        assert_eq!(err.python_location(), (1, 1));
8353    }
8354
8355    #[test]
8356    fn barry_as_flufl_does_not_rewrite_strings_or_comments() {
8357        compile(
8358            "from __future__ import barry_as_FLUFL\nx = '<>'\n# <>\n",
8359            Mode::Exec,
8360            "<barry>",
8361            CompileOpts::default(),
8362        )
8363        .expect("Barry markers in strings and comments should stay untouched");
8364    }
8365
8366    #[test]
8367    fn syntax_error_before_barry_not_equal_takes_precedence() {
8368        let err = compile(
8369            "from __future__ import barry_as_FLUFL\n<>\n2 != 3\n",
8370            Mode::Exec,
8371            "<barry>",
8372            CompileOpts::default(),
8373        )
8374        .expect_err("the earlier invalid comparison should fail");
8375        assert_eq!(err.to_string(), "invalid syntax");
8376        assert_eq!(err.python_location(), (2, 1));
8377    }
8378
8379    #[test]
8380    fn unclosed_bracket_before_barry_not_equal_takes_precedence() {
8381        let err = compile(
8382            "from __future__ import barry_as_FLUFL\n(\n2 != 3",
8383            Mode::Exec,
8384            "<barry>",
8385            CompileOpts::default(),
8386        )
8387        .expect_err("the earlier unclosed bracket should fail");
8388        assert_eq!(err.to_string(), "'(' was never closed");
8389        assert_eq!(err.python_location(), (2, 1));
8390    }
8391
8392    #[test]
8393    fn compile_phello() {
8394        let code = r#"
8395initialized = True
8396def main():
8397    print("Hello world!")
8398if __name__ == '__main__':
8399    main()
8400"#;
8401        let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8402        dbg!(compiled.expect("compile error"));
8403    }
8404
8405    #[test]
8406    fn compile_if_elif_else() {
8407        let code = r#"
8408if False:
8409    pass
8410elif False:
8411    pass
8412elif False:
8413    pass
8414else:
8415    pass
8416"#;
8417        let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8418        dbg!(compiled.expect("compile error"));
8419    }
8420
8421    #[test]
8422    fn compile_lambda() {
8423        let code = r#"
8424lambda: 'a'
8425"#;
8426        let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8427        dbg!(compiled.expect("compile error"));
8428    }
8429
8430    #[test]
8431    fn compile_lambda2() {
8432        let code = r#"
8433(lambda x: f'hello, {x}')('world}')
8434"#;
8435        let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8436        dbg!(compiled.expect("compile error"));
8437    }
8438
8439    #[test]
8440    fn compile_lambda3() {
8441        let code = r#"
8442def g():
8443    pass
8444def f():
8445    if False:
8446        return lambda x: g(x)
8447    elif False:
8448        return g
8449    else:
8450        return g
8451"#;
8452        let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8453        dbg!(compiled.expect("compile error"));
8454    }
8455
8456    #[test]
8457    fn compile_call_arg_lambda_default() {
8458        let code = "signature((lambda a=10: a))";
8459        let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8460        dbg!(compiled.expect("compile error"));
8461    }
8462
8463    #[test]
8464    fn compile_generic_function_parameter_default() {
8465        let code = "def __repr__[T: str](self, default: T = '') -> str: pass";
8466        let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8467        dbg!(compiled.expect("compile error"));
8468    }
8469
8470    #[test]
8471    fn compile_int() {
8472        let code = r#"
8473a = 0xFF
8474"#;
8475        let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8476        dbg!(compiled.expect("compile error"));
8477    }
8478
8479    #[test]
8480    fn compile_bigint() {
8481        let code = r#"
8482a = 0xFFFFFFFFFFFFFFFFFFFFFFFF
8483"#;
8484        let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8485        dbg!(compiled.expect("compile error"));
8486    }
8487
8488    #[test]
8489    fn compile_fstring() {
8490        let code1 = r#"
8491assert f"1" == '1'
8492    "#;
8493        let compiled = compile(code1, Mode::Exec, "<>", CompileOpts::default());
8494        dbg!(compiled.expect("compile error"));
8495
8496        let code2 = r#"
8497assert f"{1}" == '1'
8498    "#;
8499        let compiled = compile(code2, Mode::Exec, "<>", CompileOpts::default());
8500        dbg!(compiled.expect("compile error"));
8501        let code3 = r#"
8502assert f"{1+1}" == '2'
8503    "#;
8504        let compiled = compile(code3, Mode::Exec, "<>", CompileOpts::default());
8505        dbg!(compiled.expect("compile error"));
8506
8507        let code4 = r#"
8508assert f"{{{(lambda: f'{1}')}" == '{1'
8509    "#;
8510        let compiled = compile(code4, Mode::Exec, "<>", CompileOpts::default());
8511        dbg!(compiled.expect("compile error"));
8512
8513        let code5 = r#"
8514assert f"a{1}" == 'a1'
8515    "#;
8516        let compiled = compile(code5, Mode::Exec, "<>", CompileOpts::default());
8517        dbg!(compiled.expect("compile error"));
8518
8519        let code6 = r#"
8520assert f"{{{(lambda x: f'hello, {x}')('world}')}" == '{hello, world}'
8521    "#;
8522        let compiled = compile(code6, Mode::Exec, "<>", CompileOpts::default());
8523        dbg!(compiled.expect("compile error"));
8524    }
8525
8526    #[test]
8527    fn simple_enum() {
8528        let code = r#"
8529import enum
8530@enum._simple_enum(enum.IntFlag, boundary=enum.KEEP)
8531class RegexFlag:
8532    NOFLAG = 0
8533    DEBUG = 1
8534print(RegexFlag.NOFLAG & RegexFlag.DEBUG)
8535"#;
8536        let compiled = compile(code, Mode::Exec, "<string>", CompileOpts::default());
8537        dbg!(compiled.expect("compile error"));
8538    }
8539}