1extern crate alloc;
2
3use alloc::borrow::Cow;
4pub use ruff_python_ast::token::{TokenKind, Tokens};
5use ruff_python_parser::ParseErrorType;
6use ruff_source_file::{PositionEncoding, SourceFile, SourceFileBuilder, SourceLocation};
7use ruff_text_size::{Ranged, TextSize, TextSlice};
8use rustpython_codegen::{compile, symboltable};
9use thiserror::Error;
10
11pub use rustpython_codegen::compile::CompileOpts;
12pub use rustpython_compiler_core::{Mode, bytecode::CodeObject};
13
14pub use ruff_python_ast as ast;
16pub use ruff_python_parser as parser;
17pub use rustpython_codegen as codegen;
18pub use rustpython_compiler_core as core;
19
20#[derive(Error, Debug)]
21pub enum CompileErrorType {
22 #[error(transparent)]
23 Codegen(#[from] codegen::error::CodegenErrorType),
24 #[error(transparent)]
25 Parse(#[from] ParseErrorType),
26}
27
28#[derive(Error, Debug)]
29pub struct ParseError {
30 #[source]
31 pub error: ParseErrorType,
32 pub raw_location: ruff_text_size::TextRange,
33 pub location: SourceLocation,
34 pub end_location: SourceLocation,
35 pub source_path: String,
36 pub is_unclosed_bracket: bool,
38 pub is_unclosed_string: bool,
40}
41
42impl ::core::fmt::Display for ParseError {
43 fn fmt(&self, f: &mut ::core::fmt::Formatter<'_>) -> ::core::fmt::Result {
44 self.error.fmt(f)
45 }
46}
47
48#[derive(Error, Debug)]
49pub enum CompileError {
50 #[error(transparent)]
51 Codegen(#[from] codegen::error::CodegenError),
52 #[error(transparent)]
53 Parse(#[from] ParseError),
54}
55
56impl CompileError {
57 #[must_use]
58 pub fn from_ruff_parse_error(
59 error: parser::ParseError,
60 source_file: &SourceFile,
61 mode: Mode,
62 ) -> Self {
63 let raw_location = error.location;
64 let diagnostic = match cpython_parse_diagnostic_override(&error, source_file, mode) {
65 Some(diagnostic) => diagnostic,
66 None => default_parse_diagnostic(error, source_file),
67 };
68
69 Self::Parse(ParseError {
70 error: diagnostic.error,
71 raw_location,
72 location: diagnostic.location,
73 end_location: diagnostic.end_location,
74 source_path: source_file.name().to_owned(),
75 is_unclosed_bracket: diagnostic.is_unclosed_bracket,
76 is_unclosed_string: diagnostic.is_unclosed_string,
77 })
78 }
79
80 fn from_source_error(source_file: &SourceFile, diagnostic: CpythonDiagnostic) -> Self {
81 let (location, end_location) = source_locations(
82 source_file,
83 diagnostic.range.start(),
84 diagnostic.range.end(),
85 );
86 Self::Parse(ParseError {
87 error: parser::ParseErrorType::OtherError(diagnostic.message),
88 raw_location: diagnostic.range,
89 location,
90 end_location,
91 source_path: source_file.name().to_owned(),
92 is_unclosed_bracket: diagnostic.is_unclosed_bracket,
93 is_unclosed_string: diagnostic.is_unclosed_string,
94 })
95 }
96
97 #[must_use]
98 pub const fn location(&self) -> Option<SourceLocation> {
99 match self {
100 Self::Codegen(codegen_error) => codegen_error.location,
101 Self::Parse(parse_error) => Some(parse_error.location),
102 }
103 }
104
105 #[must_use]
106 pub const fn python_location(&self) -> (usize, usize) {
107 if let Some(location) = self.location() {
108 (location.line.get(), location.character_offset.get())
109 } else {
110 (0, 0)
111 }
112 }
113
114 #[must_use]
115 pub fn python_end_location(&self) -> Option<(usize, usize)> {
116 match self {
117 Self::Codegen(codegen_error) => codegen_error
118 .end_location
119 .map(|end| (end.line.get(), end.character_offset.get())),
120 Self::Parse(parse_error) => Some((
121 parse_error.end_location.line.get(),
122 parse_error.end_location.character_offset.get(),
123 )),
124 }
125 }
126
127 #[must_use]
128 pub fn source_path(&self) -> &str {
129 match self {
130 Self::Codegen(codegen_error) => &codegen_error.source_path,
131 Self::Parse(parse_error) => &parse_error.source_path,
132 }
133 }
134}
135
136fn source_location(source_file: &SourceFile, offset: TextSize) -> SourceLocation {
141 let text = source_file.source_text();
142 let mut index = offset.to_usize().min(text.len());
143 while !text.is_char_boundary(index) {
144 index -= 1;
145 }
146 source_file
147 .to_source_code()
148 .source_location(TextSize::new(index as u32), PositionEncoding::Utf32)
149}
150
151fn source_locations(
152 source_file: &SourceFile,
153 start: TextSize,
154 end: TextSize,
155) -> (SourceLocation, SourceLocation) {
156 (
157 source_location(source_file, start),
158 source_location(source_file, end),
159 )
160}
161
162struct NormalizedParseDiagnostic {
163 error: parser::ParseErrorType,
164 location: SourceLocation,
165 end_location: SourceLocation,
166 is_unclosed_bracket: bool,
167 is_unclosed_string: bool,
168}
169
170impl NormalizedParseDiagnostic {
171 const fn new(
172 error: parser::ParseErrorType,
173 location: SourceLocation,
174 end_location: SourceLocation,
175 ) -> Self {
176 Self {
177 error,
178 location,
179 end_location,
180 is_unclosed_bracket: false,
181 is_unclosed_string: false,
182 }
183 }
184
185 fn other(source_file: &SourceFile, diagnostic: CpythonDiagnostic) -> Self {
186 let (location, end_location) = source_locations(
187 source_file,
188 diagnostic.range.start(),
189 diagnostic.range.end(),
190 );
191 let mut diagnostic_out = Self::new(
192 parser::ParseErrorType::OtherError(diagnostic.message),
193 location,
194 end_location,
195 );
196 diagnostic_out.is_unclosed_string = diagnostic.is_unclosed_string;
197 diagnostic_out.is_unclosed_bracket = diagnostic.is_unclosed_bracket;
198 diagnostic_out
199 }
200
201 const fn with_unclosed_bracket(mut self, is_unclosed_bracket: bool) -> Self {
202 self.is_unclosed_bracket = is_unclosed_bracket;
203 self
204 }
205}
206
207#[derive(Clone)]
212struct CpythonDiagnostic {
213 message: String,
214 range: ruff_text_size::TextRange,
215 is_unclosed_string: bool,
216 is_unclosed_bracket: bool,
217}
218
219impl CpythonDiagnostic {
220 fn new(message: String, start: usize, end: usize) -> Self {
225 Self {
226 message,
227 range: ruff_text_size::TextRange::new(
228 TextSize::new(start as u32),
229 TextSize::new(end as u32),
230 ),
231 is_unclosed_string: false,
232 is_unclosed_bracket: false,
233 }
234 }
235
236 const fn with_unclosed_string(mut self) -> Self {
237 self.is_unclosed_string = true;
238 self
239 }
240
241 const fn with_unclosed_bracket(mut self) -> Self {
242 self.is_unclosed_bracket = true;
243 self
244 }
245}
246
247#[derive(Clone, Copy, PartialEq, Eq)]
254enum OverrideClass {
255 Lexer,
256 Decode,
257 Print,
258}
259
260struct RankedOverride {
261 diagnostic: CpythonDiagnostic,
262 unclosed_bracket: bool,
263 class: OverrideClass,
264}
265
266fn consider_override(
267 best: &mut Option<RankedOverride>,
268 diagnostic: CpythonDiagnostic,
269 class: OverrideClass,
270) {
271 let unclosed_bracket = diagnostic.is_unclosed_bracket;
272 consider_ranked(best, diagnostic, unclosed_bracket, class);
273}
274
275fn consider_ranked(
276 best: &mut Option<RankedOverride>,
277 diagnostic: CpythonDiagnostic,
278 unclosed_bracket: bool,
279 class: OverrideClass,
280) {
281 if best
282 .as_ref()
283 .is_none_or(|current| diagnostic.range.start() < current.diagnostic.range.start())
284 {
285 *best = Some(RankedOverride {
286 diagnostic,
287 unclosed_bracket,
288 class,
289 });
290 }
291}
292
293fn cpython_parse_diagnostic_override(
294 error: &parser::ParseError,
295 source_file: &SourceFile,
296 mode: Mode,
297) -> Option<NormalizedParseDiagnostic> {
298 let source_text = source_file.source_text();
299
300 macro_rules! source_error {
301 ($expr:expr) => {
302 if let Some(error) = $expr {
303 return Some(NormalizedParseDiagnostic::other(source_file, error));
304 }
305 };
306 }
307
308 let mut earliest: Option<RankedOverride> = None;
309 if let Some(diagnostic) = invalid_number_literal_error(source_text) {
310 consider_override(&mut earliest, diagnostic, OverrideClass::Lexer);
311 }
312 if let Some(diagnostic) = incompatible_string_prefix_error(source_text) {
313 consider_override(&mut earliest, diagnostic, OverrideClass::Lexer);
314 }
315 if let Some(diagnostic) = non_printable_character_error(source_text) {
316 consider_override(&mut earliest, diagnostic, OverrideClass::Lexer);
317 }
318 let bracket = bracket_syntax_error(source_text);
319 if let Some(bracket) = bracket.as_ref() {
320 if !bracket.unclosed {
325 consider_ranked(
326 &mut earliest,
327 bracket.diagnostic.clone(),
328 false,
329 OverrideClass::Lexer,
330 );
331 }
332 }
333 let mut saw_decode = false;
334 if let Some(diagnostic) = malformed_unicode_n_escape_error(source_text) {
335 saw_decode = true;
336 consider_override(&mut earliest, diagnostic, OverrideClass::Decode);
337 }
338 if let Some(diagnostic) = invalid_interpolated_string_error(source_text) {
339 saw_decode = true;
340 consider_override(&mut earliest, diagnostic, OverrideClass::Decode);
341 }
342 if let Some(diagnostic) = mixed_tstring_literal_error(error, source_text) {
343 saw_decode = true;
344 consider_override(&mut earliest, diagnostic, OverrideClass::Decode);
345 }
346 let line_continuation = matches!(
349 &error.error,
350 parser::ParseErrorType::Lexical(parser::LexicalErrorType::LineContinuationError)
351 );
352 if !saw_decode
353 && !line_continuation
354 && let Some(diagnostic) = unterminated_string_error(source_text, mode)
355 {
356 consider_override(&mut earliest, diagnostic, OverrideClass::Lexer);
357 }
358 let indent_error = matches!(
360 &error.error,
361 parser::ParseErrorType::Lexical(parser::LexicalErrorType::IndentationError)
362 | parser::ParseErrorType::UnexpectedIndentation
363 ) || expected_indented_block_error(error, source_text).is_some();
364 if !indent_error
365 && earliest
366 .as_ref()
367 .is_none_or(|current| current.class != OverrideClass::Lexer)
368 && let Some(diagnostic) = invalid_legacy_statement_error(source_text)
369 {
370 consider_override(&mut earliest, diagnostic, OverrideClass::Print);
371 }
372 if !saw_decode
373 && earliest
374 .as_ref()
375 .is_none_or(|current| current.class == OverrideClass::Print)
376 && let Some(bracket) = bracket.filter(|bracket| bracket.unclosed)
377 {
378 consider_ranked(
379 &mut earliest,
380 bracket.diagnostic,
381 true,
382 OverrideClass::Lexer,
383 );
384 }
385 if let Some(override_diag) = earliest {
386 return Some(
387 NormalizedParseDiagnostic::other(source_file, override_diag.diagnostic)
388 .with_unclosed_bracket(override_diag.unclosed_bracket),
389 );
390 }
391
392 if matches!(
393 &error.error,
394 parser::ParseErrorType::Lexical(parser::LexicalErrorType::LineContinuationError)
395 ) {
396 let terminal_backslash = source_text.len().checked_sub(1);
400 if matches!(mode, Mode::Exec)
401 && terminal_backslash == Some(error.location.start().to_usize())
402 {
403 let loc = source_line_end_location(source_file, error.location.start());
404 return Some(NormalizedParseDiagnostic::new(
405 parser::ParseErrorType::OtherError("unexpected EOF while parsing".to_owned()),
406 loc,
407 loc,
408 ));
409 }
410 let loc = source_location(source_file, error.location.start() + TextSize::from(1));
411 return Some(NormalizedParseDiagnostic::new(
412 parser::ParseErrorType::OtherError(
413 "unexpected character after line continuation character".to_owned(),
414 ),
415 loc,
416 loc,
417 ));
418 }
419
420 source_error!(unterminated_string_error(source_text, mode));
421 source_error!(expected_indented_block_error(error, source_text));
422
423 if matches!(
424 &error.error,
425 parser::ParseErrorType::Lexical(parser::LexicalErrorType::Eof)
426 ) {
427 return Some(eof_parse_diagnostic(error, source_file));
428 }
429
430 source_error!(invalid_type_param_error(source_text));
431 source_error!(invalid_comprehension_error(source_text));
432 source_error!(invalid_parameter_star_annotation_error(source_text));
433 source_error!(invalid_parameter_list_error(source_text));
434 source_error!(invalid_call_argument_error(source_text));
435
436 if is_missing_comma_between_literals(error) {
437 let (loc, end_loc) = adjusted_error_locations(source_file, error.location);
438 let msg = "invalid syntax. Perhaps you forgot a comma?".into();
439 return Some(NormalizedParseDiagnostic::new(
440 parser::ParseErrorType::OtherError(msg),
441 loc,
442 end_loc,
443 ));
444 }
445
446 source_error!(invalid_dict_error(source_text));
447 if !matches!(mode, Mode::Eval) {
448 source_error!(invalid_collection_assignment_error(source_text));
449 }
450 source_error!(invalid_group_error(source_text));
451 source_error!(invalid_def_type_params_error(source_text));
452 source_error!(invalid_expression_error(source_text));
453 source_error!(invalid_named_expression_error(source_text));
454 if !matches!(mode, Mode::Eval) {
457 source_error!(invalid_plain_assignment_error(source_text));
458 source_error!(expression_assignment_error(source_text));
459 source_error!(invalid_annotation_target_error(source_text));
460 source_error!(invalid_assignment_target_error(source_text));
461 source_error!(invalid_condition_assignment_error(
462 source_text,
463 error.location.start().to_usize()
464 ));
465 source_error!(invalid_augassign_target_error(source_text));
466 source_error!(invalid_for_target_error(source_text));
467 source_error!(invalid_with_target_error(source_text));
468 source_error!(invalid_delete_target_error(source_text));
469 source_error!(invalid_standalone_except_error(source_text));
470 source_error!(invalid_import_statement_error(source_text));
471 source_error!(invalid_import_target_error(source_text));
472 source_error!(invalid_except_as_target_error(source_text));
473 source_error!(invalid_match_mapping_rest_wildcard_error(source_text));
474 source_error!(invalid_match_as_target_error(source_text));
475 source_error!(invalid_for_if_clause_error(source_text));
476 source_error!(invalid_if_expression_statement_error(source_text));
477 source_error!(invalid_else_elif_error(source_text));
478 source_error!(mixed_except_handlers_error(source_text));
479 }
480
481 if matches!(
482 &error.error,
483 parser::ParseErrorType::Lexical(parser::LexicalErrorType::InvalidByteLiteral)
484 ) && let Some((start, end)) =
485 bytes_literal_span(source_text, error.location.start().to_usize())
486 {
487 let (loc, end_loc) = source_locations(
488 source_file,
489 TextSize::new(start as u32),
490 TextSize::new(end as u32),
491 );
492 return Some(NormalizedParseDiagnostic::new(
493 error.error.clone(),
494 loc,
495 end_loc,
496 ));
497 }
498
499 if matches!(
500 &error.error,
501 parser::ParseErrorType::Lexical(parser::LexicalErrorType::IndentationError)
502 ) {
503 let end_loc = source_line_end_location(source_file, error.location.start());
504 return Some(NormalizedParseDiagnostic::new(
505 error.error.clone(),
506 end_loc,
507 end_loc,
508 ));
509 }
510
511 if matches!(
512 &error.error,
513 parser::ParseErrorType::InvalidAssignmentTarget
514 ) {
515 return Some(invalid_assignment_target_diagnostic(error, source_file));
516 }
517
518 if matches!(
519 &error.error,
520 parser::ParseErrorType::InvalidNamedAssignmentTarget
521 ) {
522 let (loc, end_loc) = adjusted_error_locations(source_file, error.location);
523 let target = source_file.source_text().slice(error.location);
524 let msg = format!("cannot use assignment expressions with {target}");
525 return Some(NormalizedParseDiagnostic::new(
526 parser::ParseErrorType::OtherError(msg),
527 loc,
528 end_loc,
529 ));
530 }
531
532 if matches!(
537 &error.error,
538 parser::ParseErrorType::ExpectedExpression
539 | parser::ParseErrorType::UnexpectedExpressionToken
540 ) {
541 let (loc, end_loc) = adjusted_error_locations(source_file, error.location);
542 return Some(NormalizedParseDiagnostic::new(
543 parser::ParseErrorType::OtherError("invalid syntax".into()),
544 loc,
545 end_loc,
546 ));
547 }
548
549 None
550}
551
552fn eof_parse_diagnostic(
553 error: &parser::ParseError,
554 source_file: &SourceFile,
555) -> NormalizedParseDiagnostic {
556 let source_text = source_file.source_text();
557 if let Some((bracket_char, bracket_offset)) = find_unclosed_bracket(source_text) {
558 let loc = source_location(source_file, TextSize::new(bracket_offset as u32));
559 let end_loc = SourceLocation {
560 line: loc.line,
561 character_offset: loc.character_offset.saturating_add(1),
562 };
563 let msg = format!("'{bracket_char}' was never closed");
564 NormalizedParseDiagnostic::new(parser::ParseErrorType::OtherError(msg), loc, end_loc)
565 .with_unclosed_bracket(true)
566 } else {
567 let end_loc = source_line_end_location(source_file, error.location.start());
568 NormalizedParseDiagnostic::new(error.error.clone(), end_loc, end_loc)
569 }
570}
571
572fn invalid_assignment_target_diagnostic(
573 error: &parser::ParseError,
574 source_file: &SourceFile,
575) -> NormalizedParseDiagnostic {
576 let (loc, end_loc) = adjusted_error_locations(source_file, error.location);
577 let expr_str = source_file.source_text().slice(error.location);
578
579 let msg = parser::parse_expression(expr_str).map_or_else(
580 |_| match expr_str {
581 "yield" => "assignment to yield expression not possible".into(),
582 _ => format!("cannot assign to {expr_str}"),
583 },
584 |parsed| match *parsed.syntax().body {
585 ast::Expr::Call(_) => "cannot assign to function call".into(),
586 ast::Expr::BinOp(_) => "cannot assign to expression".into(),
587 ast::Expr::If(_) => "cannot assign to conditional expression".into(),
588 ast::Expr::Generator(_) => "cannot assign to generator expression".into(),
589 ast::Expr::Yield(_) | ast::Expr::YieldFrom(_) => {
590 "cannot assign to yield expression here. Maybe you meant '==' instead of '='?"
591 .into()
592 }
593 ast::Expr::FString(_) => "invalid syntax".into(),
594 ast::Expr::StringLiteral(_)
595 | ast::Expr::BytesLiteral(_)
596 | ast::Expr::NumberLiteral(_) => {
597 "cannot assign to literal here. Maybe you meant '==' instead of '='?".into()
598 }
599 ast::Expr::EllipsisLiteral(_) => {
600 "cannot assign to ellipsis here. Maybe you meant '==' instead of '='?".into()
601 }
602 _ => format!("cannot assign to {expr_str}"),
603 },
604 );
605
606 NormalizedParseDiagnostic::new(parser::ParseErrorType::OtherError(msg), loc, end_loc)
607}
608
609fn default_parse_diagnostic(
610 error: parser::ParseError,
611 source_file: &SourceFile,
612) -> NormalizedParseDiagnostic {
613 let (loc, end_loc) = adjusted_error_locations(source_file, error.location);
614 NormalizedParseDiagnostic::new(error.error, loc, end_loc)
615}
616
617fn adjusted_error_locations(
618 source_file: &SourceFile,
619 range: ruff_text_size::TextRange,
620) -> (SourceLocation, SourceLocation) {
621 let mut locations = source_locations(source_file, range.start(), range.end());
622 if locations.1.character_offset.get() == 1 && locations.1.line > locations.0.line {
623 locations.1 = source_location(source_file, range.end() - TextSize::from(1));
624 locations.1.character_offset = locations.1.character_offset.saturating_add(1);
625 } else if range.is_empty() {
626 locations.1.character_offset = locations.1.character_offset.saturating_add(1);
629 }
630 locations
631}
632
633fn source_line_end_location(source_file: &SourceFile, offset: TextSize) -> SourceLocation {
634 let loc = source_location(source_file, offset);
635 let line_idx = loc.line.to_zero_indexed();
636 let line = source_file
637 .source_text()
638 .split('\n')
639 .nth(line_idx)
640 .unwrap_or("");
641 let line_end_col = line.chars().count() + 1;
642 SourceLocation {
643 line: loc.line,
644 character_offset: ruff_source_file::OneIndexed::new(line_end_col)
645 .unwrap_or(loc.character_offset),
646 }
647}
648
649fn is_missing_comma_between_literals(error: &parser::ParseError) -> bool {
650 matches!(
651 &error.error,
652 parser::ParseErrorType::ExpectedToken { expected, found }
653 if matches!((expected, found), (TokenKind::Comma, TokenKind::Int))
654 )
655}
656
657fn is_ascii_identifier_char(byte: u8) -> bool {
658 byte == b'_' || byte.is_ascii_alphanumeric()
659}
660
661fn identifier_continue_before(bytes: &[u8], index: usize) -> bool {
662 if index == 0 {
663 return false;
664 }
665 if bytes[index - 1].is_ascii() {
666 return is_ascii_identifier_char(bytes[index - 1]);
667 }
668 let mut start = index - 1;
669 while start > 0 && bytes[start] & 0b1100_0000 == 0b1000_0000 {
670 start -= 1;
671 }
672 ::core::str::from_utf8(&bytes[start..index])
673 .ok()
674 .and_then(|text| text.chars().next_back())
675 .is_some_and(|ch| ch == '_' || ch.is_alphanumeric())
676}
677
678fn numeric_keyword_suffix(rest: &[u8]) -> bool {
679 rest.starts_with(b"and")
680 || rest.starts_with(b"else")
681 || rest.starts_with(b"for")
682 || rest.starts_with(b"if")
683 || rest.starts_with(b"in")
684 || rest.starts_with(b"is")
685 || rest.starts_with(b"or")
686 || rest.starts_with(b"not")
687}
688
689fn consume_decimal_digits(bytes: &[u8], mut index: usize) -> usize {
690 while index < bytes.len() {
691 match bytes[index] {
692 b'0'..=b'9' => index += 1,
693 b'_' if bytes
694 .get(index + 1)
695 .is_some_and(|byte| byte.is_ascii_digit()) =>
696 {
697 index += 2;
698 }
699 _ => break,
700 }
701 }
702 index
703}
704
705fn consume_radix_digits(bytes: &[u8], mut index: usize, is_digit: impl Fn(u8) -> bool) -> usize {
706 while index < bytes.len() {
707 if is_digit(bytes[index]) {
708 index += 1;
709 } else if bytes.get(index) == Some(&b'_')
710 && bytes.get(index + 1).is_some_and(|&byte| is_digit(byte))
711 {
712 index += 2;
713 } else {
714 break;
715 }
716 }
717 index
718}
719
720fn invalid_radix_literal_error(
721 bytes: &[u8],
722 start: usize,
723 kind: &'static str,
724 is_digit: impl Fn(u8) -> bool,
725) -> Option<(String, usize)> {
726 let mut index = start + 2;
727 let mut has_digit = false;
728 loop {
729 let Some(&byte) = bytes.get(index) else {
730 return if has_digit {
731 None
732 } else {
733 Some((format!("invalid {kind} literal"), start + 1))
734 };
735 };
736 if byte == b'_' {
737 let Some(&next) = bytes.get(index + 1) else {
738 return Some((format!("invalid {kind} literal"), index));
739 };
740 if is_digit(next) {
741 has_digit = true;
742 index += 2;
743 continue;
744 }
745 if next.is_ascii_digit() && matches!(kind, "binary" | "octal") {
746 return Some((
747 format!("invalid digit '{}' in {kind} literal", next as char),
748 index + 1,
749 ));
750 }
751 return Some((format!("invalid {kind} literal"), index));
752 }
753 if is_digit(byte) {
754 has_digit = true;
755 index += 1;
756 continue;
757 }
758 if byte.is_ascii_digit() && matches!(kind, "binary" | "octal") {
759 return Some((
760 format!("invalid digit '{}' in {kind} literal", byte as char),
761 index,
762 ));
763 }
764 if has_digit {
765 return None;
766 }
767 return Some((format!("invalid {kind} literal"), start + 1));
768 }
769}
770
771fn decimal_tail_error(bytes: &[u8], mut index: usize) -> Option<usize> {
772 loop {
773 while bytes.get(index).is_some_and(|byte| byte.is_ascii_digit()) {
774 index += 1;
775 }
776 if bytes.get(index) != Some(&b'_') {
777 return None;
778 }
779 let underscore = index;
780 index += 1;
781 if !bytes.get(index).is_some_and(|byte| byte.is_ascii_digit()) {
782 return Some(underscore);
783 }
784 }
785}
786
787fn decimal_tail_end(bytes: &[u8], mut index: usize) -> usize {
788 loop {
789 while bytes.get(index).is_some_and(|byte| byte.is_ascii_digit()) {
790 index += 1;
791 }
792 if bytes.get(index) == Some(&b'_')
793 && bytes
794 .get(index + 1)
795 .is_some_and(|byte| byte.is_ascii_digit())
796 {
797 index += 2;
798 } else {
799 return index;
800 }
801 }
802}
803
804fn invalid_decimal_literal_error(bytes: &[u8], start: usize) -> Option<(String, usize)> {
805 if bytes.get(start) == Some(&b'.') {
806 return None;
807 }
808 let message = "invalid decimal literal".to_owned();
809 if let Some(offset) = decimal_tail_error(bytes, start) {
810 return Some((message, offset));
811 }
812
813 let mut index = decimal_tail_end(bytes, start);
814 if bytes.get(index) == Some(&b'.') {
815 if bytes.get(index + 1) == Some(&b'_') {
816 return Some((message, index));
817 }
818 if let Some(offset) = decimal_tail_error(bytes, index + 1) {
819 return Some((message, offset));
820 }
821 index = decimal_tail_end(bytes, index + 1);
822 }
823 if matches!(bytes.get(index), Some(b'e' | b'E')) {
824 let exponent = index;
825 index += 1;
826 let sign = if matches!(bytes.get(index), Some(b'+' | b'-')) {
827 let sign = index;
828 index += 1;
829 Some(sign)
830 } else {
831 None
832 };
833 if !bytes.get(index).is_some_and(|byte| byte.is_ascii_digit()) {
834 return Some((message, sign.unwrap_or_else(|| exponent.saturating_sub(1))));
837 }
838 if let Some(offset) = decimal_tail_error(bytes, index) {
839 return Some((message, offset));
840 }
841 }
842 None
843}
844
845fn leading_zero_decimal_literal_error(bytes: &[u8], start: usize) -> Option<CpythonDiagnostic> {
846 if bytes.get(start) != Some(&b'0') {
847 return None;
848 }
849 let mut index = start;
850 loop {
851 match bytes.get(index) {
852 Some(b'0') => index += 1,
853 Some(b'_')
854 if bytes
855 .get(index + 1)
856 .is_some_and(|byte| byte.is_ascii_digit()) =>
857 {
858 index += 1;
859 }
860 _ => break,
861 }
862 }
863 if bytes.get(index).is_some_and(|byte| byte.is_ascii_digit()) {
864 let after_digits = decimal_tail_end(bytes, index);
865 if !matches!(
866 bytes.get(after_digits),
867 Some(b'.' | b'e' | b'E' | b'j' | b'J')
868 ) {
869 return Some(CpythonDiagnostic::new(
870 "leading zeros in decimal integer literals are not permitted; use an 0o prefix for octal integers"
871 .to_owned(),
872 start,
873 index,
874 ));
875 }
876 }
877 None
878}
879
880fn invalid_numeric_literal_error(bytes: &[u8], start: usize) -> Option<CpythonDiagnostic> {
881 if bytes.get(start) == Some(&b'0') {
882 let radix = match bytes.get(start + 1) {
883 Some(b'x' | b'X') => invalid_radix_literal_error(bytes, start, "hexadecimal", |byte| {
884 byte.is_ascii_hexdigit()
885 }),
886 Some(b'o' | b'O') => invalid_radix_literal_error(bytes, start, "octal", |byte| {
887 matches!(byte, b'0'..=b'7')
888 }),
889 Some(b'b' | b'B') => invalid_radix_literal_error(bytes, start, "binary", |byte| {
890 matches!(byte, b'0' | b'1')
891 }),
892 _ => None,
893 };
894 if let Some(radix) = radix {
895 return Some(point_span(radix));
896 }
897 if let Some(err) = leading_zero_decimal_literal_error(bytes, start) {
898 return Some(err);
899 }
900 }
901 invalid_decimal_literal_error(bytes, start).map(point_span)
902}
903
904fn consume_exponent(bytes: &[u8], index: usize) -> usize {
905 if !matches!(bytes.get(index), Some(b'e' | b'E')) {
906 return index;
907 }
908 let mut cursor = index + 1;
909 if matches!(bytes.get(cursor), Some(b'+' | b'-')) {
910 cursor += 1;
911 }
912 if bytes.get(cursor).is_some_and(|byte| byte.is_ascii_digit()) {
913 consume_decimal_digits(bytes, cursor)
914 } else {
915 index
916 }
917}
918
919fn number_literal_end(bytes: &[u8], start: usize) -> Option<(&'static str, usize)> {
920 if bytes.get(start) == Some(&b'.') {
921 if !bytes
922 .get(start + 1)
923 .is_some_and(|byte| byte.is_ascii_digit())
924 {
925 return None;
926 }
927 let mut index = consume_decimal_digits(bytes, start + 1);
928 index = consume_exponent(bytes, index);
929 if matches!(bytes.get(index), Some(b'j' | b'J')) {
930 return Some(("imaginary", index + 1));
931 }
932 return Some(("decimal", index));
933 }
934
935 if !bytes.get(start).is_some_and(|byte| byte.is_ascii_digit()) {
936 return None;
937 }
938
939 if bytes.get(start) == Some(&b'0') {
940 match bytes.get(start + 1) {
941 Some(b'x' | b'X') => {
942 let end = consume_radix_digits(bytes, start + 2, |byte| byte.is_ascii_hexdigit());
943 return Some(("hexadecimal", end));
944 }
945 Some(b'o' | b'O') => {
946 let end =
947 consume_radix_digits(bytes, start + 2, |byte| matches!(byte, b'0'..=b'7'));
948 return Some(("octal", end));
949 }
950 Some(b'b' | b'B') => {
951 let end =
952 consume_radix_digits(bytes, start + 2, |byte| matches!(byte, b'0' | b'1'));
953 return Some(("binary", end));
954 }
955 _ => {}
956 }
957 }
958
959 let mut index = consume_decimal_digits(bytes, start);
960 if bytes.get(index) == Some(&b'.') {
961 index = consume_decimal_digits(bytes, index + 1);
962 }
963 index = consume_exponent(bytes, index);
964 if matches!(bytes.get(index), Some(b'j' | b'J')) {
965 return Some(("imaginary", index + 1));
966 }
967 Some(("decimal", index))
968}
969
970fn quoted_string_is_closed(bytes: &[u8], start: usize) -> bool {
971 let quote = bytes[start];
972 let triple = bytes.get(start + 1) == Some("e) && bytes.get(start + 2) == Some("e);
973 let end = skip_quoted_string(bytes, start);
974 if triple {
975 end >= start + 6
976 && bytes[end - 3] == quote
977 && bytes[end - 2] == quote
978 && bytes[end - 1] == quote
979 } else {
980 end > start + 1 && bytes[end - 1] == quote
981 }
982}
983
984fn bytes_literal_span(source: &str, error_at: usize) -> Option<(usize, usize)> {
985 let bytes = source.as_bytes();
986 if error_at > bytes.len() || bytes.is_empty() {
987 return None;
988 }
989 let mut quote_idx = error_at.min(bytes.len().saturating_sub(1));
990 loop {
991 if matches!(bytes[quote_idx], b'\'' | b'"') {
992 break;
993 }
994 if quote_idx == 0 {
995 return None;
996 }
997 quote_idx -= 1;
998 }
999 let mut start = quote_idx;
1000 while start > 0 && matches!(bytes[start - 1], b'b' | b'B' | b'r' | b'R') {
1001 start -= 1;
1002 }
1003 if !matches!(bytes.get(start), Some(b'b' | b'B' | b'r' | b'R')) {
1004 return None;
1005 }
1006 let end = skip_quoted_string(bytes, quote_idx);
1007 Some((start, end))
1008}
1009
1010fn skip_quoted_string(bytes: &[u8], mut index: usize) -> usize {
1011 let quote = bytes[index];
1012 let triple = bytes.get(index + 1) == Some("e) && bytes.get(index + 2) == Some("e);
1013 let quote_len = if triple { 3 } else { 1 };
1014 index += quote_len;
1015 while index < bytes.len() {
1016 if bytes[index] == b'\\' {
1017 index = (index + 2).min(bytes.len());
1018 } else if triple
1019 && bytes.get(index) == Some("e)
1020 && bytes.get(index + 1) == Some("e)
1021 && bytes.get(index + 2) == Some("e)
1022 {
1023 return index + 3;
1024 } else if !triple && bytes[index] == quote {
1025 return index + 1;
1026 } else {
1027 index += 1;
1028 }
1029 }
1030 index
1031}
1032
1033fn point_span((message, offset): (String, usize)) -> CpythonDiagnostic {
1035 CpythonDiagnostic::new(message, offset, offset)
1036}
1037
1038fn invalid_number_literal_error(source: &str) -> Option<CpythonDiagnostic> {
1039 let bytes = source.as_bytes();
1040 let mut index = 0;
1041 while index < bytes.len() {
1042 match bytes[index] {
1043 b'#' => {
1044 while index < bytes.len() && bytes[index] != b'\n' {
1045 index += 1;
1046 }
1047 }
1048 b'\'' | b'"' => {
1049 index = skip_quoted_string(bytes, index);
1050 }
1051 byte if byte >= 0x80 || byte == b'_' || byte.is_ascii_alphabetic() => {
1052 index += 1;
1053 while index < bytes.len()
1054 && (bytes[index] >= 0x80 || is_ascii_identifier_char(bytes[index]))
1055 {
1056 index += 1;
1057 }
1058 }
1059 b'.' | b'0'..=b'9' => {
1060 if let Some(err) = invalid_numeric_literal_error(bytes, index) {
1061 return Some(err);
1062 }
1063 let Some((kind, end)) = number_literal_end(bytes, index) else {
1064 index += 1;
1065 continue;
1066 };
1067 if end > index {
1068 if source[end..].starts_with('⁄') {
1069 return Some(CpythonDiagnostic::new(
1070 "invalid character '⁄' (U+2044)".to_owned(),
1071 end,
1072 end,
1073 ));
1074 }
1075 if bytes
1076 .get(end)
1077 .is_some_and(|byte| *byte < 128 && is_ascii_identifier_char(*byte))
1078 && !numeric_keyword_suffix(&bytes[end..])
1079 {
1080 let offset = end.saturating_sub(1);
1081 return Some(CpythonDiagnostic::new(
1082 format!("invalid {kind} literal"),
1083 offset,
1084 offset,
1085 ));
1086 }
1087 }
1088 index = end.max(index + 1);
1089 }
1090 _ => index += 1,
1091 }
1092 }
1093 None
1094}
1095
1096fn cpython_indented_block_clause(message: &str) -> Option<&'static str> {
1097 let clause = message.strip_prefix("Expected an indented block after ")?;
1098 Some(match clause {
1099 "`if` statement" => "'if' statement",
1100 "`elif` clause" => "'elif' statement",
1101 "`else` clause" => "'else' statement",
1102 "`for` statement" => "'for' statement",
1103 "`with` statement" => "'with' statement",
1104 "`while` statement" => "'while' statement",
1105 "`try` statement" => "'try' statement",
1106 "`except` clause" => "'except' statement",
1107 "`finally` clause" => "'finally' statement",
1108 "`match` statement" => "'match' statement",
1109 "`case` block" => "'case' statement",
1110 "`class` definition" => "class definition",
1111 "function definition" => "function definition",
1112 _ => return None,
1113 })
1114}
1115
1116fn previous_non_empty_line_number(source: &str, offset: usize) -> Option<usize> {
1117 let bytes = source.as_bytes();
1118 let mut index = offset.min(bytes.len());
1119 while index > 0 {
1120 let line_end = index;
1121 while index > 0 && bytes[index - 1] != b'\n' {
1122 index -= 1;
1123 }
1124 let line_start = index;
1125 let content_start = skip_horizontal_whitespace(bytes, line_start);
1126 let mut content_end = line_end;
1127 while content_end > content_start
1128 && matches!(
1129 bytes.get(content_end - 1),
1130 Some(b' ' | b'\t' | b'\r' | b'\x0c')
1131 )
1132 {
1133 content_end -= 1;
1134 }
1135 if content_start < content_end {
1136 return Some(
1137 source[..line_start]
1138 .bytes()
1139 .filter(|byte| *byte == b'\n')
1140 .count()
1141 + 1,
1142 );
1143 }
1144 index = line_start.saturating_sub(1);
1145 }
1146 None
1147}
1148
1149fn expected_indented_block_error(
1150 error: &parser::ParseError,
1151 source: &str,
1152) -> Option<CpythonDiagnostic> {
1153 let parser::ParseErrorType::OtherError(message) = &error.error else {
1154 return None;
1155 };
1156 let mut clause = cpython_indented_block_clause(message)?;
1157 let start = error.location.start().to_usize();
1158 let end = error.location.end().to_usize();
1159 let line = previous_non_empty_line_number(source, start)?;
1160 if clause == "'except' statement"
1161 && let Some(previous_line) = previous_non_empty_line(source, start)
1162 && matches!(
1163 previous_line.trim_start(),
1164 line if line.starts_with("except*") || line.starts_with("except *")
1165 )
1166 {
1167 clause = "'except*' statement";
1168 }
1169 Some(CpythonDiagnostic::new(
1170 format!("expected an indented block after {clause} on line {line}"),
1171 start,
1172 end,
1173 ))
1174}
1175
1176fn previous_non_empty_line(source: &str, offset: usize) -> Option<&str> {
1177 let bytes = source.as_bytes();
1178 let mut index = offset.min(bytes.len());
1179 while index > 0 {
1180 let line_end = index;
1181 while index > 0 && bytes[index - 1] != b'\n' {
1182 index -= 1;
1183 }
1184 let line_start = index;
1185 let mut content_start = line_start;
1186 while content_start < line_end
1187 && matches!(bytes[content_start], b' ' | b'\t' | b'\n' | b'\r' | b'\x0c')
1188 {
1189 content_start += 1;
1190 }
1191 let mut content_end = line_end;
1192 while content_end > content_start
1193 && matches!(bytes[content_end - 1], b' ' | b'\t' | b'\r' | b'\x0c')
1194 {
1195 content_end -= 1;
1196 }
1197 if content_start < content_end {
1198 return source.get(line_start..line_end);
1199 }
1200 index = line_start.saturating_sub(1);
1201 }
1202 None
1203}
1204
1205fn starts_identifier(bytes: &[u8], index: usize, word: &[u8]) -> bool {
1206 bytes.get(index..index + word.len()) == Some(word)
1207 && index
1208 .checked_sub(1)
1209 .and_then(|before| bytes.get(before))
1210 .is_none_or(|byte| !is_ascii_identifier_char(*byte))
1211 && bytes
1212 .get(index + word.len())
1213 .is_none_or(|byte| !is_ascii_identifier_char(*byte))
1214}
1215
1216fn is_plain_assignment_operator(bytes: &[u8], index: usize) -> bool {
1217 bytes.get(index) == Some(&b'=')
1218 && bytes.get(index + 1) != Some(&b'=')
1219 && !matches!(
1220 index.checked_sub(1).and_then(|before| bytes.get(before)),
1221 Some(b'=' | b'!' | b'<' | b'>' | b':')
1222 )
1223}
1224
1225fn is_simple_keyword_name(bytes: &[u8], mut start: usize, mut end: usize) -> bool {
1226 while matches!(
1227 bytes.get(start),
1228 Some(b' ' | b'\t' | b'\n' | b'\r' | b'\x0c')
1229 ) {
1230 start += 1;
1231 }
1232 while end > start
1233 && matches!(
1234 bytes.get(end - 1),
1235 Some(b' ' | b'\t' | b'\n' | b'\r' | b'\x0c')
1236 )
1237 {
1238 end -= 1;
1239 }
1240 let Some(&first) = bytes.get(start) else {
1241 return false;
1242 };
1243 if !(first == b'_' || first.is_ascii_alphabetic() || first >= 0x80) {
1244 return false;
1245 }
1246 let mut index = start + 1;
1247 while index < end {
1248 if bytes[index] < 0x80 && !is_ascii_identifier_char(bytes[index]) {
1249 return false;
1250 }
1251 index += 1;
1252 }
1253 true
1254}
1255
1256fn is_function_parameter_list(bytes: &[u8], paren: usize) -> bool {
1257 let mut cursor = paren;
1258 while cursor > 0 && matches!(bytes.get(cursor - 1), Some(b' ' | b'\t' | b'\x0c')) {
1259 cursor -= 1;
1260 }
1261 if cursor > 0 && bytes.get(cursor - 1) == Some(&b']') {
1262 let mut bracket = cursor;
1263 let mut level = 0usize;
1264 while bracket > 0 {
1265 bracket -= 1;
1266 match bytes[bracket] {
1267 b']' => level += 1,
1268 b'[' => {
1269 level = level.saturating_sub(1);
1270 if level == 0 {
1271 cursor = bracket;
1272 break;
1273 }
1274 }
1275 _ => {}
1276 }
1277 }
1278 while cursor > 0 && matches!(bytes.get(cursor - 1), Some(b' ' | b'\t' | b'\x0c')) {
1279 cursor -= 1;
1280 }
1281 }
1282 while cursor > 0
1283 && bytes
1284 .get(cursor - 1)
1285 .is_some_and(|byte| *byte >= 0x80 || is_ascii_identifier_char(*byte))
1286 {
1287 cursor -= 1;
1288 }
1289 while cursor > 0 && matches!(bytes.get(cursor - 1), Some(b' ' | b'\t' | b'\x0c')) {
1290 cursor -= 1;
1291 }
1292 cursor >= 3
1293 && starts_identifier(bytes, cursor - 3, b"def")
1294 && cursor
1295 .checked_sub(4)
1296 .and_then(|before| bytes.get(before))
1297 .is_none_or(|byte| !is_ascii_identifier_char(*byte))
1298}
1299
1300#[derive(Clone, Copy)]
1301enum ParameterListKind {
1302 Function,
1303 Lambda,
1304}
1305
1306fn matching_delimiter(bytes: &[u8], open: usize, close: u8) -> Option<usize> {
1307 let mut index = open;
1308 let mut level = 0usize;
1309 while index < bytes.len() {
1310 match bytes[index] {
1311 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1312 b'(' | b'[' | b'{' => {
1313 level += 1;
1314 index += 1;
1315 }
1316 byte if byte == close => {
1317 level = level.saturating_sub(1);
1318 if level == 0 {
1319 return Some(index);
1320 }
1321 index += 1;
1322 }
1323 b')' | b']' | b'}' => {
1324 level = level.saturating_sub(1);
1325 index += 1;
1326 }
1327 _ => index += 1,
1328 }
1329 }
1330 None
1331}
1332
1333fn find_lambda_parameter_end(bytes: &[u8], mut index: usize) -> Option<usize> {
1334 let mut level = 0usize;
1335 while index < bytes.len() {
1336 match bytes[index] {
1337 b'#' if level == 0 => return None,
1338 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1339 b'(' | b'[' | b'{' => {
1340 level += 1;
1341 index += 1;
1342 }
1343 b')' | b']' | b'}' => {
1344 level = level.saturating_sub(1);
1345 index += 1;
1346 }
1347 b':' if level == 0 => return Some(index),
1348 _ => index += 1,
1349 }
1350 }
1351 None
1352}
1353
1354fn top_level_byte(bytes: &[u8], mut index: usize, end: usize, needle: u8) -> Option<usize> {
1355 let mut level = 0usize;
1356 while index < end {
1357 match bytes[index] {
1358 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1359 byte if level == 0 && byte == needle => return Some(index),
1360 b'(' | b'[' | b'{' => {
1361 level += 1;
1362 index += 1;
1363 }
1364 b')' | b']' | b'}' => {
1365 level = level.saturating_sub(1);
1366 index += 1;
1367 }
1368 _ => index += 1,
1369 }
1370 }
1371 None
1372}
1373
1374fn identifier_end(bytes: &[u8], mut index: usize, end: usize) -> usize {
1375 if !bytes
1376 .get(index)
1377 .is_some_and(|byte| *byte >= 0x80 || *byte == b'_' || byte.is_ascii_alphabetic())
1378 {
1379 return index;
1380 }
1381 index += 1;
1382 while index < end
1383 && bytes
1384 .get(index)
1385 .is_some_and(|byte| *byte >= 0x80 || is_ascii_identifier_char(*byte))
1386 {
1387 index += 1;
1388 }
1389 index
1390}
1391
1392fn expression_slice_is_tuple(source: &str, start: usize, end: usize) -> bool {
1393 let bytes = source.as_bytes();
1394 let (start, end) = trim_target_range(bytes, start, end);
1395 if start >= end {
1396 return false;
1397 }
1398 let Ok(parsed) = parser::parse(&source[start..end], parser::Mode::Expression.into()) else {
1399 return false;
1400 };
1401 matches!(parsed.into_syntax(), ast::Mod::Expression(expression) if matches!(*expression.body, ast::Expr::Tuple(_)))
1402}
1403
1404fn type_param_list_open(bytes: &[u8], open: usize) -> bool {
1405 let mut cursor = open;
1406 while cursor > 0 && matches!(bytes.get(cursor - 1), Some(b' ' | b'\t' | b'\x0c')) {
1407 cursor -= 1;
1408 }
1409 while cursor > 0
1410 && bytes
1411 .get(cursor - 1)
1412 .is_some_and(|byte| *byte >= 0x80 || is_ascii_identifier_char(*byte))
1413 {
1414 cursor -= 1;
1415 }
1416 while cursor > 0 && matches!(bytes.get(cursor - 1), Some(b' ' | b'\t' | b'\x0c')) {
1417 cursor -= 1;
1418 }
1419 (cursor >= 3 && starts_identifier(bytes, cursor - 3, b"def"))
1420 || (cursor >= 5 && starts_identifier(bytes, cursor - 5, b"class"))
1421 || (cursor >= 4 && starts_identifier(bytes, cursor - 4, b"type"))
1422}
1423
1424fn invalid_type_param_item_error(
1425 source: &str,
1426 start: usize,
1427 end: usize,
1428) -> Option<CpythonDiagnostic> {
1429 let bytes = source.as_bytes();
1430 let (start, end) = trim_target_range(bytes, start, end);
1431 if start >= end || bytes.get(start) != Some(&b'*') {
1432 return None;
1433 }
1434 let is_param_spec = bytes.get(start + 1) == Some(&b'*');
1435 let name_start = start + if is_param_spec { 2 } else { 1 };
1436 let name_end = identifier_end(bytes, name_start, end);
1437 if name_start == name_end {
1438 return None;
1439 }
1440 let colon = next_non_horizontal_whitespace(bytes, name_end);
1441 if colon >= end || bytes.get(colon) != Some(&b':') {
1442 return None;
1443 }
1444 let has_constraints = expression_slice_is_tuple(source, colon + 1, end);
1445 let message = match (is_param_spec, has_constraints) {
1446 (false, false) => "cannot use bound with TypeVarTuple",
1447 (false, true) => "cannot use constraints with TypeVarTuple",
1448 (true, false) => "cannot use bound with ParamSpec",
1449 (true, true) => "cannot use constraints with ParamSpec",
1450 };
1451 Some(CpythonDiagnostic::new(message.to_owned(), colon, colon + 1))
1452}
1453
1454fn invalid_type_param_list_error(
1455 source: &str,
1456 open: usize,
1457 close: usize,
1458) -> Option<CpythonDiagnostic> {
1459 let bytes = source.as_bytes();
1460 let mut item_start = open + 1;
1461 let mut index = item_start;
1462 let mut level = 0usize;
1463 while index <= close {
1464 if index == close || (level == 0 && bytes.get(index) == Some(&b',')) {
1465 if let Some(error) = invalid_type_param_item_error(source, item_start, index) {
1466 return Some(error);
1467 }
1468 item_start = index + 1;
1469 index += 1;
1470 continue;
1471 }
1472 match bytes[index] {
1473 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1474 b'(' | b'[' | b'{' => {
1475 level += 1;
1476 index += 1;
1477 }
1478 b')' | b']' | b'}' => {
1479 level = level.saturating_sub(1);
1480 index += 1;
1481 }
1482 _ => index += 1,
1483 }
1484 }
1485 None
1486}
1487
1488fn invalid_type_param_error(source: &str) -> Option<CpythonDiagnostic> {
1489 let bytes = source.as_bytes();
1490 let mut index = 0usize;
1491 while index < bytes.len() {
1492 match bytes[index] {
1493 b'#' => {
1494 while index < bytes.len() && bytes[index] != b'\n' {
1495 index += 1;
1496 }
1497 }
1498 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1499 b'[' if type_param_list_open(bytes, index) => {
1500 let Some(close) = matching_delimiter(bytes, index, b']') else {
1501 index += 1;
1502 continue;
1503 };
1504 if let Some(error) = invalid_type_param_list_error(source, index, close) {
1505 return Some(error);
1506 }
1507 index = close + 1;
1508 }
1509 _ => index += 1,
1510 }
1511 }
1512 None
1513}
1514
1515fn invalid_comprehension_in_slice(
1516 bytes: &[u8],
1517 open: usize,
1518 close: usize,
1519) -> Option<CpythonDiagnostic> {
1520 let for_index = find_keyword_at_level(bytes, open + 1, close, b"for")?;
1521 let item_start = next_non_horizontal_whitespace(bytes, open + 1);
1522 if item_start >= for_index {
1523 return None;
1524 }
1525 if bytes.get(item_start..item_start + 2) == Some(b"**") && bytes.get(open) == Some(&b'{') {
1526 return Some(CpythonDiagnostic::new(
1527 "dict unpacking cannot be used in dict comprehension".to_owned(),
1528 item_start,
1529 item_start + 2,
1530 ));
1531 }
1532 if bytes.get(item_start..item_start + 2) == Some(b"**") && bytes.get(open) == Some(&b'(') {
1533 return Some(CpythonDiagnostic::new(
1534 "invalid syntax".to_owned(),
1535 for_index,
1536 for_index + 3,
1537 ));
1538 }
1539 if bytes.get(item_start) == Some(&b'*') {
1540 return Some(CpythonDiagnostic::new(
1541 "iterable unpacking cannot be used in comprehension".to_owned(),
1542 item_start,
1543 item_start + 1,
1544 ));
1545 }
1546 if !matches!(bytes.get(open), Some(b'[' | b'{')) {
1547 return None;
1548 }
1549 if top_level_colon(bytes, open + 1, for_index).is_none()
1550 && let Some(comma) = top_level_byte(bytes, open + 1, for_index, b',')
1551 {
1552 let (start, _) = trim_target_range(bytes, open + 1, comma);
1553 return Some(CpythonDiagnostic::new(
1554 "did you forget parentheses around the comprehension target?".to_owned(),
1555 start,
1556 comma + 1,
1557 ));
1558 }
1559 None
1560}
1561
1562fn invalid_comprehension_error(source: &str) -> Option<CpythonDiagnostic> {
1563 let bytes = source.as_bytes();
1564 let mut index = 0usize;
1565 while index < bytes.len() {
1566 match bytes[index] {
1567 b'#' => {
1568 while index < bytes.len() && bytes[index] != b'\n' {
1569 index += 1;
1570 }
1571 }
1572 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1573 b'(' | b'[' | b'{' => {
1574 let close_byte = match bytes[index] {
1575 b'(' => b')',
1576 b'[' => b']',
1577 _ => b'}',
1578 };
1579 let Some(close) = matching_delimiter(bytes, index, close_byte) else {
1580 index += 1;
1581 continue;
1582 };
1583 if let Some(error) = invalid_comprehension_in_slice(bytes, index, close) {
1584 return Some(error);
1585 }
1586 index = close + 1;
1587 }
1588 _ => index += 1,
1589 }
1590 }
1591 None
1592}
1593
1594fn invalid_group_in_slice(bytes: &[u8], open: usize, close: usize) -> Option<CpythonDiagnostic> {
1595 let (item_start, item_end) = trim_target_range(bytes, open + 1, close);
1596 if item_start >= item_end
1597 || top_level_byte(bytes, item_start, item_end, b',').is_some()
1598 || top_level_colon(bytes, item_start, item_end).is_some()
1599 || find_keyword_at_level(bytes, item_start, item_end, b"for").is_some()
1600 {
1601 return None;
1602 }
1603 if bytes.get(item_start..item_start + 2) == Some(b"**") {
1604 return Some(CpythonDiagnostic::new(
1605 "cannot use double starred expression here".to_owned(),
1606 item_start,
1607 item_start + 2,
1608 ));
1609 }
1610 if bytes.get(item_start) == Some(&b'*') {
1611 return Some(CpythonDiagnostic::new(
1612 "cannot use starred expression here".to_owned(),
1613 item_start,
1614 item_start + 1,
1615 ));
1616 }
1617 None
1618}
1619
1620fn invalid_group_error(source: &str) -> Option<CpythonDiagnostic> {
1621 let bytes = source.as_bytes();
1622 let mut index = 0usize;
1623 while index < bytes.len() {
1624 match bytes[index] {
1625 b'#' => {
1626 while index < bytes.len() && bytes[index] != b'\n' {
1627 index += 1;
1628 }
1629 }
1630 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1631 b'(' => {
1632 let Some(close) = matching_delimiter(bytes, index, b')') else {
1633 index += 1;
1634 continue;
1635 };
1636 if let Some(error) = invalid_group_in_slice(bytes, index, close) {
1637 return Some(error);
1638 }
1639 index = close + 1;
1640 }
1641 _ => index += 1,
1642 }
1643 }
1644 None
1645}
1646
1647fn invalid_parameter_star_annotation_error(source: &str) -> Option<CpythonDiagnostic> {
1648 let bytes = source.as_bytes();
1649 let mut index = 0usize;
1650 while index < bytes.len() {
1651 match bytes[index] {
1652 b'#' => {
1653 while index < bytes.len() && bytes[index] != b'\n' {
1654 index += 1;
1655 }
1656 }
1657 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1658 b'(' => {
1659 let Some(close) = matching_delimiter(bytes, index, b')') else {
1660 index += 1;
1661 continue;
1662 };
1663 let mut param_start = index + 1;
1664 while param_start < close {
1665 let param_end =
1666 find_byte_at_level(bytes, param_start, close, b',').unwrap_or(close);
1667 if let Some(colon) = top_level_colon(bytes, param_start, param_end) {
1668 let value_start = next_non_horizontal_whitespace(bytes, colon + 1);
1669 if bytes.get(value_start) == Some(&b'*') {
1670 return Some(CpythonDiagnostic::new(
1671 "invalid syntax".to_owned(),
1672 value_start,
1673 value_start + 1,
1674 ));
1675 }
1676 }
1677 param_start = param_end.saturating_add(1);
1678 }
1679 index = close + 1;
1680 }
1681 _ => index += 1,
1682 }
1683 }
1684 None
1685}
1686
1687fn invalid_def_type_params_error(source: &str) -> Option<CpythonDiagnostic> {
1688 let bytes = source.as_bytes();
1689 let mut index = 0;
1690 while index < bytes.len() {
1691 match bytes[index] {
1692 b'#' => {
1693 while index < bytes.len() && bytes[index] != b'\n' {
1694 index += 1;
1695 }
1696 }
1697 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1698 _ if starts_identifier(bytes, index, b"def") => {
1699 let name_start = skip_horizontal_whitespace(bytes, index + 3);
1700 let name_end = identifier_end(bytes, name_start, bytes.len());
1701 let bracket = skip_horizontal_whitespace(bytes, name_end);
1702 if bytes.get(bracket) == Some(&b'[') {
1703 let Some(close) = matching_delimiter(bytes, bracket, b']') else {
1704 index = bracket + 1;
1705 continue;
1706 };
1707 let after_close = skip_horizontal_whitespace(bytes, close + 1);
1708 if bytes.get(after_close) == Some(&b'(')
1709 && type_param_list_is_malformed(bytes, bracket + 1, close)
1710 {
1711 return Some(CpythonDiagnostic::new(
1712 "expected '('".to_owned(),
1713 bracket,
1714 bracket + 1,
1715 ));
1716 }
1717 }
1718 index = name_end.max(index + 3);
1719 }
1720 _ => index += 1,
1721 }
1722 }
1723 None
1724}
1725
1726fn type_param_list_is_malformed(bytes: &[u8], start: usize, end: usize) -> bool {
1727 let mut index = start;
1728 let mut expect_item = true;
1729 while index < end {
1730 index = skip_horizontal_whitespace(bytes, index);
1731 if index >= end {
1732 break;
1733 }
1734 if bytes[index] == b',' {
1735 if expect_item {
1736 return true;
1737 }
1738 expect_item = true;
1739 index += 1;
1740 continue;
1741 }
1742 if !expect_item {
1743 return true;
1744 }
1745 if bytes.get(index..index + 2) == Some(b"**") {
1746 index += 2;
1747 } else if bytes.get(index) == Some(&b'*') {
1748 index += 1;
1749 }
1750 let item_start = skip_horizontal_whitespace(bytes, index);
1751 let item_end = identifier_end(bytes, item_start, end);
1752 if item_end == item_start {
1753 return true;
1754 }
1755 index = item_end;
1756 if bytes.get(skip_horizontal_whitespace(bytes, index)) == Some(&b':') {
1757 index = skip_horizontal_whitespace(bytes, index) + 1;
1758 while index < end && bytes[index] != b',' {
1759 index = match bytes[index] {
1760 b'\'' | b'"' => skip_quoted_string(bytes, index),
1761 _ => index + 1,
1762 };
1763 }
1764 }
1765 expect_item = false;
1766 }
1767 false
1768}
1769
1770fn invalid_parameter_list_slice_error(
1771 source: &str,
1772 start: usize,
1773 end: usize,
1774 kind: ParameterListKind,
1775) -> Option<CpythonDiagnostic> {
1776 let bytes = source.as_bytes();
1777 let mut index = start;
1778 let mut level = 0usize;
1779 let mut default_seen = false;
1780 let mut keyword_only = false;
1781 let mut slash_seen = false;
1782 let mut var_keyword_seen = false;
1783 while index < end {
1784 match bytes[index] {
1785 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1786 _ if level == 0
1787 && var_keyword_seen
1788 && bytes.get(index).is_some_and(|byte| {
1789 *byte >= 0x80 || *byte == b'_' || byte.is_ascii_alphabetic()
1790 }) =>
1791 {
1792 let name_end = identifier_end(bytes, index, end);
1793 return Some(CpythonDiagnostic::new(
1794 "arguments cannot follow var-keyword argument".to_owned(),
1795 index,
1796 name_end,
1797 ));
1798 }
1799 _ if level == 0
1800 && !keyword_only
1801 && bytes.get(index).is_some_and(|byte| {
1802 *byte >= 0x80 || *byte == b'_' || byte.is_ascii_alphabetic()
1803 }) =>
1804 {
1805 let param_end = find_byte_at_level(bytes, index, end, b',')
1806 .or_else(|| top_level_byte(bytes, index, end, b')'))
1807 .or_else(|| {
1808 matches!(kind, ParameterListKind::Lambda)
1809 .then(|| top_level_byte(bytes, index, end, b':'))
1810 .flatten()
1811 })
1812 .unwrap_or(end);
1813 let name_end = identifier_end(bytes, index, param_end);
1814 if let Some(eq) = top_level_byte(bytes, index, param_end, b'=') {
1815 let value_start = next_non_horizontal_whitespace(bytes, eq + 1);
1816 if value_start >= param_end
1817 || matches!(bytes.get(value_start), Some(b',' | b')' | b':'))
1818 {
1819 let message = if matches!(kind, ParameterListKind::Lambda)
1820 && matches!(bytes.get(value_start), Some(b':'))
1821 {
1822 "invalid syntax"
1823 } else {
1824 "expected default value expression"
1825 };
1826 return Some(CpythonDiagnostic::new(message.to_owned(), eq, eq + 1));
1827 }
1828 default_seen = true;
1829 } else if default_seen {
1830 return Some(CpythonDiagnostic::new(
1831 "parameter without a default follows parameter with a default".to_owned(),
1832 index,
1833 name_end,
1834 ));
1835 }
1836 index = param_end;
1837 }
1838 b'(' if level == 0 => {
1839 let close = matching_delimiter(bytes, index, b')')
1840 .filter(|close| *close <= end)
1841 .unwrap_or(index + 1);
1842 let message = match kind {
1843 ParameterListKind::Function => "Function parameters cannot be parenthesized",
1844 ParameterListKind::Lambda => {
1845 "Lambda expression parameters cannot be parenthesized"
1846 }
1847 };
1848 return Some(CpythonDiagnostic::new(message.to_owned(), index, close + 1));
1849 }
1850 b'(' | b'[' | b'{' => {
1851 level += 1;
1852 index += 1;
1853 }
1854 b')' | b']' | b'}' => {
1855 level = level.saturating_sub(1);
1856 index += 1;
1857 }
1858 b'/' if level == 0 => {
1859 if var_keyword_seen {
1860 return Some(CpythonDiagnostic::new(
1861 "arguments cannot follow var-keyword argument".to_owned(),
1862 index,
1863 index + 1,
1864 ));
1865 }
1866 if slash_seen {
1867 return Some(CpythonDiagnostic::new(
1868 "/ may appear only once".to_owned(),
1869 index,
1870 index + 1,
1871 ));
1872 }
1873 slash_seen = true;
1874 let next = next_non_horizontal_whitespace(bytes, index + 1);
1875 if bytes.get(next) == Some(&b'*') {
1876 return Some(CpythonDiagnostic::new(
1877 "expected comma between / and *".to_owned(),
1878 next,
1879 next + 1,
1880 ));
1881 }
1882 index += 1;
1883 }
1884 b'*' if level == 0 => {
1885 if var_keyword_seen {
1886 return Some(CpythonDiagnostic::new(
1887 "arguments cannot follow var-keyword argument".to_owned(),
1888 index,
1889 index + 1,
1890 ));
1891 }
1892 keyword_only = true;
1893 let stars = usize::from(bytes.get(index + 1) == Some(&b'*')) + 1;
1894 let name_start = next_non_horizontal_whitespace(bytes, index + stars);
1895 for keyword in [b"True".as_slice(), b"False".as_slice(), b"None".as_slice()] {
1896 if starts_identifier(bytes, name_start, keyword) {
1897 return Some(CpythonDiagnostic::new(
1898 "invalid syntax".to_owned(),
1899 name_start,
1900 name_start + keyword.len(),
1901 ));
1902 }
1903 }
1904 let param_end = find_byte_at_level(bytes, name_start, end, b',')
1905 .or_else(|| top_level_byte(bytes, name_start, end, b')'))
1906 .or_else(|| {
1907 matches!(kind, ParameterListKind::Lambda)
1908 .then(|| top_level_byte(bytes, name_start, end, b':'))
1909 .flatten()
1910 })
1911 .unwrap_or(end);
1912 if stars == 1 && matches!(bytes.get(name_start), Some(b')' | b',' | b':')) {
1913 return Some(CpythonDiagnostic::new(
1914 "named arguments must follow bare *".to_owned(),
1915 index,
1916 index + 1,
1917 ));
1918 }
1919 if stars == 1 && top_level_byte(bytes, name_start, param_end, b'=').is_some() {
1920 return Some(CpythonDiagnostic::new(
1921 "var-positional argument cannot have default value".to_owned(),
1922 index,
1923 index + 1,
1924 ));
1925 }
1926 if stars == 2 && top_level_byte(bytes, name_start, param_end, b'=').is_some() {
1927 return Some(CpythonDiagnostic::new(
1928 "var-keyword argument cannot have default value".to_owned(),
1929 index,
1930 index + 2,
1931 ));
1932 }
1933 if stars == 2 {
1934 var_keyword_seen = true;
1935 index = param_end;
1936 continue;
1937 }
1938 index += stars;
1939 }
1940 b'=' if level == 0 => {
1941 let value_start = next_non_horizontal_whitespace(bytes, index + 1);
1942 if value_start >= end || matches!(bytes.get(value_start), Some(b',' | b')' | b':'))
1943 {
1944 if matches!(kind, ParameterListKind::Lambda)
1945 && matches!(bytes.get(value_start), Some(b':'))
1946 {
1947 return Some(CpythonDiagnostic::new(
1948 "invalid syntax".to_owned(),
1949 index,
1950 index + 1,
1951 ));
1952 }
1953 return Some(CpythonDiagnostic::new(
1954 "expected default value expression".to_owned(),
1955 index,
1956 index + 1,
1957 ));
1958 }
1959 index += 1;
1960 }
1961 _ => index += 1,
1962 }
1963 }
1964 None
1965}
1966
1967fn invalid_parameter_list_error(source: &str) -> Option<CpythonDiagnostic> {
1968 let bytes = source.as_bytes();
1969 let mut index = 0usize;
1970 while index < bytes.len() {
1971 match bytes[index] {
1972 b'#' => {
1973 while index < bytes.len() && bytes[index] != b'\n' {
1974 index += 1;
1975 }
1976 }
1977 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
1978 _ if starts_identifier(bytes, index, b"def") => {
1979 let Some(paren) = top_level_byte(bytes, index + 3, bytes.len(), b'(') else {
1980 index += 3;
1981 continue;
1982 };
1983 let Some(close) = matching_delimiter(bytes, paren, b')') else {
1984 index = paren + 1;
1985 continue;
1986 };
1987 if let Some(error) = invalid_parameter_list_slice_error(
1988 source,
1989 paren + 1,
1990 close,
1991 ParameterListKind::Function,
1992 ) {
1993 return Some(error);
1994 }
1995 index = close + 1;
1996 }
1997 _ if starts_identifier(bytes, index, b"lambda") => {
1998 let params_start = index + 6;
1999 let Some(params_end) = find_lambda_parameter_end(bytes, params_start) else {
2000 index = params_start;
2001 continue;
2002 };
2003 if let Some(error) = invalid_parameter_list_slice_error(
2004 source,
2005 params_start,
2006 params_end,
2007 ParameterListKind::Lambda,
2008 ) {
2009 return Some(error);
2010 }
2011 index = params_end + 1;
2012 }
2013 _ => index += 1,
2014 }
2015 }
2016 None
2017}
2018
2019#[derive(Clone, Copy)]
2020struct CallArgFrame {
2021 level: usize,
2022 arg_start: Option<usize>,
2023 in_call: bool,
2024}
2025
2026fn next_non_horizontal_whitespace(bytes: &[u8], mut index: usize) -> usize {
2027 while matches!(bytes.get(index), Some(b' ' | b'\t' | b'\x0c')) {
2028 index += 1;
2029 }
2030 index
2031}
2032
2033fn invalid_call_argument_assignment_error(
2034 source: &str,
2035 arg_start: usize,
2036 equal: usize,
2037) -> Option<CpythonDiagnostic> {
2038 let bytes = source.as_bytes();
2039 let start = bytes[arg_start..equal]
2040 .iter()
2041 .rposition(|byte| *byte == b'\n')
2042 .map_or(arg_start, |newline| arg_start + newline + 1);
2043 let (target_start, target_end) = trim_target_range(bytes, start, equal);
2044 if target_start >= target_end {
2045 return None;
2046 }
2047 let value_start = next_non_horizontal_whitespace(bytes, equal + 1);
2048 if matches!(bytes.get(value_start), None | Some(b',' | b')')) {
2049 return Some(CpythonDiagnostic::new(
2050 "expected argument value expression".to_owned(),
2051 target_start,
2052 equal + 1,
2053 ));
2054 }
2055 if bytes.get(target_start..target_start + 2) == Some(b"**") {
2056 return Some(CpythonDiagnostic::new(
2057 "cannot assign to keyword argument unpacking".to_owned(),
2058 target_start,
2059 value_start,
2060 ));
2061 }
2062 if bytes.get(target_start) == Some(&b'*') {
2063 return Some(CpythonDiagnostic::new(
2064 "cannot assign to iterable argument unpacking".to_owned(),
2065 target_start,
2066 value_start,
2067 ));
2068 }
2069 for keyword in [b"True".as_slice(), b"False".as_slice(), b"None".as_slice()] {
2070 if bytes.get(target_start..target_end) == Some(keyword) {
2071 let keyword = ::core::str::from_utf8(keyword).ok()?;
2072 return Some(CpythonDiagnostic::new(
2073 format!("cannot assign to {keyword}"),
2074 target_start,
2075 target_end,
2076 ));
2077 }
2078 }
2079 if is_simple_keyword_name(bytes, target_start, target_end) {
2080 if unparenthesized_comprehension(bytes, equal + 1) {
2082 return Some(CpythonDiagnostic::new(
2083 "invalid syntax. Maybe you meant '==' or ':=' instead of '='?".to_owned(),
2084 target_start,
2085 equal + 1,
2086 ));
2087 }
2088 return None;
2089 }
2090 Some(CpythonDiagnostic::new(
2091 "expression cannot contain assignment, perhaps you meant \"==\"?".to_owned(),
2092 target_start,
2093 equal,
2094 ))
2095}
2096
2097fn unparenthesized_comprehension(bytes: &[u8], mut index: usize) -> bool {
2098 index = next_non_horizontal_whitespace(bytes, index);
2099 if index >= bytes.len() {
2100 return false;
2101 }
2102 let start = index;
2103 let mut level = 0usize;
2104 while index < bytes.len() {
2105 match bytes[index] {
2106 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2107 b'(' | b'[' | b'{' => {
2108 level += 1;
2109 index += 1;
2110 }
2111 b')' | b']' | b'}' => {
2112 if level == 0 {
2113 return false;
2114 }
2115 level -= 1;
2116 index += 1;
2117 }
2118 b',' | b':' if level == 0 => return false,
2119 _ if level == 0 && index > start && starts_identifier(bytes, index, b"for") => {
2120 return true;
2121 }
2122 _ => index += 1,
2123 }
2124 }
2125 false
2126}
2127
2128fn invalid_call_star_expression_error(
2129 bytes: &[u8],
2130 arg_start: usize,
2131 index: usize,
2132) -> Option<CpythonDiagnostic> {
2133 let start = next_non_horizontal_whitespace(bytes, arg_start);
2134 if start != index || bytes.get(index) != Some(&b'*') {
2135 return None;
2136 }
2137 let after_star = next_non_horizontal_whitespace(bytes, index + 1);
2138 if matches!(bytes.get(after_star), None | Some(b',' | b')' | b':')) {
2139 return Some(CpythonDiagnostic::new(
2140 "Invalid star expression".to_owned(),
2141 index,
2142 (index + 1).min(bytes.len()),
2143 ));
2144 }
2145 None
2146}
2147
2148fn invalid_call_argument_error(source: &str) -> Option<CpythonDiagnostic> {
2149 let bytes = source.as_bytes();
2150 let mut index = 0usize;
2151 let mut level = 0usize;
2152 let mut frames: Vec<CallArgFrame> = Vec::new();
2153 while index < bytes.len() {
2154 match bytes[index] {
2155 b'#' => {
2156 while index < bytes.len() && bytes[index] != b'\n' {
2157 index += 1;
2158 }
2159 }
2160 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2161 _ if starts_identifier(bytes, index, b"lambda") => {
2162 let params_start = index + 6;
2163 if let Some(params_end) = find_lambda_parameter_end(bytes, params_start) {
2164 index = params_end + 1;
2165 } else {
2166 index = params_start;
2167 }
2168 }
2169 b'(' => {
2170 level += 1;
2171 let in_call = opening_paren_is_call(bytes, index)
2172 || frames.last().is_some_and(|frame| frame.in_call);
2173 frames.push(CallArgFrame {
2174 level,
2175 arg_start: (in_call && !is_function_parameter_list(bytes, index))
2176 .then_some(index + 1),
2177 in_call,
2178 });
2179 index += 1;
2180 }
2181 b')' => {
2182 if matches!(frames.last(), Some(frame) if frame.level == level) {
2183 frames.pop();
2184 }
2185 level = level.saturating_sub(1);
2186 index += 1;
2187 }
2188 b'[' | b'{' => {
2189 level += 1;
2190 index += 1;
2191 }
2192 b']' | b'}' => {
2193 level = level.saturating_sub(1);
2194 index += 1;
2195 }
2196 b',' => {
2197 if let Some(frame) = frames.last_mut()
2198 && frame.level == level
2199 && frame.arg_start.is_some()
2200 {
2201 frame.arg_start = Some(index + 1);
2202 }
2203 index += 1;
2204 }
2205 b'*' => {
2206 if let Some(CallArgFrame {
2207 level: frame_level,
2208 arg_start: Some(arg_start),
2209 in_call: true,
2210 }) = frames.last().copied()
2211 && frame_level == level
2212 && let Some(error) = invalid_call_star_expression_error(bytes, arg_start, index)
2213 {
2214 return Some(error);
2215 }
2216 index += 1;
2217 }
2218 b'=' if is_plain_assignment_operator(bytes, index) => {
2219 if let Some(CallArgFrame {
2220 level: frame_level,
2221 arg_start: Some(arg_start),
2222 in_call: true,
2223 }) = frames.last().copied()
2224 && frame_level == level
2225 && let Some(error) =
2226 invalid_call_argument_assignment_error(source, arg_start, index)
2227 {
2228 return Some(error);
2229 }
2230 index += 1;
2231 }
2232 _ => index += 1,
2233 }
2234 }
2235 None
2236}
2237
2238fn top_level_colon(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
2239 let mut level = 0usize;
2240 while index < end {
2241 match bytes[index] {
2242 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2243 b'(' | b'[' | b'{' => {
2244 level += 1;
2245 index += 1;
2246 }
2247 b')' | b']' | b'}' => {
2248 level = level.saturating_sub(1);
2249 index += 1;
2250 }
2251 b':' if level == 0 => return Some(index),
2252 _ => index += 1,
2253 }
2254 }
2255 None
2256}
2257
2258fn expression_slice_is_valid(source: &str, start: usize, end: usize) -> bool {
2259 let bytes = source.as_bytes();
2260 let (start, end) = trim_target_range(bytes, start, end);
2261 start < end
2262 && parser::parse(&source[start..end], parser::Mode::Expression.into())
2263 .is_ok_and(|parsed| matches!(parsed.into_syntax(), ast::Mod::Expression(_)))
2264}
2265
2266fn invalid_dict_entry_error(
2267 source: &str,
2268 item_start: usize,
2269 item_end: usize,
2270 colon: Option<usize>,
2271 saw_dict_item: bool,
2272) -> Option<CpythonDiagnostic> {
2273 let bytes = source.as_bytes();
2274 let (item_start, item_end) = trim_target_range(bytes, item_start, item_end);
2275 if item_start >= item_end {
2276 return None;
2277 }
2278 if let Some(colon) = colon {
2279 let value_start = next_non_horizontal_whitespace(bytes, colon + 1);
2280 if value_start >= item_end {
2281 return Some(CpythonDiagnostic::new(
2282 "expression expected after dictionary key and ':'".to_owned(),
2283 colon,
2284 colon + 1,
2285 ));
2286 }
2287 if bytes.get(value_start) == Some(&b'*') {
2288 return Some(CpythonDiagnostic::new(
2289 "cannot use a starred expression in a dictionary value".to_owned(),
2290 value_start,
2291 value_start + 1,
2292 ));
2293 }
2294 if !expression_slice_is_valid(source, value_start, item_end) {
2295 return Some(CpythonDiagnostic::new(
2296 "invalid syntax".to_owned(),
2297 value_start,
2298 value_start,
2299 ));
2300 }
2301 } else if saw_dict_item {
2302 return Some(CpythonDiagnostic::new(
2303 "':' expected after dictionary key".to_owned(),
2304 item_end.saturating_sub(1),
2305 item_end,
2306 ));
2307 }
2308 None
2309}
2310
2311fn invalid_dict_literal_error(
2312 source: &str,
2313 open: usize,
2314 close: usize,
2315) -> Option<CpythonDiagnostic> {
2316 let bytes = source.as_bytes();
2317 let mut item_start = open + 1;
2318 let mut index = item_start;
2319 let mut level = 0usize;
2320 let mut saw_dict_item = false;
2321 let mut item_colon = None;
2322 while index <= close {
2323 if index == close || (level == 0 && bytes.get(index) == Some(&b',')) {
2324 if let Some(error) =
2325 invalid_dict_entry_error(source, item_start, index, item_colon, saw_dict_item)
2326 {
2327 return Some(error);
2328 }
2329 saw_dict_item |= item_colon.is_some();
2330 item_start = index + 1;
2331 item_colon = None;
2332 index += 1;
2333 continue;
2334 }
2335 match bytes[index] {
2336 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2337 b'(' | b'[' | b'{' => {
2338 level += 1;
2339 index += 1;
2340 }
2341 b')' | b']' | b'}' => {
2342 level = level.saturating_sub(1);
2343 index += 1;
2344 }
2345 b':' if level == 0 && item_colon.is_none() => {
2346 item_colon = Some(index);
2347 saw_dict_item = true;
2348 index += 1;
2349 }
2350 _ => index += 1,
2351 }
2352 }
2353 None
2354}
2355
2356fn invalid_dict_error(source: &str) -> Option<CpythonDiagnostic> {
2357 let bytes = source.as_bytes();
2358 let mut index = 0usize;
2359 while index < bytes.len() {
2360 match bytes[index] {
2361 b'#' => {
2362 while index < bytes.len() && bytes[index] != b'\n' {
2363 index += 1;
2364 }
2365 }
2366 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2367 b'{' => {
2368 let Some(close) = matching_delimiter(bytes, index, b'}') else {
2369 index += 1;
2370 continue;
2371 };
2372 if top_level_colon(bytes, index + 1, close).is_some()
2373 && let Some(error) = invalid_dict_literal_error(source, index, close)
2374 {
2375 return Some(error);
2376 }
2377 index = close + 1;
2378 }
2379 _ => index += 1,
2380 }
2381 }
2382 None
2383}
2384
2385fn collection_open_is_call(bytes: &[u8], open: usize) -> bool {
2386 if bytes.get(open) != Some(&b'(') {
2387 return false;
2388 }
2389 let mut cursor = open;
2390 while cursor > 0 && matches!(bytes.get(cursor - 1), Some(b' ' | b'\t' | b'\x0c')) {
2391 cursor -= 1;
2392 }
2393 matches!(
2394 cursor.checked_sub(1).and_then(|before| bytes.get(before)),
2395 Some(b')' | b']' | b'_' | b'a'..=b'z' | b'A'..=b'Z' | 0x80..=0xff)
2396 )
2397}
2398
2399fn invalid_collection_assignment_in_slice(
2400 source: &str,
2401 bytes: &[u8],
2402 start: usize,
2403 end: usize,
2404) -> Option<CpythonDiagnostic> {
2405 let mut item_start = start;
2406 let mut index = start;
2407 let mut level = 0usize;
2408 while index < end {
2409 match bytes[index] {
2410 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2411 b'(' | b'[' | b'{' => {
2412 level += 1;
2413 index += 1;
2414 }
2415 b')' | b']' | b'}' => {
2416 level = level.saturating_sub(1);
2417 index += 1;
2418 }
2419 b',' if level == 0 => {
2420 item_start = index + 1;
2421 index += 1;
2422 }
2423 b'=' if level == 0 && is_plain_assignment_operator(bytes, index) => {
2424 if top_level_colon(bytes, item_start, index).is_none() {
2425 let start = next_non_horizontal_whitespace(bytes, item_start);
2426 let target_end = trim_end_horizontal_whitespace(bytes, start, index);
2427 if start < target_end
2428 && let Some((expr_name, expr_start, expr_end, _)) =
2429 expression_name_and_range(&source[start..target_end])
2430 {
2431 if matches!(expr_name, "list" | "tuple") {
2432 return None;
2433 }
2434 if matches!(expr_name, "expression" | "attribute" | "subscript") {
2435 return Some(CpythonDiagnostic::new(
2436 format!(
2437 "cannot assign to {expr_name} here. Maybe you meant '==' instead of '='?"
2438 ),
2439 start + expr_start,
2440 start + expr_end,
2441 ));
2442 }
2443 }
2444 return Some(CpythonDiagnostic::new(
2445 "invalid syntax. Maybe you meant '==' or ':=' instead of '='?".to_owned(),
2446 start,
2447 index + 1,
2448 ));
2449 }
2450 index += 1;
2451 }
2452 _ => index += 1,
2453 }
2454 }
2455 None
2456}
2457
2458fn invalid_collection_assignment_error(source: &str) -> Option<CpythonDiagnostic> {
2459 let bytes = source.as_bytes();
2460 let mut index = 0usize;
2461 while index < bytes.len() {
2462 match bytes[index] {
2463 b'#' => {
2464 while index < bytes.len() && bytes[index] != b'\n' {
2465 index += 1;
2466 }
2467 }
2468 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2469 b'(' | b'[' | b'{' => {
2470 let close_byte = match bytes[index] {
2471 b'(' => b')',
2472 b'[' => b']',
2473 _ => b'}',
2474 };
2475 let Some(close) = matching_delimiter(bytes, index, close_byte) else {
2476 index += 1;
2477 continue;
2478 };
2479 if !collection_open_is_call(bytes, index)
2480 && let Some(error) =
2481 invalid_collection_assignment_in_slice(source, bytes, index + 1, close)
2482 {
2483 return Some(error);
2484 }
2485 index = close + 1;
2486 }
2487 _ => index += 1,
2488 }
2489 }
2490 None
2491}
2492
2493fn expression_assignment_error(source: &str) -> Option<CpythonDiagnostic> {
2494 let bytes = source.as_bytes();
2495 let mut index = 0;
2496 let mut paren_arg_starts: Vec<(Option<usize>, bool)> = Vec::new();
2497 while index < bytes.len() {
2498 match bytes[index] {
2499 b'#' => {
2500 while index < bytes.len() && bytes[index] != b'\n' {
2501 index += 1;
2502 }
2503 }
2504 b'\'' | b'"' => {
2505 index = skip_quoted_string(bytes, index);
2506 }
2507 _ if starts_identifier(bytes, index, b"lambda") => {
2508 let params_start = index + 6;
2509 if let Some(params_end) = find_lambda_parameter_end(bytes, params_start) {
2510 index = params_end + 1;
2511 } else {
2512 index = params_start;
2513 }
2514 }
2515 b'(' => {
2516 let in_call_context = opening_paren_is_call(bytes, index)
2517 || paren_arg_starts.last().is_some_and(|(_, in_call)| *in_call);
2518 paren_arg_starts.push((
2519 (!is_function_parameter_list(bytes, index)).then_some(index + 1),
2520 in_call_context,
2521 ));
2522 index += 1;
2523 }
2524 b')' => {
2525 paren_arg_starts.pop();
2526 index += 1;
2527 }
2528 b',' => {
2529 if let Some((start, _)) = paren_arg_starts.last_mut()
2530 && start.is_some()
2531 {
2532 *start = Some(index + 1);
2533 }
2534 index += 1;
2535 }
2536 b'=' if is_plain_assignment_operator(bytes, index) => {
2537 if let Some((Some(start), true)) = paren_arg_starts.last().copied()
2538 && !is_simple_keyword_name(bytes, start, index)
2539 {
2540 let mut expr_start = start;
2541 while matches!(bytes.get(expr_start), Some(b' ' | b'\t' | b'\x0c')) {
2542 expr_start += 1;
2543 }
2544 return Some(CpythonDiagnostic::new(
2545 "expression cannot contain assignment, perhaps you meant \"==\"?"
2546 .to_owned(),
2547 expr_start,
2548 index,
2549 ));
2550 }
2551 index += 1;
2552 }
2553 _ => index += 1,
2554 }
2555 }
2556 None
2557}
2558
2559fn invalid_named_expression_error(source: &str) -> Option<CpythonDiagnostic> {
2560 let bytes = source.as_bytes();
2561 let mut index = 0;
2562 while index + 1 < bytes.len() {
2563 match bytes[index] {
2564 b'#' => {
2565 while index < bytes.len() && bytes[index] != b'\n' {
2566 index += 1;
2567 }
2568 }
2569 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2570 b':' if bytes.get(index + 1) == Some(&b'=') => {
2571 let target_start = named_expression_target_start(bytes, index);
2572 let target_end = trim_end_horizontal_whitespace(bytes, target_start, index);
2573 if target_start < target_end
2574 && let Some((expr_name, start, end, is_name)) =
2575 expression_name_and_range(&source[target_start..target_end])
2576 && !is_name
2577 {
2578 return Some(CpythonDiagnostic::new(
2579 format!("cannot use assignment expressions with {expr_name}"),
2580 target_start + start,
2581 target_start + end,
2582 ));
2583 }
2584 index += 2;
2585 }
2586 _ => index += 1,
2587 }
2588 }
2589 None
2590}
2591
2592#[derive(Clone, Copy)]
2593struct AssignmentContext {
2594 start: usize,
2595 call: bool,
2596}
2597
2598fn invalid_plain_assignment_error(source: &str) -> Option<CpythonDiagnostic> {
2599 let bytes = source.as_bytes();
2600 let mut stack: Vec<AssignmentContext> = Vec::new();
2601 let mut index = 0;
2602 while index < bytes.len() {
2603 match bytes[index] {
2604 b'#' => {
2605 while index < bytes.len() && bytes[index] != b'\n' {
2606 index += 1;
2607 }
2608 }
2609 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
2610 b'(' | b'[' | b'{' => {
2611 stack.push(AssignmentContext {
2612 start: index + 1,
2613 call: bytes[index] == b'(' && opening_paren_is_call(bytes, index),
2614 });
2615 index += 1;
2616 }
2617 b')' | b']' | b'}' => {
2618 stack.pop();
2619 index += 1;
2620 }
2621 b',' => {
2622 if let Some(context) = stack.last_mut()
2623 && !context.call
2624 {
2625 context.start = index + 1;
2626 }
2627 index += 1;
2628 }
2629 b'=' if is_plain_assignment_operator(bytes, index) => {
2630 if let Some(context) = stack.last().copied()
2631 && !context.call
2632 {
2633 let target_start = skip_horizontal_whitespace(bytes, context.start);
2634 let target_end = trim_end_horizontal_whitespace(bytes, target_start, index);
2635 if target_start < target_end
2636 && let Some((expr_name, start, end, _)) =
2637 expression_name_and_range(&source[target_start..target_end])
2638 && matches!(expr_name, "expression" | "attribute" | "subscript")
2639 {
2640 return Some(CpythonDiagnostic::new(
2641 format!(
2642 "cannot assign to {expr_name} here. Maybe you meant '==' instead of '='?"
2643 ),
2644 target_start + start,
2645 target_start + end,
2646 ));
2647 }
2648 }
2649 index += 1;
2650 }
2651 _ => index += 1,
2652 }
2653 }
2654 None
2655}
2656
2657fn opening_paren_is_call(bytes: &[u8], paren: usize) -> bool {
2658 let mut cursor = paren;
2659 while cursor > 0 && matches!(bytes[cursor - 1], b' ' | b'\t' | b'\x0c') {
2660 cursor -= 1;
2661 }
2662 cursor > 0
2663 && (bytes[cursor - 1] >= 0x80
2664 || is_ascii_identifier_char(bytes[cursor - 1])
2665 || matches!(bytes[cursor - 1], b')' | b']'))
2666}
2667
2668fn named_expression_target_start(bytes: &[u8], walrus: usize) -> usize {
2669 let mut index = walrus;
2670 let mut level = 0usize;
2671 while index > 0 {
2672 index -= 1;
2673 match bytes[index] {
2674 b')' | b']' | b'}' => level += 1,
2675 b'(' | b'[' | b'{' if level > 0 => level -= 1,
2676 b'(' | b'[' | b'{' if level == 0 => return index + 1,
2677 b',' | b'\n' | b';' if level == 0 => return index + 1,
2678 _ => {}
2679 }
2680 }
2681 0
2682}
2683
2684fn trim_end_horizontal_whitespace(bytes: &[u8], start: usize, mut end: usize) -> usize {
2685 while end > start && matches!(bytes[end - 1], b' ' | b'\t' | b'\x0c') {
2686 end -= 1;
2687 }
2688 end
2689}
2690
2691fn annotation_target_error_for_slice(
2692 source: &str,
2693 start: usize,
2694 colon: usize,
2695) -> Option<CpythonDiagnostic> {
2696 let bytes = source.as_bytes();
2697 let (target_start, target_end) = trim_target_range(bytes, start, colon);
2698 if target_start >= target_end {
2699 return None;
2700 }
2701 let target_text = &source[target_start..target_end];
2702 let Ok(parsed) = parser::parse(target_text, parser::Mode::Expression.into()) else {
2703 return None;
2704 };
2705 let ast::Mod::Expression(expression) = parsed.into_syntax() else {
2706 return None;
2707 };
2708 match expression.body.as_ref() {
2709 ast::Expr::Name(_) | ast::Expr::Attribute(_) | ast::Expr::Subscript(_) => None,
2710 ast::Expr::List(_) => Some(CpythonDiagnostic::new(
2711 "only single target (not list) can be annotated".to_owned(),
2712 target_start,
2713 target_end,
2714 )),
2715 ast::Expr::Tuple(_) => Some(CpythonDiagnostic::new(
2716 "only single target (not tuple) can be annotated".to_owned(),
2717 target_start,
2718 target_end,
2719 )),
2720 _ => Some(CpythonDiagnostic::new(
2721 "illegal target for annotation".to_owned(),
2722 target_start,
2723 target_end,
2724 )),
2725 }
2726}
2727
2728fn invalid_annotation_line_start(bytes: &[u8], line_start: usize) -> bool {
2729 let column = skip_horizontal_whitespace(bytes, line_start);
2730 for keyword in [
2731 b"async".as_slice(),
2732 b"case",
2733 b"class",
2734 b"def",
2735 b"elif",
2736 b"else",
2737 b"except",
2738 b"finally",
2739 b"for",
2740 b"if",
2741 b"match",
2742 b"try",
2743 b"while",
2744 b"with",
2745 ] {
2746 if starts_identifier(bytes, column, keyword) {
2747 return false;
2748 }
2749 }
2750 true
2751}
2752
2753fn invalid_annotation_target_error(source: &str) -> Option<CpythonDiagnostic> {
2754 let bytes = source.as_bytes();
2755 let mut line_start = 0usize;
2756 for line in source.split_inclusive('\n') {
2757 let line_end = line_start + line.len();
2758 if invalid_annotation_line_start(bytes, line_start)
2759 && let Some(colon) = find_byte_at_level(bytes, line_start, line_end, b':')
2760 && bytes.get(colon + 1) != Some(&b'=')
2761 && colon.checked_sub(1).and_then(|before| bytes.get(before)) != Some(&b':')
2762 && let Some(error) = annotation_target_error_for_slice(source, line_start, colon)
2763 {
2764 return Some(error);
2765 }
2766 line_start = line_end;
2767 }
2768 None
2769}
2770
2771fn statement_target_end(bytes: &[u8], mut index: usize) -> usize {
2772 let mut level = 0usize;
2773 while index < bytes.len() {
2774 match bytes[index] {
2775 b'#' if level == 0 => return index,
2776 b'\n' | b';' if level == 0 => return index,
2777 b'\'' | b'"' => {
2778 index = skip_quoted_string(bytes, index);
2779 }
2780 b'(' | b'[' | b'{' => {
2781 level += 1;
2782 index += 1;
2783 }
2784 b')' | b']' | b'}' => {
2785 level = level.saturating_sub(1);
2786 index += 1;
2787 }
2788 _ => index += 1,
2789 }
2790 }
2791 index
2792}
2793
2794fn invalid_assignment_target(expression: &ast::Expr) -> Option<&ast::Expr> {
2795 match expression {
2796 ast::Expr::List(ast::ExprList { elts, .. })
2797 | ast::Expr::Tuple(ast::ExprTuple { elts, .. }) => {
2798 elts.iter().find_map(invalid_assignment_target)
2799 }
2800 ast::Expr::Starred(ast::ExprStarred { value, .. }) => invalid_assignment_target(value),
2801 ast::Expr::Name(_) | ast::Expr::Subscript(_) | ast::Expr::Attribute(_) => None,
2802 _ => Some(expression),
2803 }
2804}
2805
2806fn invalid_for_target(expression: &ast::Expr) -> Option<&ast::Expr> {
2807 match expression {
2808 ast::Expr::List(ast::ExprList { elts, .. })
2809 | ast::Expr::Tuple(ast::ExprTuple { elts, .. }) => elts.iter().find_map(invalid_for_target),
2810 ast::Expr::Starred(ast::ExprStarred { value, .. }) => invalid_for_target(value),
2811 ast::Expr::Compare(ast::ExprCompare { left, ops, .. }) => {
2812 if matches!(ops.first(), Some(ast::CmpOp::In)) {
2813 invalid_for_target(left)
2814 } else {
2815 None
2816 }
2817 }
2818 ast::Expr::Name(_) | ast::Expr::Subscript(_) | ast::Expr::Attribute(_) => None,
2819 _ => Some(expression),
2820 }
2821}
2822
2823fn invalid_delete_target(expression: &ast::Expr) -> Option<&ast::Expr> {
2824 match expression {
2825 ast::Expr::List(ast::ExprList { elts, .. })
2826 | ast::Expr::Tuple(ast::ExprTuple { elts, .. }) => {
2827 elts.iter().find_map(invalid_delete_target)
2828 }
2829 ast::Expr::Name(_) | ast::Expr::Subscript(_) | ast::Expr::Attribute(_) => None,
2830 ast::Expr::Starred(_) => Some(expression),
2831 ast::Expr::Compare(_) => Some(expression),
2832 _ => Some(expression),
2833 }
2834}
2835
2836fn delete_target_expr_name(expression: &ast::Expr) -> &'static str {
2837 match expression {
2838 ast::Expr::Attribute(_) => "attribute",
2839 ast::Expr::Subscript(_) => "subscript",
2840 ast::Expr::Starred(_) => "starred",
2841 ast::Expr::Name(_) => "name",
2842 ast::Expr::List(_) => "list",
2843 ast::Expr::Tuple(_) => "tuple",
2844 ast::Expr::Lambda(_) => "lambda",
2845 ast::Expr::Call(_) => "function call",
2846 ast::Expr::BoolOp(_) | ast::Expr::BinOp(_) | ast::Expr::UnaryOp(_) => "expression",
2847 ast::Expr::Generator(_) => "generator expression",
2848 ast::Expr::Yield(_) | ast::Expr::YieldFrom(_) => "yield expression",
2849 ast::Expr::Await(_) => "await expression",
2850 ast::Expr::ListComp(_) => "list comprehension",
2851 ast::Expr::SetComp(_) => "set comprehension",
2852 ast::Expr::DictComp(_) => "dict comprehension",
2853 ast::Expr::Dict(_) => "dict literal",
2854 ast::Expr::Set(_) => "set display",
2855 ast::Expr::FString(_) => "f-string expression",
2856 ast::Expr::TString(_) => "t-string expression",
2857 ast::Expr::NumberLiteral(_) | ast::Expr::StringLiteral(_) | ast::Expr::BytesLiteral(_) => {
2858 "literal"
2859 }
2860 ast::Expr::Constant(expr) => match &expr.value {
2861 ast::ConstantValue::None => "None",
2862 ast::ConstantValue::Boolean(true) => "True",
2863 ast::ConstantValue::Boolean(false) => "False",
2864 ast::ConstantValue::Ellipsis => "ellipsis",
2865 ast::ConstantValue::Tuple(_) => "tuple",
2866 ast::ConstantValue::Frozenset(_) => "literal",
2867 ast::ConstantValue::Str(_)
2868 | ast::ConstantValue::Bytes(_)
2869 | ast::ConstantValue::Integer(_)
2870 | ast::ConstantValue::Float(_)
2871 | ast::ConstantValue::Complex { .. } => "literal",
2872 },
2873 ast::Expr::BooleanLiteral(boolean) => {
2874 if boolean.value {
2875 "True"
2876 } else {
2877 "False"
2878 }
2879 }
2880 ast::Expr::NoneLiteral(_) => "None",
2881 ast::Expr::EllipsisLiteral(_) => "ellipsis",
2882 ast::Expr::Compare(_) => "comparison",
2883 ast::Expr::If(_) => "conditional expression",
2884 ast::Expr::Named(_) => "named expression",
2885 ast::Expr::Slice(_) | ast::Expr::IpyEscapeCommand(_) => "expression",
2886 }
2887}
2888
2889fn parenthesized_single_starred_delete_target(bytes: &[u8], start: usize, end: usize) -> bool {
2890 let mut cursor = start;
2891 while matches!(bytes.get(cursor), Some(b' ' | b'\t' | b'\x0c')) {
2892 cursor += 1;
2893 }
2894 if bytes.get(cursor) != Some(&b'(') {
2895 return false;
2896 }
2897 cursor += 1;
2898 while matches!(bytes.get(cursor), Some(b' ' | b'\t' | b'\x0c')) {
2899 cursor += 1;
2900 }
2901 if bytes.get(cursor) != Some(&b'*') {
2902 return false;
2903 }
2904 let mut level = 1usize;
2905 cursor += 1;
2906 while cursor < end {
2907 match bytes[cursor] {
2908 b'\'' | b'"' => {
2909 cursor = skip_quoted_string(bytes, cursor);
2910 }
2911 b'(' | b'[' | b'{' => {
2912 level += 1;
2913 cursor += 1;
2914 }
2915 b')' => {
2916 level = level.saturating_sub(1);
2917 if level == 0 {
2918 cursor += 1;
2919 while matches!(bytes.get(cursor), Some(b' ' | b'\t' | b'\x0c')) {
2920 cursor += 1;
2921 }
2922 return cursor == end;
2923 }
2924 cursor += 1;
2925 }
2926 b',' if level == 1 => return false,
2927 b']' | b'}' => {
2928 level = level.saturating_sub(1);
2929 cursor += 1;
2930 }
2931 _ => cursor += 1,
2932 }
2933 }
2934 false
2935}
2936
2937fn assignment_target_expr_range(source: &str, start: usize, end: usize) -> Option<(usize, usize)> {
2938 let bytes = source.as_bytes();
2939 let (target_start, target_end) = trim_target_range(bytes, start, end);
2940 if target_start >= target_end {
2941 return None;
2942 }
2943 if parser::parse(
2944 &source[target_start..target_end],
2945 parser::Mode::Expression.into(),
2946 )
2947 .is_ok()
2948 {
2949 return Some((target_start, target_end));
2950 }
2951 let colon = top_level_colon(bytes, target_start, target_end)?;
2956 let after = skip_horizontal_whitespace(bytes, colon + 1);
2957 let yield_at = skip_opening_parentheses(bytes, after, target_end);
2958 if yield_at >= target_end || !starts_identifier(bytes, yield_at, b"yield") {
2959 return None;
2960 }
2961 Some((after, target_end))
2962}
2963
2964fn skip_opening_parentheses(bytes: &[u8], mut index: usize, end: usize) -> usize {
2965 loop {
2966 index = skip_horizontal_whitespace(bytes, index);
2967 if index >= end || bytes[index] != b'(' {
2968 return index;
2969 }
2970 index += 1;
2971 }
2972}
2973
2974fn trim_target_range(bytes: &[u8], mut start: usize, mut end: usize) -> (usize, usize) {
2975 while start < end
2976 && matches!(
2977 bytes.get(start),
2978 Some(b' ' | b'\t' | b'\n' | b'\r' | b'\x0c')
2979 )
2980 {
2981 start += 1;
2982 }
2983 while end > start
2984 && matches!(
2985 bytes.get(end - 1),
2986 Some(b' ' | b'\t' | b'\n' | b'\r' | b'\x0c')
2987 )
2988 {
2989 end -= 1;
2990 }
2991 (start, end)
2992}
2993
2994fn invalid_assignment_message(name: &'static str, top_level_bitwise: bool) -> String {
2995 if top_level_bitwise {
2996 format!("cannot assign to {name} here. Maybe you meant '==' instead of '='?")
2997 } else {
2998 format!("cannot assign to {name}")
2999 }
3000}
3001
3002fn assignment_target_error_for_slice(
3003 source: &str,
3004 start: usize,
3005 end: usize,
3006) -> Option<CpythonDiagnostic> {
3007 let bytes = source.as_bytes();
3008 let (target_start, target_end) = assignment_target_expr_range(source, start, end)?;
3009 if starts_identifier(bytes, target_start, b"yield") {
3010 return Some(CpythonDiagnostic::new(
3011 "assignment to yield expression not possible".to_owned(),
3012 target_start,
3013 target_start + 5,
3014 ));
3015 }
3016 let target_text = &source[target_start..target_end];
3017 let Ok(parsed) = parser::parse(target_text, parser::Mode::Expression.into()) else {
3018 return None;
3019 };
3020 let ast::Mod::Expression(expression) = parsed.into_syntax() else {
3021 return None;
3022 };
3023 let invalid_target = invalid_assignment_target(&expression.body)?;
3024 let invalid_start = target_start + invalid_target.range().start().to_usize();
3025 let invalid_end = target_start + invalid_target.range().end().to_usize();
3026 let name = delete_target_expr_name(invalid_target);
3027 let top_level = invalid_target.range() == expression.body.range();
3028 let bitwise_like = matches!(
3029 invalid_target,
3030 ast::Expr::Call(_)
3031 | ast::Expr::BoolOp(_)
3032 | ast::Expr::BinOp(_)
3033 | ast::Expr::UnaryOp(_)
3034 | ast::Expr::NumberLiteral(_)
3035 | ast::Expr::StringLiteral(_)
3036 | ast::Expr::BytesLiteral(_)
3037 | ast::Expr::EllipsisLiteral(_)
3038 | ast::Expr::Yield(_)
3039 | ast::Expr::YieldFrom(_)
3040 | ast::Expr::Set(_)
3041 | ast::Expr::Dict(_)
3042 | ast::Expr::FString(_)
3043 | ast::Expr::TString(_)
3044 );
3045 Some(CpythonDiagnostic::new(
3046 invalid_assignment_message(name, top_level && bitwise_like),
3047 invalid_start,
3048 invalid_end,
3049 ))
3050}
3051
3052fn star_target_error_for_slice(
3053 source: &str,
3054 start: usize,
3055 end: usize,
3056) -> Option<CpythonDiagnostic> {
3057 invalid_target_error_for_slice(source, start, end, invalid_assignment_target)
3058}
3059
3060fn for_target_error_for_slice(source: &str, start: usize, end: usize) -> Option<CpythonDiagnostic> {
3061 invalid_target_error_for_slice(source, start, end, invalid_for_target)
3062}
3063
3064fn invalid_target_error_for_slice(
3065 source: &str,
3066 start: usize,
3067 end: usize,
3068 invalid_target: for<'a> fn(&'a ast::Expr) -> Option<&'a ast::Expr>,
3069) -> Option<CpythonDiagnostic> {
3070 let bytes = source.as_bytes();
3071 let (target_start, target_end) = trim_target_range(bytes, start, end);
3072 if target_start >= target_end {
3073 return None;
3074 }
3075 let target_text = &source[target_start..target_end];
3076 let Ok(parsed) = parser::parse(target_text, parser::Mode::Expression.into()) else {
3077 return None;
3078 };
3079 let ast::Mod::Expression(expression) = parsed.into_syntax() else {
3080 return None;
3081 };
3082 let invalid_target = invalid_target(&expression.body)?;
3083 let name = delete_target_expr_name(invalid_target);
3084 let invalid_start = target_start + invalid_target.range().start().to_usize();
3085 let invalid_end = target_start + invalid_target.range().end().to_usize();
3086 Some(CpythonDiagnostic::new(
3087 format!("cannot assign to {name}"),
3088 invalid_start,
3089 invalid_end,
3090 ))
3091}
3092
3093fn first_compare_operator_at_level(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
3094 let mut level = 0usize;
3095 while index < end {
3096 match bytes[index] {
3097 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3098 b'(' | b'[' | b'{' => {
3099 level += 1;
3100 index += 1;
3101 }
3102 b')' | b']' | b'}' => {
3103 level = level.saturating_sub(1);
3104 index += 1;
3105 }
3106 b'<' | b'>' if level == 0 => return Some(index),
3107 b'=' if level == 0 && bytes.get(index + 1) == Some(&b'=') => return Some(index),
3108 b'!' if level == 0 && bytes.get(index + 1) == Some(&b'=') => return Some(index),
3109 _ if level == 0 && starts_identifier(bytes, index, b"is") => return Some(index),
3110 _ if level == 0 && starts_identifier(bytes, index, b"not") => return Some(index),
3111 _ => index += 1,
3112 }
3113 }
3114 None
3115}
3116
3117fn non_in_compare_for_target_error(
3118 source: &str,
3119 start: usize,
3120 end: usize,
3121) -> Option<CpythonDiagnostic> {
3122 let bytes = source.as_bytes();
3123 let (target_start, target_end) = trim_target_range(bytes, start, end);
3124 if target_start >= target_end {
3125 return None;
3126 }
3127 let target_text = &source[target_start..target_end];
3128 let Ok(parsed) = parser::parse(target_text, parser::Mode::Expression.into()) else {
3129 return None;
3130 };
3131 let ast::Mod::Expression(expression) = parsed.into_syntax() else {
3132 return None;
3133 };
3134 let ast::Expr::Compare(ast::ExprCompare { ops, .. }) = expression.body.as_ref() else {
3135 return None;
3136 };
3137 if matches!(ops.first(), Some(ast::CmpOp::In)) {
3138 return None;
3139 }
3140 let operator = first_compare_operator_at_level(bytes, target_start, target_end)?;
3141 Some(CpythonDiagnostic::new(
3142 "invalid syntax".to_owned(),
3143 operator,
3144 (operator + 1).min(target_end),
3145 ))
3146}
3147
3148fn top_level_plain_assignment_offsets(bytes: &[u8]) -> Vec<usize> {
3149 let mut offsets = Vec::new();
3150 let mut index = 0usize;
3151 let mut level = 0usize;
3152 while index < bytes.len() {
3153 match bytes[index] {
3154 b'#' if level == 0 => {
3155 while index < bytes.len() && bytes[index] != b'\n' {
3156 index += 1;
3157 }
3158 }
3159 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3160 b'(' | b'[' | b'{' => {
3161 level += 1;
3162 index += 1;
3163 }
3164 b')' | b']' | b'}' => {
3165 level = level.saturating_sub(1);
3166 index += 1;
3167 }
3168 b'=' if level == 0 && is_plain_assignment_operator(bytes, index) => {
3169 offsets.push(index);
3170 index += 1;
3171 }
3172 _ => index += 1,
3173 }
3174 }
3175 offsets
3176}
3177
3178fn invalid_condition_assignment_error(
3179 source: &str,
3180 parse_error_offset: usize,
3181) -> Option<CpythonDiagnostic> {
3182 let bytes = source.as_bytes();
3183 let mut index = 0usize;
3184 let mut line_start = 0usize;
3185 let mut level = 0usize;
3186 while index < bytes.len() {
3187 match bytes[index] {
3188 b'#' if level == 0 => {
3189 while index < bytes.len() && bytes[index] != b'\n' {
3190 index += 1;
3191 }
3192 }
3193 b'\n' => {
3194 line_start = index + 1;
3195 index += 1;
3196 }
3197 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3198 b'(' | b'[' | b'{' => {
3199 level += 1;
3200 index += 1;
3201 }
3202 b')' | b']' | b'}' => {
3203 level = level.saturating_sub(1);
3204 index += 1;
3205 }
3206 _ if level == 0
3207 && index == skip_horizontal_whitespace(bytes, line_start)
3208 && (starts_identifier(bytes, index, b"if")
3209 || starts_identifier(bytes, index, b"elif")
3210 || starts_identifier(bytes, index, b"while")) =>
3211 {
3212 let keyword_len = if starts_identifier(bytes, index, b"while") {
3213 5
3214 } else if starts_identifier(bytes, index, b"elif") {
3215 4
3216 } else {
3217 2
3218 };
3219 let cond_start = skip_horizontal_whitespace(bytes, index + keyword_len);
3220 let Some(colon) = top_level_colon(bytes, cond_start, bytes.len()) else {
3221 index += keyword_len;
3222 continue;
3223 };
3224 if parse_error_offset < index || parse_error_offset > colon {
3225 index += keyword_len;
3226 continue;
3227 }
3228 let Some(equal) = condition_plain_assignment(bytes, cond_start, colon) else {
3229 index += keyword_len;
3230 continue;
3231 };
3232 let (target_start, target_end) = trim_target_range(bytes, cond_start, equal);
3233 if target_start >= target_end {
3234 index += keyword_len;
3235 continue;
3236 }
3237 if is_simple_keyword_name(bytes, target_start, target_end) {
3238 return Some(CpythonDiagnostic::new(
3239 "invalid syntax. Maybe you meant '==' or ':=' instead of '='?".to_owned(),
3240 target_start,
3241 equal + 1,
3242 ));
3243 }
3244 if let Some((expr_name, start, end, _)) =
3245 expression_name_and_range(&source[target_start..target_end])
3246 {
3247 return Some(CpythonDiagnostic::new(
3248 format!(
3249 "cannot assign to {expr_name} here. Maybe you meant '==' instead of '='?"
3250 ),
3251 target_start + start,
3252 target_start + end,
3253 ));
3254 }
3255 return Some(CpythonDiagnostic::new(
3256 "invalid syntax. Maybe you meant '==' or ':=' instead of '='?".to_owned(),
3257 target_start,
3258 equal + 1,
3259 ));
3260 }
3261 _ => index += 1,
3262 }
3263 }
3264 None
3265}
3266
3267fn condition_plain_assignment(bytes: &[u8], start: usize, end: usize) -> Option<usize> {
3268 let mut index = start;
3269 let mut nest = Vec::new();
3270 while index < end {
3271 match bytes[index] {
3272 b'#' => {
3273 while index < end && bytes[index] != b'\n' {
3274 index += 1;
3275 }
3276 }
3277 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3278 b'(' => {
3279 nest.push(if is_call_open(bytes, start, index) {
3280 b'c'
3281 } else {
3282 b'g'
3283 });
3284 index += 1;
3285 }
3286 b'[' => {
3287 nest.push(b'[');
3288 index += 1;
3289 }
3290 b'{' => {
3291 nest.push(b'{');
3292 index += 1;
3293 }
3294 b')' | b']' | b'}' => {
3295 nest.pop();
3296 index += 1;
3297 }
3298 b':' if nest.last() == Some(&b'l') => {
3299 nest.pop();
3300 index += 1;
3301 }
3302 b'=' if is_plain_assignment_operator(bytes, index)
3303 && !nest.contains(&b'c')
3304 && !nest.contains(&b'l') =>
3305 {
3306 return Some(index);
3307 }
3308 _ if starts_identifier(bytes, index, b"lambda") => {
3309 nest.push(b'l');
3310 index += 6;
3311 }
3312 _ => index += 1,
3313 }
3314 }
3315 None
3316}
3317
3318fn is_call_open(bytes: &[u8], start: usize, open: usize) -> bool {
3319 let mut index = open;
3320 while index > start {
3321 index -= 1;
3322 match bytes[index] {
3323 b' ' | b'\t' | b'\n' | b'\r' | b'\x0c' => {}
3324 b')' | b']' => return true,
3325 byte if byte >= 0x80 || is_ascii_identifier_char(byte) => {
3326 let mut ident_start = index;
3327 while ident_start > start
3328 && bytes
3329 .get(ident_start - 1)
3330 .is_some_and(|b| *b >= 0x80 || is_ascii_identifier_char(*b))
3331 {
3332 ident_start -= 1;
3333 }
3334 return !is_condition_keyword(&bytes[ident_start..=index]);
3335 }
3336 _ => return false,
3337 }
3338 }
3339 false
3340}
3341
3342fn is_condition_keyword(word: &[u8]) -> bool {
3343 matches!(
3344 word,
3345 b"not" | b"and" | b"or" | b"in" | b"is" | b"if" | b"elif" | b"while" | b"await" | b"lambda"
3346 )
3347}
3348
3349fn invalid_assignment_target_error(source: &str) -> Option<CpythonDiagnostic> {
3350 let bytes = source.as_bytes();
3351 let offsets = top_level_plain_assignment_offsets(bytes);
3352 if offsets.is_empty() {
3353 return None;
3354 }
3355 let mut start = 0usize;
3356 for offset in offsets {
3357 if let Some(error) = assignment_target_error_for_slice(source, start, offset) {
3358 return Some(error);
3359 }
3360 start = offset + 1;
3361 }
3362 None
3363}
3364
3365fn top_level_augassign_offset(bytes: &[u8]) -> Option<(usize, usize)> {
3366 let mut index = 0usize;
3367 let mut level = 0usize;
3368 while index < bytes.len() {
3369 match bytes[index] {
3370 b'#' if level == 0 => {
3371 while index < bytes.len() && bytes[index] != b'\n' {
3372 index += 1;
3373 }
3374 }
3375 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3376 b'(' | b'[' | b'{' => {
3377 level += 1;
3378 index += 1;
3379 }
3380 b')' | b']' | b'}' => {
3381 level = level.saturating_sub(1);
3382 index += 1;
3383 }
3384 b'+' | b'-' | b'*' | b'@' | b'/' | b'%' | b'&' | b'|' | b'^'
3385 if level == 0 && bytes.get(index + 1) == Some(&b'=') =>
3386 {
3387 return Some((index, 2));
3388 }
3389 b'<' | b'>'
3390 if level == 0
3391 && bytes.get(index + 1) == Some(&bytes[index])
3392 && bytes.get(index + 2) == Some(&b'=') =>
3393 {
3394 return Some((index, 3));
3395 }
3396 b'*' if level == 0
3397 && bytes.get(index + 1) == Some(&b'*')
3398 && bytes.get(index + 2) == Some(&b'=') =>
3399 {
3400 return Some((index, 3));
3401 }
3402 b'/' if level == 0
3403 && bytes.get(index + 1) == Some(&b'/')
3404 && bytes.get(index + 2) == Some(&b'=') =>
3405 {
3406 return Some((index, 3));
3407 }
3408 _ => index += 1,
3409 }
3410 }
3411 None
3412}
3413
3414fn invalid_augassign_target_error(source: &str) -> Option<CpythonDiagnostic> {
3415 let bytes = source.as_bytes();
3416 let (operator, _) = top_level_augassign_offset(bytes)?;
3417 let (target_start, target_end) = assignment_target_expr_range(source, 0, operator)?;
3418 let target_text = &source[target_start..target_end];
3419 let Ok(parsed) = parser::parse(target_text, parser::Mode::Expression.into()) else {
3420 return None;
3421 };
3422 let ast::Mod::Expression(expression) = parsed.into_syntax() else {
3423 return None;
3424 };
3425 let name = delete_target_expr_name(&expression.body);
3426 Some(CpythonDiagnostic::new(
3427 format!("'{name}' is an illegal expression for augmented assignment"),
3428 target_start,
3429 target_end,
3430 ))
3431}
3432
3433fn find_for_target_delimiter(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
3434 let mut level = 0usize;
3435 while index < end {
3436 match bytes[index] {
3437 b'#' if level == 0 => return None,
3438 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3439 b'(' | b'[' | b'{' => {
3440 level += 1;
3441 index += 1;
3442 }
3443 b')' | b']' | b'}' => {
3444 level = level.saturating_sub(1);
3445 index += 1;
3446 }
3447 _ if level == 0 && starts_identifier(bytes, index, b"in") => return Some(index),
3448 b':' if level == 0 => return Some(index),
3449 _ => index += 1,
3450 }
3451 }
3452 None
3453}
3454
3455fn invalid_for_target_error(source: &str) -> Option<CpythonDiagnostic> {
3456 let bytes = source.as_bytes();
3457 let mut index = 0usize;
3458 while index < bytes.len() {
3459 match bytes[index] {
3460 b'#' => {
3461 while index < bytes.len() && bytes[index] != b'\n' {
3462 index += 1;
3463 }
3464 }
3465 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3466 _ if starts_identifier(bytes, index, b"for") => {
3467 let target_start = skip_horizontal_whitespace(bytes, index + 3);
3468 let line_end = source[index..]
3469 .find('\n')
3470 .map_or(bytes.len(), |newline| index + newline);
3471 if let Some(target_end) = find_for_target_delimiter(bytes, target_start, line_end) {
3472 if let Some(error) =
3473 for_target_error_for_slice(source, target_start, target_end)
3474 {
3475 return Some(error);
3476 }
3477 if let Some(error) =
3478 non_in_compare_for_target_error(source, target_start, target_end)
3479 {
3480 return Some(error);
3481 }
3482 }
3483 index = target_start.max(index + 3);
3484 }
3485 _ => index += 1,
3486 }
3487 }
3488 None
3489}
3490
3491fn find_with_target_delimiter(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
3492 let mut level = 0usize;
3493 while index < end {
3494 match bytes[index] {
3495 b'#' if level == 0 => return None,
3496 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3497 b'(' | b'[' | b'{' => {
3498 level += 1;
3499 index += 1;
3500 }
3501 b')' if level == 0 => return Some(index),
3502 b')' | b']' | b'}' => {
3503 level = level.saturating_sub(1);
3504 index += 1;
3505 }
3506 b',' | b':' if level == 0 => return Some(index),
3507 _ => index += 1,
3508 }
3509 }
3510 None
3511}
3512
3513fn invalid_with_target_error(source: &str) -> Option<CpythonDiagnostic> {
3514 let bytes = source.as_bytes();
3515 let mut line_start = 0usize;
3516 for line in source.split_inclusive('\n') {
3517 let line_end = line_start + line.len();
3518 let mut column = skip_horizontal_whitespace(bytes, line_start);
3519 if starts_identifier(bytes, column, b"async") {
3520 column = skip_horizontal_whitespace(bytes, column + 5);
3521 }
3522 if !starts_identifier(bytes, column, b"with") {
3523 line_start = line_end;
3524 continue;
3525 }
3526 let mut index = column + 4;
3527 while let Some(as_index) = find_keyword_at_level(bytes, index, line_end, b"as") {
3528 let target_start = skip_horizontal_whitespace(bytes, as_index + 2);
3529 if let Some(target_end) = find_with_target_delimiter(bytes, target_start, line_end) {
3530 if let Some(error) = star_target_error_for_slice(source, target_start, target_end) {
3531 return Some(error);
3532 }
3533 index = target_end.saturating_add(1);
3534 } else {
3535 break;
3536 }
3537 }
3538 line_start = line_end;
3539 }
3540 None
3541}
3542
3543fn find_missing_in_if_keyword(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
3544 let mut level = 0usize;
3545 while index < end {
3546 match bytes[index] {
3547 b'#' if level == 0 => return None,
3548 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3549 b'(' | b'[' | b'{' => {
3550 level += 1;
3551 index += 1;
3552 }
3553 b')' | b']' | b'}' if level == 0 => return None,
3554 b')' | b']' | b'}' => {
3555 level = level.saturating_sub(1);
3556 index += 1;
3557 }
3558 _ if level == 0 && starts_identifier(bytes, index, b"in") => return None,
3559 _ if level == 0 && starts_identifier(bytes, index, b"if") => return Some(index),
3560 _ => index += 1,
3561 }
3562 }
3563 None
3564}
3565
3566fn invalid_for_if_clause_error(source: &str) -> Option<CpythonDiagnostic> {
3567 let bytes = source.as_bytes();
3568 let mut index = 0usize;
3569 let mut level = 0usize;
3570 while index < bytes.len() {
3571 match bytes[index] {
3572 b'#' if level == 0 => {
3573 while index < bytes.len() && bytes[index] != b'\n' {
3574 index += 1;
3575 }
3576 }
3577 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3578 b'(' | b'[' | b'{' => {
3579 level += 1;
3580 index += 1;
3581 }
3582 b')' | b']' | b'}' => {
3583 level = level.saturating_sub(1);
3584 index += 1;
3585 }
3586 _ if level > 0 && starts_identifier(bytes, index, b"for") => {
3587 let target_start = skip_horizontal_whitespace(bytes, index + 3);
3588 let line_end = source[index..]
3589 .find('\n')
3590 .map_or(bytes.len(), |newline| index + newline);
3591 if let Some(if_index) = find_missing_in_if_keyword(bytes, target_start, line_end) {
3592 return Some(CpythonDiagnostic::new(
3593 "'in' expected after for-loop variables".to_owned(),
3594 if_index,
3595 (if_index + 2).min(line_end),
3596 ));
3597 }
3598 index = target_start.max(index + 3);
3599 }
3600 _ => index += 1,
3601 }
3602 }
3603 None
3604}
3605
3606fn invalid_delete_target_error(source: &str) -> Option<CpythonDiagnostic> {
3607 let bytes = source.as_bytes();
3608 let mut index = 0;
3609 while index < bytes.len() {
3610 match bytes[index] {
3611 b'#' => {
3612 while index < bytes.len() && bytes[index] != b'\n' {
3613 index += 1;
3614 }
3615 }
3616 b'\'' | b'"' => {
3617 index = skip_quoted_string(bytes, index);
3618 }
3619 b'd' if starts_identifier(bytes, index, b"del") => {
3620 let mut target_start = index + 3;
3621 if !matches!(bytes.get(target_start), Some(b' ' | b'\t' | b'\x0c')) {
3622 index += 3;
3623 continue;
3624 }
3625 while matches!(bytes.get(target_start), Some(b' ' | b'\t' | b'\x0c')) {
3626 target_start += 1;
3627 }
3628 let mut target_end = statement_target_end(bytes, target_start);
3629 while target_end > target_start
3630 && matches!(bytes.get(target_end - 1), Some(b' ' | b'\t' | b'\x0c'))
3631 {
3632 target_end -= 1;
3633 }
3634 if target_start >= target_end {
3635 index = target_end.max(index + 3);
3636 continue;
3637 }
3638 if parenthesized_single_starred_delete_target(bytes, target_start, target_end) {
3639 return Some(CpythonDiagnostic::new(
3640 "cannot use starred expression here".to_owned(),
3641 target_start,
3642 target_end,
3643 ));
3644 }
3645 if bytes.get(target_start) == Some(&b'*') {
3646 return Some(CpythonDiagnostic::new(
3647 "cannot delete starred".to_owned(),
3648 target_start,
3649 (target_start + 1).min(target_end),
3650 ));
3651 }
3652 let target_text = &source[target_start..target_end];
3653 let Ok(parsed) = parser::parse(target_text, parser::Mode::Expression.into()) else {
3654 index = target_end;
3655 continue;
3656 };
3657 let ast::Mod::Expression(expression) = parsed.into_syntax() else {
3658 index = target_end;
3659 continue;
3660 };
3661 let Some(invalid_target) = invalid_delete_target(&expression.body) else {
3662 index = target_end;
3663 continue;
3664 };
3665 let start = target_start + invalid_target.range().start().to_usize();
3666 let end = target_start + invalid_target.range().end().to_usize();
3667 if matches!(invalid_target, ast::Expr::FString(_)) {
3668 return Some(CpythonDiagnostic::new(
3669 "invalid syntax".to_owned(),
3670 start,
3671 end,
3672 ));
3673 }
3674 let name = delete_target_expr_name(invalid_target);
3675 return Some(CpythonDiagnostic::new(
3676 format!("cannot delete {name}"),
3677 start,
3678 end,
3679 ));
3680 }
3681 _ => index += 1,
3682 }
3683 }
3684 None
3685}
3686
3687fn skip_horizontal_whitespace(bytes: &[u8], mut index: usize) -> usize {
3688 while matches!(bytes.get(index), Some(b' ' | b'\t' | b'\x0c')) {
3689 index += 1;
3690 }
3691 index
3692}
3693
3694fn find_keyword_at_level(
3695 bytes: &[u8],
3696 mut index: usize,
3697 end: usize,
3698 keyword: &[u8],
3699) -> Option<usize> {
3700 let mut level = 0usize;
3701 while index < end {
3702 match bytes[index] {
3703 b'#' if level == 0 => return None,
3704 b'\'' | b'"' => {
3705 index = skip_quoted_string(bytes, index);
3706 }
3707 b'(' | b'[' | b'{' => {
3708 level += 1;
3709 index += 1;
3710 }
3711 b')' | b']' | b'}' => {
3712 level = level.saturating_sub(1);
3713 index += 1;
3714 }
3715 _ if level == 0 && starts_identifier(bytes, index, keyword) => return Some(index),
3716 _ => index += 1,
3717 }
3718 }
3719 None
3720}
3721
3722fn find_byte_at_level(bytes: &[u8], mut index: usize, end: usize, needle: u8) -> Option<usize> {
3723 let mut level = 0usize;
3724 while index < end {
3725 match bytes[index] {
3726 b'#' if level == 0 => return None,
3727 b'\'' | b'"' => {
3728 index = skip_quoted_string(bytes, index);
3729 }
3730 b'(' | b'[' | b'{' => {
3731 level += 1;
3732 index += 1;
3733 }
3734 b')' | b']' | b'}' => {
3735 level = level.saturating_sub(1);
3736 index += 1;
3737 }
3738 byte if level == 0 && byte == needle => return Some(index),
3739 _ => index += 1,
3740 }
3741 }
3742 None
3743}
3744
3745fn expression_name_and_range(source: &str) -> Option<(&'static str, usize, usize, bool)> {
3746 let parsed = parser::parse(source, parser::Mode::Expression.into()).ok()?;
3747 let ast::Mod::Expression(expression) = parsed.into_syntax() else {
3748 return None;
3749 };
3750 let is_name = matches!(expression.body.as_ref(), ast::Expr::Name(_));
3751 Some((
3752 delete_target_expr_name(&expression.body),
3753 expression.body.range().start().to_usize(),
3754 expression.body.range().end().to_usize(),
3755 is_name,
3756 ))
3757}
3758
3759fn invalid_standalone_except_error(source: &str) -> Option<CpythonDiagnostic> {
3760 let bytes = source.as_bytes();
3761 let mut line_start = 0usize;
3762 let mut seen_try = false;
3763 for line in source.split_inclusive('\n') {
3764 let line_end = line_start + line.len();
3765 let column = skip_horizontal_whitespace(bytes, line_start);
3766 if column >= line_end {
3767 line_start = line_end;
3768 continue;
3769 }
3770 if starts_identifier(bytes, column, b"try") {
3771 seen_try = true;
3772 } else if (bytes.get(column..column + 7) == Some(b"except*")
3773 || starts_identifier(bytes, column, b"except"))
3774 && !seen_try
3775 {
3776 let end = if bytes.get(column..column + 7) == Some(b"except*") {
3777 column + 7
3778 } else {
3779 column + 6
3780 };
3781 return Some(CpythonDiagnostic::new(
3782 "invalid syntax".to_owned(),
3783 column,
3784 end,
3785 ));
3786 }
3787 line_start = line_end;
3788 }
3789 None
3790}
3791
3792fn invalid_import_statement_error(source: &str) -> Option<CpythonDiagnostic> {
3793 let bytes = source.as_bytes();
3794 let mut line_start = 0usize;
3795 for line in source.split_inclusive('\n') {
3796 let line_end = line_start + line.len();
3797 let column = skip_horizontal_whitespace(bytes, line_start);
3798 if column < line_end
3799 && starts_identifier(bytes, column, b"import")
3800 && find_keyword_at_level(bytes, column + 6, line_end, b"from").is_some()
3801 {
3802 return Some(CpythonDiagnostic::new(
3803 "Did you mean to use 'from ... import ...' instead?".to_owned(),
3804 column,
3805 column + 6,
3806 ));
3807 }
3808 line_start = line_end;
3809 }
3810 None
3811}
3812
3813fn import_as_target_end(bytes: &[u8], mut index: usize) -> usize {
3814 let mut level = 0usize;
3815 while index < bytes.len() {
3816 match bytes[index] {
3817 b'#' if level == 0 => return index,
3818 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
3819 b'(' | b'[' | b'{' => {
3820 level += 1;
3821 index += 1;
3822 }
3823 b')' if level == 0 => return index,
3824 b')' | b']' | b'}' => {
3825 level = level.saturating_sub(1);
3826 index += 1;
3827 }
3828 b',' | b';' | b'\n' if level == 0 => return index,
3829 _ => index += 1,
3830 }
3831 }
3832 index
3833}
3834
3835fn valid_import_alias_name(bytes: &[u8], mut start: usize, end: usize) -> bool {
3836 start = skip_horizontal_whitespace(bytes, start);
3837 let Some(&first) = bytes.get(start) else {
3838 return false;
3839 };
3840 if !(first == b'_' || first.is_ascii_alphabetic() || first >= 0x80) {
3841 return false;
3842 }
3843 let mut index = start + 1;
3844 while index < end {
3845 match bytes[index] {
3846 b' ' | b'\t' | b'\x0c' => break,
3847 byte if byte >= 0x80 || is_ascii_identifier_char(byte) => index += 1,
3848 _ => return false,
3849 }
3850 }
3851 let index = skip_horizontal_whitespace(bytes, index);
3852 matches!(
3853 bytes.get(index),
3854 None | Some(b',' | b')' | b';' | b'\n' | b'\r')
3855 )
3856}
3857
3858fn import_target_error_for_slice(
3859 source: &str,
3860 start: usize,
3861 end: usize,
3862) -> Option<CpythonDiagnostic> {
3863 let bytes = source.as_bytes();
3864 let (target_start, target_end) = trim_target_range(bytes, start, end);
3865 if target_start >= target_end || valid_import_alias_name(bytes, target_start, target_end) {
3866 return None;
3867 }
3868 let parsed = parser::parse(
3869 &source[target_start..target_end],
3870 parser::Mode::Expression.into(),
3871 )
3872 .ok()?;
3873 let ast::Mod::Expression(expression) = parsed.into_syntax() else {
3874 return None;
3875 };
3876 let name = delete_target_expr_name(&expression.body);
3877 let start = target_start + expression.body.range().start().to_usize();
3878 let end = target_start + expression.body.range().end().to_usize();
3879 Some(CpythonDiagnostic::new(
3880 format!("cannot use {name} as import target"),
3881 start,
3882 end,
3883 ))
3884}
3885
3886fn statement_starts_import(bytes: &[u8], line_start: usize, line_end: usize) -> bool {
3887 let column = skip_horizontal_whitespace(bytes, line_start);
3888 if starts_identifier(bytes, column, b"import") {
3889 return true;
3890 }
3891 starts_identifier(bytes, column, b"from")
3892 && find_keyword_at_level(bytes, column + 4, line_end, b"import").is_some()
3893}
3894
3895fn invalid_import_target_error(source: &str) -> Option<CpythonDiagnostic> {
3896 let bytes = source.as_bytes();
3897 let mut line_start = 0usize;
3898 let mut in_parenthesized_from_import = false;
3899 for line in source.split_inclusive('\n') {
3900 let line_end = line_start + line.len();
3901 let starts_import = statement_starts_import(bytes, line_start, line_end);
3902 if starts_import && bytes[line_start..line_end].contains(&b'(') {
3903 in_parenthesized_from_import = true;
3904 }
3905 if starts_import || in_parenthesized_from_import {
3906 let mut index = line_start;
3907 while index < line_end {
3908 if starts_identifier(bytes, index, b"as") {
3909 let target_start = skip_horizontal_whitespace(bytes, index + 2);
3910 let target_end = import_as_target_end(bytes, target_start);
3911 if let Some(error) =
3912 import_target_error_for_slice(source, target_start, target_end)
3913 {
3914 return Some(error);
3915 }
3916 index = target_end.max(index + 2);
3917 } else {
3918 index += 1;
3919 }
3920 }
3921 }
3922 if in_parenthesized_from_import && bytes[line_start..line_end].contains(&b')') {
3923 in_parenthesized_from_import = false;
3924 }
3925 line_start = line_end;
3926 }
3927 None
3928}
3929
3930fn invalid_except_as_target_error(source: &str) -> Option<CpythonDiagnostic> {
3931 let bytes = source.as_bytes();
3932 let mut line_start = 0usize;
3933 let mut seen_try = false;
3934 for line in source.split_inclusive('\n') {
3935 let line_end = line_start + line.len();
3936 let mut column = skip_horizontal_whitespace(bytes, line_start);
3937 if column >= line_end {
3938 line_start = line_end;
3939 continue;
3940 }
3941 if starts_identifier(bytes, column, b"try") {
3942 seen_try = true;
3943 line_start = line_end;
3944 continue;
3945 }
3946 let (keyword_len, starred) = if bytes.get(column..column + 7) == Some(b"except*") {
3947 (7, true)
3948 } else if starts_identifier(bytes, column, b"except") {
3949 (6, false)
3950 } else {
3951 line_start = line_end;
3952 continue;
3953 };
3954 if !seen_try {
3955 line_start = line_end;
3956 continue;
3957 }
3958 column += keyword_len;
3959 let Some(as_index) = find_keyword_at_level(bytes, column, line_end, b"as") else {
3960 line_start = line_end;
3961 continue;
3962 };
3963 let target_start = skip_horizontal_whitespace(bytes, as_index + 2);
3964 let Some(delimiter) = find_byte_at_level(bytes, target_start, line_end, b':')
3965 .into_iter()
3966 .chain(find_byte_at_level(bytes, target_start, line_end, b','))
3967 .min()
3968 else {
3969 line_start = line_end;
3970 continue;
3971 };
3972 let mut target_end = delimiter;
3973 while target_end > target_start
3974 && matches!(bytes.get(target_end - 1), Some(b' ' | b'\t' | b'\x0c'))
3975 {
3976 target_end -= 1;
3977 }
3978 let Some((expr_name, start, end, is_name)) =
3979 expression_name_and_range(&source[target_start..target_end])
3980 else {
3981 line_start = line_end;
3982 continue;
3983 };
3984 if !is_name {
3985 let statement = if starred { "except*" } else { "except" };
3986 return Some(CpythonDiagnostic::new(
3987 format!("cannot use {statement} statement with {expr_name}"),
3988 target_start + start,
3989 target_start + end,
3990 ));
3991 }
3992 line_start = line_end;
3993 }
3994 None
3995}
3996
3997fn invalid_match_as_target_error(source: &str) -> Option<CpythonDiagnostic> {
3998 let bytes = source.as_bytes();
3999 let quoted_ranges = quoted_string_ranges(bytes);
4000 let mut quoted_range = 0usize;
4001 let mut line_start = 0usize;
4002 for line in source.split_inclusive('\n') {
4003 let line_end = line_start + line.len();
4004 let mut column = skip_horizontal_whitespace(bytes, line_start);
4005 if column >= line_end
4006 || offset_in_ranges("ed_ranges, &mut quoted_range, column)
4007 || !starts_identifier(bytes, column, b"case")
4008 {
4009 line_start = line_end;
4010 continue;
4011 }
4012 column += 4;
4013 let Some(as_index) = find_keyword_at_level(bytes, column, line_end, b"as") else {
4014 line_start = line_end;
4015 continue;
4016 };
4017 let target_start = skip_horizontal_whitespace(bytes, as_index + 2);
4018 let Some(delimiter) = find_byte_at_level(bytes, target_start, line_end, b':')
4019 .into_iter()
4020 .chain(find_byte_at_level(bytes, target_start, line_end, b','))
4021 .min()
4022 else {
4023 line_start = line_end;
4024 continue;
4025 };
4026 let mut target_end = delimiter;
4027 while target_end > target_start
4028 && matches!(bytes.get(target_end - 1), Some(b' ' | b'\t' | b'\x0c'))
4029 {
4030 target_end -= 1;
4031 }
4032 if source[target_start..target_end].trim() == "_" {
4033 return Some(CpythonDiagnostic::new(
4034 "cannot use '_' as a target".to_owned(),
4035 target_start,
4036 target_end,
4037 ));
4038 }
4039 let Some((expr_name, start, end, is_name)) =
4040 expression_name_and_range(&source[target_start..target_end])
4041 else {
4042 line_start = line_end;
4043 continue;
4044 };
4045 if !is_name {
4046 if matches!(expr_name, "expression" | "subscript") {
4047 line_start = line_end;
4048 continue;
4049 }
4050 return Some(CpythonDiagnostic::new(
4051 format!("cannot use {expr_name} as pattern target"),
4052 target_start + start,
4053 target_start + end,
4054 ));
4055 }
4056 line_start = line_end;
4057 }
4058 None
4059}
4060
4061fn quoted_string_ranges(bytes: &[u8]) -> Vec<(usize, usize)> {
4062 let mut ranges = Vec::new();
4063 let mut index = 0usize;
4064 while index < bytes.len() {
4065 match bytes[index] {
4066 b'#' => {
4067 while index < bytes.len() && bytes[index] != b'\n' {
4068 index += 1;
4069 }
4070 }
4071 b'\'' | b'"' => {
4072 let end = skip_quoted_string(bytes, index);
4073 ranges.push((index, end));
4074 index = end;
4075 }
4076 _ => index += 1,
4077 }
4078 }
4079 ranges
4080}
4081
4082fn offset_in_ranges(ranges: &[(usize, usize)], range_index: &mut usize, offset: usize) -> bool {
4083 while ranges
4084 .get(*range_index)
4085 .is_some_and(|(_, end)| *end <= offset)
4086 {
4087 *range_index += 1;
4088 }
4089 ranges
4090 .get(*range_index)
4091 .is_some_and(|(start, end)| *start <= offset && offset < *end)
4092}
4093
4094fn invalid_match_mapping_rest_wildcard_error(source: &str) -> Option<CpythonDiagnostic> {
4095 let bytes = source.as_bytes();
4096 let next_line_end = |line_start: usize| {
4097 line_start
4098 + bytes[line_start..]
4099 .iter()
4100 .position(|byte| *byte == b'\n')
4101 .unwrap_or(bytes.len() - line_start)
4102 };
4103 let mut index = 0usize;
4104 let mut line_start = 0usize;
4105 let mut line_end = next_line_end(line_start);
4106 while index < bytes.len() {
4107 match bytes[index] {
4108 b'#' => {
4109 while index < bytes.len() && bytes[index] != b'\n' {
4110 index += 1;
4111 }
4112 }
4113 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
4114 b'\n' => {
4115 index += 1;
4116 line_start = index;
4117 line_end = next_line_end(line_start);
4118 }
4119 _ => {
4120 let column = skip_horizontal_whitespace(bytes, line_start);
4121 if index != column
4122 || column >= line_end
4123 || !starts_identifier(bytes, column, b"case")
4124 {
4125 index += 1;
4126 continue;
4127 }
4128 let mut cursor = column + 4;
4129 while cursor < line_end {
4130 match bytes[cursor] {
4131 b'#' => break,
4132 b'\'' | b'"' => cursor = skip_quoted_string(bytes, cursor),
4133 b'{' => {
4134 let rest = next_non_horizontal_whitespace(bytes, cursor + 1);
4135 if bytes.get(rest..rest + 2) == Some(b"**") {
4136 let name_start = next_non_horizontal_whitespace(bytes, rest + 2);
4137 let name_end = identifier_end(bytes, name_start, line_end);
4138 if source.get(name_start..name_end) == Some("_") {
4139 return Some(CpythonDiagnostic::new(
4140 "invalid syntax".to_owned(),
4141 name_start,
4142 name_end,
4143 ));
4144 }
4145 }
4146 cursor += 1;
4147 }
4148 _ => cursor += 1,
4149 }
4150 }
4151 index = line_end;
4152 }
4153 }
4154 }
4155 None
4156}
4157
4158fn invalid_if_expression_statement_error(source: &str) -> Option<CpythonDiagnostic> {
4159 let bytes = source.as_bytes();
4160 let mut line_start = 0usize;
4161 for line in source.split_inclusive('\n') {
4162 let line_end = line_start + line.len();
4163 if let Some(if_index) = find_keyword_at_level(bytes, line_start, line_end, b"if")
4164 && let Some((start, end)) = statement_before_if_expression(bytes, line_start, if_index)
4165 && find_keyword_at_level(bytes, if_index + 2, line_end, b"else").is_some()
4166 {
4167 return Some(CpythonDiagnostic::new(
4168 "expected expression before 'if', but statement is given".to_owned(),
4169 start,
4170 end,
4171 ));
4172 }
4173 if let Some(else_index) = find_keyword_at_level(bytes, line_start, line_end, b"else")
4174 && find_keyword_at_level(bytes, line_start, else_index, b"if").is_some()
4175 && let Some((start, end)) =
4176 statement_after_else_expression(bytes, else_index + 4, line_end)
4177 {
4178 return Some(CpythonDiagnostic::new(
4179 "expected expression after 'else', but statement is given".to_owned(),
4180 start,
4181 end,
4182 ));
4183 }
4184 line_start = line_end;
4185 }
4186 None
4187}
4188
4189fn statement_before_if_expression(
4190 bytes: &[u8],
4191 line_start: usize,
4192 if_index: usize,
4193) -> Option<(usize, usize)> {
4194 let mut start = if_index;
4195 while start > line_start && matches!(bytes.get(start - 1), Some(b' ' | b'\t' | b'\x0c')) {
4196 start -= 1;
4197 }
4198 while start > line_start
4199 && !matches!(
4200 bytes.get(start - 1),
4201 Some(b'=' | b':' | b',' | b'(' | b'[' | b'{')
4202 )
4203 {
4204 start -= 1;
4205 }
4206 start = skip_horizontal_whitespace(bytes, start);
4207 for keyword in [b"pass".as_slice(), b"break", b"continue"] {
4208 if starts_identifier(bytes, start, keyword) {
4209 return Some((start, start + keyword.len()));
4210 }
4211 }
4212 None
4213}
4214
4215fn statement_after_else_expression(
4216 bytes: &[u8],
4217 else_end: usize,
4218 line_end: usize,
4219) -> Option<(usize, usize)> {
4220 let start = skip_horizontal_whitespace(bytes, else_end);
4221 for keyword in [
4222 b"pass".as_slice(),
4223 b"return",
4224 b"raise",
4225 b"del",
4226 b"yield",
4227 b"assert",
4228 b"break",
4229 b"continue",
4230 b"import",
4231 b"from",
4232 ] {
4233 if starts_identifier(bytes, start, keyword) {
4234 let end = statement_target_end(bytes, start).min(line_end);
4235 return Some((start, end.max(start + keyword.len())));
4236 }
4237 }
4238 None
4239}
4240
4241fn invalid_else_elif_error(source: &str) -> Option<CpythonDiagnostic> {
4242 let bytes = source.as_bytes();
4243 let mut line_start = 0usize;
4244 let mut else_indents: Vec<usize> = Vec::new();
4245 for line in source.split_inclusive('\n') {
4246 let line_end = line_start + line.len();
4247 let column = skip_horizontal_whitespace(bytes, line_start);
4248 let line_column = column.saturating_sub(line_start);
4249 if column >= line_end {
4250 line_start = line_end;
4251 continue;
4252 }
4253 while else_indents
4254 .last()
4255 .is_some_and(|indent| line_column < *indent)
4256 {
4257 else_indents.pop();
4258 }
4259 if starts_identifier(bytes, column, b"else")
4260 && find_byte_at_level(bytes, column + 4, line_end, b':').is_some()
4261 {
4262 else_indents.push(line_column);
4263 } else if starts_identifier(bytes, column, b"elif") && else_indents.contains(&line_column) {
4264 return Some(CpythonDiagnostic::new(
4265 "'elif' block follows an 'else' block".to_owned(),
4266 column,
4267 column + 4,
4268 ));
4269 }
4270 line_start = line_end;
4271 }
4272 None
4273}
4274
4275fn mixed_except_handlers_error(source: &str) -> Option<CpythonDiagnostic> {
4276 let message = "cannot have both 'except' and 'except*' on the same 'try'".to_owned();
4277 let mut seen_except = false;
4278 let mut seen_except_star = false;
4279 let mut line_start = 0usize;
4280 for line in source.split_inclusive('\n') {
4281 let bytes = line.as_bytes();
4282 let mut column = 0usize;
4283 while matches!(bytes.get(column), Some(b' ' | b'\t' | b'\x0c')) {
4284 column += 1;
4285 }
4286 let token_start = line_start + column;
4287 if bytes.get(column..column + 7) == Some(b"except*") {
4288 if seen_except {
4289 return Some(CpythonDiagnostic::new(
4290 message,
4291 token_start,
4292 token_start + 7,
4293 ));
4294 }
4295 seen_except_star = true;
4296 } else if starts_identifier(bytes, column, b"except") {
4297 if seen_except_star {
4298 return Some(CpythonDiagnostic::new(
4299 message,
4300 token_start,
4301 token_start + 6,
4302 ));
4303 }
4304 seen_except = true;
4305 }
4306 line_start += line.len();
4307 }
4308 None
4309}
4310
4311fn non_printable_character_error(source: &str) -> Option<CpythonDiagnostic> {
4312 let bytes = source.as_bytes();
4313 let mut index = 0;
4314 while index < bytes.len() {
4315 match bytes[index] {
4316 b'#' => {
4317 while index < bytes.len() && bytes[index] != b'\n' {
4318 index += 1;
4319 }
4320 }
4321 b'\'' | b'"' => {
4322 index = skip_quoted_string(bytes, index);
4323 }
4324 byte if byte.is_ascii_control() && !matches!(byte, b'\t' | b'\n' | b'\r' | b'\x0c') => {
4325 return Some(CpythonDiagnostic::new(
4326 format!("invalid non-printable character U+{byte:04X}"),
4327 index,
4328 index + 1,
4329 ));
4330 }
4331 byte if byte >= 0x80 => {
4332 let ch = source[index..].chars().next()?;
4333 if ch.is_control() {
4334 return Some(CpythonDiagnostic::new(
4335 format!("invalid non-printable character U+{:04X}", ch as u32),
4336 index,
4337 index + ch.len_utf8(),
4338 ));
4339 }
4340 index += ch.len_utf8();
4341 }
4342 _ => index += 1,
4343 }
4344 }
4345 None
4346}
4347
4348fn unterminated_string_error(source: &str, mode: Mode) -> Option<CpythonDiagnostic> {
4349 let bytes = source.as_bytes();
4350 let mut index = 0;
4351 let mut line = 1usize;
4352 while index < bytes.len() {
4353 match bytes[index] {
4354 b'#' => {
4355 while index < bytes.len() && bytes[index] != b'\n' {
4356 index += 1;
4357 }
4358 }
4359 b'\n' => {
4360 line += 1;
4361 index += 1;
4362 }
4363 quote @ (b'\'' | b'"') => {
4364 let start = index;
4365 let start_line = line;
4366 let quote_size = if bytes.get(index + 1) == Some("e)
4367 && bytes.get(index + 2) == Some("e)
4368 {
4369 3
4370 } else {
4371 1
4372 };
4373 index += quote_size;
4374 let mut has_escaped_quote = false;
4375 let mut ended_with_escape = false;
4376 let mut closed = false;
4377 while index < bytes.len() {
4378 let c = bytes[index];
4379 if c == b'\n' {
4380 if quote_size == 1 {
4381 break;
4385 }
4386 line += 1;
4387 index += 1;
4388 } else if c == quote {
4389 if quote_size == 3 {
4390 if bytes.get(index + 1) == Some("e)
4391 && bytes.get(index + 2) == Some("e)
4392 {
4393 index += 3;
4394 closed = true;
4395 break;
4396 }
4397 index += 1;
4398 } else {
4399 index += 1;
4400 closed = true;
4401 break;
4402 }
4403 } else if c == b'\\' {
4404 if bytes.get(index + 1) == Some("e) {
4405 has_escaped_quote = true;
4406 }
4407 ended_with_escape = index + 1 >= bytes.len();
4408 index = (index + 2).min(bytes.len());
4409 } else {
4410 index += 1;
4411 }
4412 }
4413 if !closed {
4414 if let Some(error) =
4415 unclosed_replacement_field_error(bytes, start, start + quote_size, index)
4416 {
4417 return Some(error);
4418 }
4419 let detected_line = if quote_size == 3 { line } else { start_line };
4420 let interpolated = interpolated_string_prefix(bytes, start);
4421 let diagnostic = CpythonDiagnostic::new(
4422 unterminated_string_message(
4423 detected_line,
4424 quote_size == 3,
4425 has_escaped_quote,
4426 interpolated,
4427 ),
4428 start,
4429 start,
4430 );
4431 let exec_implicit_newline =
4436 matches!(mode, Mode::Exec) && quote_size == 1 && !ended_with_escape;
4437 let eval_assignment =
4438 matches!(mode, Mode::Eval) && eval_has_assignment_before(bytes, start);
4439 let continuable = index >= bytes.len()
4440 && !(interpolated.is_some() && quote_size == 1)
4441 && !exec_implicit_newline
4442 && !eval_assignment;
4443 return Some(if continuable {
4444 diagnostic.with_unclosed_string()
4445 } else {
4446 diagnostic
4447 });
4448 }
4449 }
4450 _ => index += 1,
4451 }
4452 }
4453 None
4454}
4455
4456fn eval_has_assignment_before(bytes: &[u8], end: usize) -> bool {
4457 let mut index = 0;
4458 let mut level = 0usize;
4459 let mut in_lambda_params = false;
4460 while index < end {
4461 match bytes[index] {
4462 b'#' => {
4463 while index < end && bytes[index] != b'\n' {
4464 index += 1;
4465 }
4466 }
4467 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
4468 b'(' | b'[' | b'{' => {
4469 level += 1;
4470 index += 1;
4471 }
4472 b')' | b']' | b'}' => {
4473 level = level.saturating_sub(1);
4474 index += 1;
4475 }
4476 b':' if level == 0 => {
4477 in_lambda_params = false;
4478 index += 1;
4479 }
4480 b'=' if level == 0
4481 && !in_lambda_params
4482 && bytes.get(index + 1) != Some(&b'=')
4483 && !matches!(
4484 bytes.get(index.saturating_sub(1)),
4485 Some(b':' | b'!' | b'<' | b'>')
4486 ) =>
4487 {
4488 return true;
4489 }
4490 _ if level == 0 && starts_identifier(bytes, index, b"lambda") => {
4491 in_lambda_params = true;
4492 index += b"lambda".len();
4493 }
4494 _ => index += 1,
4495 }
4496 }
4497 false
4498}
4499
4500fn invalid_interpolated_string_error(source: &str) -> Option<CpythonDiagnostic> {
4501 let bytes = source.as_bytes();
4502 let mut index = 0;
4503 while index < bytes.len() {
4504 match bytes[index] {
4505 b'#' => {
4506 while index < bytes.len() && bytes[index] != b'\n' {
4507 index += 1;
4508 }
4509 }
4510 quote @ (b'\'' | b'"') => {
4511 let Some(prefix) = interpolated_string_prefix(bytes, index) else {
4512 index = skip_quoted_string(bytes, index);
4513 continue;
4514 };
4515 if let Some(error) =
4516 single_quoted_format_spec_newline_error(bytes, index, quote, prefix)
4517 {
4518 return Some(error);
4519 }
4520 let Some((content_start, content_end)) =
4521 quoted_string_content_range(bytes, index, quote)
4522 else {
4523 index = skip_quoted_string(bytes, index);
4524 continue;
4525 };
4526 if let Some(error) =
4527 invalid_replacement_field_error(bytes, content_start, content_end, prefix)
4528 {
4529 return Some(error);
4530 }
4531 index = skip_quoted_string(bytes, index);
4532 }
4533 _ => index += 1,
4534 }
4535 }
4536 None
4537}
4538
4539fn is_bytes_literal(bytes: &[u8], quote: usize) -> bool {
4541 let mut start = quote;
4542 while quote - start < 2
4543 && start > 0
4544 && matches!(
4545 bytes[start - 1].to_ascii_lowercase(),
4546 b'r' | b'b' | b'u' | b'f' | b't'
4547 )
4548 {
4549 start -= 1;
4550 }
4551 if start > 0 && is_ascii_identifier_char(bytes[start - 1]) {
4552 return false;
4553 }
4554 bytes[start..quote]
4555 .iter()
4556 .any(|byte| byte.eq_ignore_ascii_case(&b'b'))
4557}
4558
4559fn mixed_tstring_literal_error(
4569 error: &parser::ParseError,
4570 source: &str,
4571) -> Option<CpythonDiagnostic> {
4572 let parser::ParseErrorType::OtherError(message) = &error.error else {
4573 return None;
4574 };
4575 if !message.eq_ignore_ascii_case("bytes literal cannot be mixed with non-bytes literals") {
4576 return None;
4577 }
4578
4579 let start = error.location.start().to_usize();
4580 let end = error.location.end().to_usize();
4581 let bytes = source.as_bytes();
4582 if end > bytes.len() {
4583 return None;
4584 }
4585
4586 let mut index = start;
4587 let (mut bytes_seen, mut nonbytes_seen) = (false, false);
4588 while index < end {
4589 match bytes[index] {
4590 b'\'' | b'"' => {
4591 if interpolated_string_prefix(bytes, index) == Some("t-string") {
4592 let message = if bytes_seen && nonbytes_seen {
4593 "cannot mix bytes and nonbytes literals"
4596 } else {
4597 "cannot mix t-string literals with string or bytes literals"
4598 };
4599 return Some(CpythonDiagnostic::new(message.to_owned(), start, end));
4600 }
4601 if is_bytes_literal(bytes, index) {
4602 bytes_seen = true;
4603 } else {
4604 nonbytes_seen = true;
4605 }
4606 index = skip_quoted_string(bytes, index);
4607 }
4608 _ => index += 1,
4609 }
4610 }
4611 None
4612}
4613
4614fn single_quoted_format_spec_newline_error(
4615 bytes: &[u8],
4616 quote_index: usize,
4617 quote: u8,
4618 prefix: &str,
4619) -> Option<CpythonDiagnostic> {
4620 if bytes.get(quote_index + 1) == Some("e) && bytes.get(quote_index + 2) == Some("e) {
4621 return None;
4622 }
4623
4624 let (content_start, content_end) = quoted_string_content_range(bytes, quote_index, quote)?;
4625 let mut index = content_start;
4626 while index < content_end {
4627 match bytes[index] {
4628 b'{' if bytes.get(index + 1) == Some(&b'{') => index += 2,
4629 b'}' if bytes.get(index + 1) == Some(&b'}') => index += 2,
4630 b'{' => {
4631 let expr_start = skip_ascii_whitespace(bytes, index + 1, content_end);
4632 if let Some(separator) = replacement_field_separator(bytes, expr_start, content_end)
4633 && bytes[separator] == b':'
4634 {
4635 let format_end =
4636 replacement_field_closing_brace(bytes, separator + 1, content_end)
4637 .unwrap_or(content_end);
4638 if bytes[separator + 1..format_end].contains(&b'\n') {
4639 return Some(CpythonDiagnostic::new(
4640 format!(
4641 "{prefix}: newlines are not allowed in format specifiers for single quoted {prefix}s"
4642 ),
4643 quote_index,
4644 quote_index + 1,
4645 ));
4646 }
4647 }
4648 index += 1;
4649 }
4650 _ => index += 1,
4651 }
4652 }
4653 None
4654}
4655
4656fn interpolated_string_prefix(bytes: &[u8], quote: usize) -> Option<&'static str> {
4657 let prev = quote.checked_sub(1).and_then(|index| bytes.get(index))?;
4658 let lower_prev = prev.to_ascii_lowercase();
4659 let (prefix_start, marker) = if matches!(lower_prev, b'f' | b't') {
4660 if quote >= 2 && bytes[quote - 2].eq_ignore_ascii_case(&b'r') {
4661 (quote - 2, lower_prev)
4662 } else {
4663 (quote - 1, lower_prev)
4664 }
4665 } else if lower_prev == b'r'
4666 && quote >= 2
4667 && matches!(bytes[quote - 2].to_ascii_lowercase(), b'f' | b't')
4668 {
4669 (quote - 2, bytes[quote - 2].to_ascii_lowercase())
4670 } else {
4671 return None;
4672 };
4673
4674 if prefix_start > 0 && identifier_continue_before(bytes, prefix_start) {
4675 return None;
4676 }
4677
4678 Some(if marker == b'f' {
4679 "f-string"
4680 } else {
4681 "t-string"
4682 })
4683}
4684
4685const MAXFSTRINGLEVEL: usize = 150;
4686
4687fn skip_string_token(bytes: &[u8], quote_index: usize) -> usize {
4688 if interpolated_string_prefix(bytes, quote_index).is_some() {
4689 interpolated_string_end(bytes, quote_index).unwrap_or(bytes.len())
4690 } else {
4691 skip_quoted_string(bytes, quote_index)
4692 }
4693}
4694
4695fn interpolated_string_end(bytes: &[u8], quote_index: usize) -> Option<usize> {
4696 let quote = bytes[quote_index];
4697 let triple =
4698 bytes.get(quote_index + 1) == Some("e) && bytes.get(quote_index + 2) == Some("e);
4699 let quote_len = if triple { 3 } else { 1 };
4700 let mut index = quote_index + quote_len;
4701 let mut brace_depth = 0usize;
4702 while index < bytes.len() {
4703 if brace_depth == 0 {
4704 if bytes[index] == b'\\' {
4705 index = (index + 2).min(bytes.len());
4706 continue;
4707 }
4708 if triple
4709 && bytes.get(index) == Some("e)
4710 && bytes.get(index + 1) == Some("e)
4711 && bytes.get(index + 2) == Some("e)
4712 {
4713 return Some(index + 3);
4714 }
4715 if !triple && bytes[index] == quote {
4716 return Some(index + 1);
4717 }
4718 if bytes[index] == b'{' {
4719 if bytes.get(index + 1) == Some(&b'{') {
4720 index += 2;
4721 } else {
4722 brace_depth = 1;
4723 index += 1;
4724 }
4725 } else if bytes[index] == b'}' && bytes.get(index + 1) == Some(&b'}') {
4726 index += 2;
4727 } else {
4728 index += 1;
4729 }
4730 } else {
4731 match bytes[index] {
4732 quote_ch @ (b'\'' | b'"') => {
4733 if quoted_string_is_closed(bytes, index) {
4734 index = skip_string_token(bytes, index).max(index + 1);
4735 } else if !triple && quote_ch == quote {
4736 return Some(index + 1);
4737 } else {
4738 index = skip_string_token(bytes, index).max(index + 1);
4739 }
4740 }
4741 b'#' => {
4742 while index < bytes.len() && bytes[index] != b'\n' {
4743 index += 1;
4744 }
4745 }
4746 b'{' => {
4747 brace_depth += 1;
4748 index += 1;
4749 }
4750 b'}' => {
4751 brace_depth -= 1;
4752 index += 1;
4753 }
4754 b'\\' => index = (index + 2).min(bytes.len()),
4755 _ => index += 1,
4756 }
4757 }
4758 }
4759 None
4760}
4761
4762fn quote_len_at(bytes: &[u8], quote_index: usize) -> usize {
4763 let quote = bytes[quote_index];
4764 if bytes.get(quote_index + 1) == Some("e) && bytes.get(quote_index + 2) == Some("e) {
4765 3
4766 } else {
4767 1
4768 }
4769}
4770
4771fn interpolated_string_content_range(bytes: &[u8], quote_index: usize) -> Option<(usize, usize)> {
4772 let end = interpolated_string_end(bytes, quote_index)?;
4773 let quote_len = quote_len_at(bytes, quote_index);
4774 Some((quote_index + quote_len, end - quote_len))
4775}
4776
4777fn quoted_string_content_range(
4778 bytes: &[u8],
4779 quote_index: usize,
4780 quote: u8,
4781) -> Option<(usize, usize)> {
4782 let triple =
4783 bytes.get(quote_index + 1) == Some("e) && bytes.get(quote_index + 2) == Some("e);
4784 let quote_len = if triple { 3 } else { 1 };
4785 let content_start = quote_index + quote_len;
4786 let mut index = content_start;
4787 while index < bytes.len() {
4788 if bytes[index] == b'\\' {
4789 index = (index + 2).min(bytes.len());
4790 } else if (triple
4791 && bytes.get(index) == Some("e)
4792 && bytes.get(index + 1) == Some("e)
4793 && bytes.get(index + 2) == Some("e))
4794 || (!triple && bytes[index] == quote)
4795 {
4796 return Some((content_start, index));
4797 } else {
4798 index += 1;
4799 }
4800 }
4801 None
4802}
4803
4804fn invalid_replacement_field_error(
4805 bytes: &[u8],
4806 start: usize,
4807 end: usize,
4808 prefix: &str,
4809) -> Option<CpythonDiagnostic> {
4810 let mut index = start;
4811 while index < end {
4812 match bytes[index] {
4813 b'{' if bytes.get(index + 1) == Some(&b'{') => index += 2,
4814 b'}' if bytes.get(index + 1) == Some(&b'}') => index += 2,
4815 b'{' => {
4816 if let Some(error) = replacement_field_error(bytes, index, end, prefix, 0) {
4817 return Some(error);
4818 }
4819 index = replacement_field_end(bytes, index, end).unwrap_or(end);
4822 }
4823 _ => index += 1,
4824 }
4825 }
4826 None
4827}
4828
4829fn replacement_field_end(bytes: &[u8], open: usize, end: usize) -> Option<usize> {
4832 let expr_start = skip_replacement_field_trivia(bytes, open + 1, end);
4833 let separator = replacement_field_separator(bytes, expr_start, end)?;
4834 if bytes[separator] == b'}' {
4835 return Some(separator + 1);
4836 }
4837 replacement_field_closing_brace(bytes, separator + 1, end).map(|brace| brace + 1)
4838}
4839
4840fn line_of_offset(bytes: &[u8], offset: usize) -> usize {
4844 1 + bytes[..offset]
4845 .iter()
4846 .filter(|byte| **byte == b'\n')
4847 .count()
4848}
4849
4850fn replacement_field_bracket_error(
4855 bytes: &[u8],
4856 start: usize,
4857 end: usize,
4858 prefix: &str,
4859) -> Option<CpythonDiagnostic> {
4860 let mut stack: Vec<(u8, usize)> = Vec::new();
4862 let mut index = start;
4863 while index < end {
4864 let byte = bytes[index];
4865 match byte {
4866 b'\'' | b'"' => {
4867 index = skip_quoted_string(bytes, index);
4868 continue;
4869 }
4870 b'#' => {
4871 index = skip_replacement_field_comment(bytes, index, end);
4872 continue;
4873 }
4874 b'(' | b'[' | b'{' => stack.push((byte, index)),
4875 b')' | b']' | b'}' => {
4876 let expected = expected_opening_bracket(byte as char) as u8;
4877 match stack.last() {
4878 None if byte == b'}' => return None,
4880 None => {
4881 return Some(CpythonDiagnostic::new(
4882 format!("{prefix}: unmatched '{}'", byte as char),
4883 index,
4884 index + 1,
4885 ));
4886 }
4887 Some(&(opening, _)) if opening == expected => {
4888 stack.pop();
4889 }
4890 Some(&(opening, opened_at)) => {
4891 let opening_line = line_of_offset(bytes, opened_at);
4894 let suffix = if opening_line == line_of_offset(bytes, index) {
4895 String::new()
4896 } else {
4897 format!(" on line {opening_line}")
4898 };
4899 return Some(CpythonDiagnostic::new(
4900 format!(
4901 "closing parenthesis '{}' does not match opening parenthesis '{}'{suffix}",
4902 byte as char, opening as char
4903 ),
4904 index,
4905 index + 1,
4906 ));
4907 }
4908 }
4909 }
4910 _ => {}
4911 }
4912 index += 1;
4913 }
4914 None
4915}
4916
4917fn replacement_field_comment_error(
4922 bytes: &[u8],
4923 open: usize,
4924 start: usize,
4925 end: usize,
4926 literal_end: usize,
4927) -> Option<CpythonDiagnostic> {
4928 let mut index = start;
4929 while index < end {
4930 match bytes[index] {
4931 b'\'' | b'"' => {
4932 index = skip_quoted_string(bytes, index);
4933 continue;
4934 }
4935 b'#' => {
4936 if !bytes[index..literal_end].contains(&b'\n') {
4937 return Some(
4938 CpythonDiagnostic::new("'{' was never closed".to_owned(), open, open + 1)
4939 .with_unclosed_bracket(),
4940 );
4941 }
4942 index = skip_replacement_field_comment(bytes, index, end);
4943 continue;
4944 }
4945 _ => index += 1,
4946 }
4947 }
4948 None
4949}
4950
4951fn skip_replacement_field_comment(bytes: &[u8], mut index: usize, end: usize) -> usize {
4953 while index < end && bytes[index] != b'\n' {
4954 index += 1;
4955 }
4956 index
4957}
4958
4959fn skip_replacement_field_trivia(bytes: &[u8], mut index: usize, end: usize) -> usize {
4961 loop {
4962 index = skip_ascii_whitespace(bytes, index, end);
4963 if index < end && bytes[index] == b'#' {
4964 index = skip_replacement_field_comment(bytes, index, end);
4965 continue;
4966 }
4967 return index;
4968 }
4969}
4970
4971const MAX_REPLACEMENT_FIELD_DEPTH: usize = 32;
4974
4975fn replacement_field_error(
4976 bytes: &[u8],
4977 open: usize,
4978 end: usize,
4979 prefix: &str,
4980 depth: usize,
4981) -> Option<CpythonDiagnostic> {
4982 let expr_start = skip_replacement_field_trivia(bytes, open + 1, end);
4983
4984 let separator = replacement_field_separator(bytes, expr_start, end);
4988 let expr_end = separator.unwrap_or(end);
4989
4990 if let Some(backslash) = replacement_field_line_continuation(bytes, expr_start, expr_end) {
4991 return Some(CpythonDiagnostic::new(
4992 "unexpected character after line continuation character".to_owned(),
4993 backslash + 1,
4994 (backslash + 2).min(end),
4995 ));
4996 }
4997 if let Some(quote) = unterminated_string_in_replacement_field(bytes, expr_start, expr_end) {
4998 return Some(CpythonDiagnostic::new(
4999 unterminated_string_message(1, false, false, None),
5000 quote,
5001 quote + 1,
5002 ));
5003 }
5004 if let Some(error) = replacement_field_bracket_error(bytes, expr_start, expr_end, prefix) {
5008 return Some(error);
5009 }
5010 if let Some(error) = replacement_field_comment_error(bytes, open, open + 1, expr_end, end) {
5011 return Some(error);
5012 }
5013
5014 if expr_start >= end {
5017 return Some(CpythonDiagnostic::new(
5018 format!("{prefix}: expecting '}}'"),
5019 open,
5020 open + 1,
5021 ));
5022 }
5023
5024 if let marker @ (b'=' | b'!' | b':' | b'}') = bytes[expr_start]
5025 && is_replacement_field_marker(bytes, expr_start)
5026 {
5027 return Some(CpythonDiagnostic::new(
5028 format!(
5029 "{prefix}: valid expression required before '{}'",
5030 marker as char
5031 ),
5032 expr_start,
5033 expr_start + 1,
5034 ));
5035 }
5036
5037 if bytes[expr_start] == b'{' {
5040 let inner = skip_ascii_whitespace(bytes, expr_start + 1, end);
5041 if inner < end
5042 && bytes[inner] != b'}'
5043 && invalid_replacement_expression_start(bytes, inner, end)
5044 {
5045 return Some(CpythonDiagnostic::new(
5046 format!("{prefix}: expecting a valid expression after '{{'"),
5047 expr_start,
5048 expr_start + 1,
5049 ));
5050 }
5051 }
5052
5053 if starts_identifier(bytes, expr_start, b"lambda") {
5054 return Some(CpythonDiagnostic::new(
5055 format!("{prefix}: lambda expressions are not allowed without parentheses"),
5056 expr_start,
5057 expr_start + b"lambda".len(),
5058 ));
5059 }
5060
5061 if invalid_replacement_expression_start(bytes, expr_start, end) {
5062 return Some(CpythonDiagnostic::new(
5063 format!("{prefix}: expecting a valid expression after '{{'"),
5064 open,
5065 open + 1,
5066 ));
5067 }
5068
5069 if !starts_identifier(bytes, expr_start, b"yield")
5073 && !starts_identifier(bytes, expr_start, b"await")
5074 && !starts_identifier(bytes, expr_start, b"not")
5075 && let Some(atom_end) = adjacent_atom_end(bytes, expr_start)
5076 {
5077 let next = skip_ascii_whitespace(bytes, atom_end, expr_end);
5078 if next > atom_end
5079 && expression_atom_start(bytes, next)
5080 && !expression_continuation_keyword(bytes, next)
5081 {
5082 let second_end = adjacent_atom_end(bytes, next).unwrap_or(next + 1);
5083 return Some(CpythonDiagnostic::new(
5084 "invalid syntax. Perhaps you forgot a comma?".to_owned(),
5085 expr_start,
5086 second_end,
5087 ));
5088 }
5089 }
5090
5091 if let Some(stray) = replacement_expression_stray_character(bytes, expr_start, expr_end) {
5094 return Some(CpythonDiagnostic::new(
5095 format!("{prefix}: expecting '=', or '!', or ':', or '}}'"),
5096 stray,
5097 stray + 1,
5098 ));
5099 }
5100
5101 let Some(separator) = separator else {
5102 return Some(CpythonDiagnostic::new(
5103 format!("{prefix}: expecting '}}'"),
5104 open,
5105 open + 1,
5106 ));
5107 };
5108
5109 if let Some(operator) = dangling_operator(bytes, expr_start, expr_end) {
5114 return Some(CpythonDiagnostic::new(
5115 format!("{prefix}: expecting '=', or '!', or ':', or '}}'"),
5116 operator,
5117 operator + 1,
5118 ));
5119 }
5120
5121 if let Some(lambda_at) = top_level_lambda(bytes, expr_start, expr_end)
5122 && lambda_allowed_at_expression_position(bytes, expr_start, lambda_at)
5123 {
5124 return Some(CpythonDiagnostic::new(
5125 format!("{prefix}: lambda expressions are not allowed without parentheses"),
5126 lambda_at,
5127 lambda_at + b"lambda".len(),
5128 ));
5129 }
5130
5131 if bytes[separator] == b':'
5132 && replacement_expression_has_parse_error(bytes, expr_start, separator)
5133 {
5134 return Some(CpythonDiagnostic::new(
5135 "invalid syntax".to_owned(),
5136 expr_start,
5137 separator,
5138 ));
5139 }
5140
5141 match bytes[separator] {
5142 b'=' => invalid_debug_expression_error(bytes, separator, end, prefix, depth),
5143 b'!' => invalid_conversion_error(bytes, separator, end, prefix, depth),
5144 b':' => invalid_format_spec_error(bytes, separator, end, prefix, depth),
5145 b'}' => None,
5146 _ => unreachable!(),
5147 }
5148}
5149
5150fn replacement_field_line_continuation(
5151 bytes: &[u8],
5152 mut index: usize,
5153 end: usize,
5154) -> Option<usize> {
5155 let mut level = 0usize;
5156 while index < end {
5157 match bytes[index] {
5158 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
5159 b'#' => index = skip_replacement_field_comment(bytes, index, end),
5160 b'\\' => return Some(index),
5161 b'(' | b'[' | b'{' => {
5162 level += 1;
5163 index += 1;
5164 }
5165 b')' | b']' | b'}' if level > 0 => {
5166 level -= 1;
5167 index += 1;
5168 }
5169 b'=' | b'!' | b':' | b'}'
5170 if level == 0 && is_replacement_field_marker(bytes, index) =>
5171 {
5172 return None;
5173 }
5174 _ => index += 1,
5175 }
5176 }
5177 None
5178}
5179
5180fn unterminated_string_in_replacement_field(
5181 bytes: &[u8],
5182 mut index: usize,
5183 end: usize,
5184) -> Option<usize> {
5185 while index < end {
5186 match bytes[index] {
5187 quote @ (b'\'' | b'"') => {
5188 let string_end = skip_quoted_string(bytes, index);
5189 if string_end >= end && !bytes[index + 1..end].contains("e) {
5190 return Some(index);
5191 }
5192 index = string_end;
5193 }
5194 b'#' => index = skip_replacement_field_comment(bytes, index, end),
5195 _ => index += 1,
5196 }
5197 }
5198 None
5199}
5200
5201fn top_level_lambda(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
5202 let mut level = 0usize;
5203 while index < end {
5204 match bytes[index] {
5205 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
5206 b'#' => index = skip_replacement_field_comment(bytes, index, end),
5207 b'(' | b'[' | b'{' => {
5208 level += 1;
5209 index += 1;
5210 }
5211 b')' | b']' | b'}' if level > 0 => {
5212 level -= 1;
5213 index += 1;
5214 }
5215 _ if level == 0 && starts_identifier(bytes, index, b"lambda") => {
5216 return Some(index);
5217 }
5218 _ => index += 1,
5219 }
5220 }
5221 None
5222}
5223
5224fn lambda_allowed_at_expression_position(bytes: &[u8], mut index: usize, lambda_at: usize) -> bool {
5227 let mut after_comma = true;
5228 let mut level = 0usize;
5229 while index < lambda_at {
5230 match bytes[index] {
5231 b'\'' | b'"' => {
5232 after_comma = false;
5233 index = skip_quoted_string(bytes, index);
5234 }
5235 b'#' => index = skip_replacement_field_comment(bytes, index, lambda_at),
5236 b'(' | b'[' | b'{' => {
5237 level += 1;
5238 after_comma = false;
5239 index += 1;
5240 }
5241 b')' | b']' | b'}' if level > 0 => {
5242 level -= 1;
5243 after_comma = false;
5244 index += 1;
5245 }
5246 b',' if level == 0 => {
5247 after_comma = true;
5248 index += 1;
5249 }
5250 b' ' | b'\t' | b'\n' | b'\r' | b'\x0c' => index += 1,
5251 _ => {
5252 after_comma = false;
5253 index += 1;
5254 }
5255 }
5256 }
5257 after_comma
5258}
5259
5260fn replacement_expression_has_parse_error(bytes: &[u8], start: usize, end: usize) -> bool {
5261 let Ok(expression) = ::core::str::from_utf8(&bytes[start..end]) else {
5262 return false;
5263 };
5264 parser::parse_expression(expression).is_err()
5265}
5266
5267fn invalid_replacement_expression_start(bytes: &[u8], index: usize, end: usize) -> bool {
5268 if index >= end {
5269 return true;
5270 }
5271
5272 if bytes[index] == b'.'
5275 && (bytes[index..end].starts_with(b"...")
5276 || bytes
5277 .get(index + 1)
5278 .is_some_and(|byte| byte.is_ascii_digit()))
5279 {
5280 return false;
5281 }
5282
5283 if matches!(
5284 bytes[index],
5285 b'.' | b',' | b'*' | b'/' | b'%' | b'&' | b'|' | b'^' | b'<' | b'>' | b'@' | b'=' | b'!'
5286 ) || is_stray_in_replacement_expression(bytes[index])
5287 {
5288 return true;
5289 }
5290
5291 if matches!(bytes[index], b'+' | b'-' | b'~') {
5292 let operand = skip_ascii_whitespace(bytes, index + 1, end);
5293 if starts_identifier(bytes, operand, b"lambda") {
5294 return true;
5295 }
5296 return !bytes.get(operand).is_some_and(|byte| {
5297 *byte >= 0x80
5298 || *byte == b'_'
5299 || byte.is_ascii_alphabetic()
5300 || byte.is_ascii_digit()
5301 || matches!(*byte, b'\'' | b'"' | b'(' | b'[' | b'{')
5302 || (*byte == b'.'
5304 && bytes
5305 .get(operand + 1)
5306 .is_some_and(|next| next.is_ascii_digit()))
5307 });
5308 }
5309
5310 [
5311 b"and".as_slice(),
5312 b"as".as_slice(),
5313 b"else".as_slice(),
5314 b"for".as_slice(),
5315 b"if".as_slice(),
5316 b"in".as_slice(),
5317 b"is".as_slice(),
5318 b"or".as_slice(),
5319 ]
5320 .iter()
5321 .any(|keyword| starts_identifier(bytes, index, keyword))
5322}
5323
5324fn is_replacement_field_marker(bytes: &[u8], index: usize) -> bool {
5328 match bytes[index] {
5329 b'=' => {
5330 if bytes.get(index + 1) == Some(&b'=') {
5331 return false;
5332 }
5333 !index
5334 .checked_sub(1)
5335 .and_then(|previous| bytes.get(previous).copied())
5336 .is_some_and(precedes_equals_in_one_operator)
5337 }
5338 b'!' => bytes.get(index + 1) != Some(&b'='),
5339 _ => true,
5340 }
5341}
5342
5343fn replacement_field_separator(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
5344 let mut level = 0usize;
5345 while index < end {
5346 match bytes[index] {
5347 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
5348 b'#' => index = skip_replacement_field_comment(bytes, index, end),
5349 b'(' | b'[' | b'{' => {
5350 level += 1;
5351 index += 1;
5352 }
5353 b')' | b']' | b'}' if level > 0 => {
5354 level -= 1;
5355 index += 1;
5356 }
5357 b'=' | b'!' | b':' | b'}'
5358 if level == 0 && is_replacement_field_marker(bytes, index) =>
5359 {
5360 return Some(index);
5361 }
5362 _ => index += 1,
5363 }
5364 }
5365 None
5366}
5367
5368const fn is_stray_in_replacement_expression(byte: u8) -> bool {
5371 matches!(byte, b';' | b'$' | b'?' | b'`')
5372}
5373
5374fn replacement_expression_stray_character(
5375 bytes: &[u8],
5376 mut index: usize,
5377 end: usize,
5378) -> Option<usize> {
5379 while index < end {
5380 match bytes[index] {
5381 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
5382 b'#' => index = skip_replacement_field_comment(bytes, index, end),
5383 byte if is_stray_in_replacement_expression(byte) => return Some(index),
5384 _ => index += 1,
5385 }
5386 }
5387 None
5388}
5389
5390const fn precedes_equals_in_one_operator(byte: u8) -> bool {
5395 matches!(
5396 byte,
5397 b'!' | b'%'
5398 | b'&'
5399 | b'*'
5400 | b'+'
5401 | b'-'
5402 | b'/'
5403 | b':'
5404 | b'<'
5405 | b'='
5406 | b'>'
5407 | b'@'
5408 | b'^'
5409 | b'|'
5410 )
5411}
5412
5413const fn is_dangling_operator_byte(byte: u8) -> bool {
5417 matches!(
5418 byte,
5419 b'!' | b'%'
5420 | b'&'
5421 | b'*'
5422 | b'+'
5423 | b'-'
5424 | b'.'
5425 | b'/'
5426 | b'<'
5427 | b'='
5428 | b'>'
5429 | b'@'
5430 | b'^'
5431 | b'|'
5432 | b'~'
5433 )
5434}
5435
5436fn preceding_keyword(bytes: &[u8], start: usize, word: usize) -> Option<(usize, &[u8])> {
5438 let mut cursor = word;
5439 while cursor > start && matches!(bytes[cursor - 1], b' ' | b'\t' | b'\r' | b'\n' | 0x0c) {
5440 cursor -= 1;
5441 }
5442 let stop = cursor;
5443 while cursor > start && is_ascii_identifier_char(bytes[cursor - 1]) {
5444 cursor -= 1;
5445 }
5446 (cursor < stop).then(|| (cursor, &bytes[cursor..stop]))
5447}
5448
5449fn dangling_operator(bytes: &[u8], start: usize, end: usize) -> Option<usize> {
5453 let mut index = start;
5457 let mut tail = start;
5458 while index < end {
5459 match bytes[index] {
5460 b'\'' | b'"' => {
5461 index = skip_quoted_string(bytes, index).min(end);
5462 tail = index;
5463 }
5464 b'#' => index = skip_replacement_field_comment(bytes, index, end),
5465 b' ' | b'\t' | b'\r' | b'\n' | 0x0c => index += 1,
5466 _ => {
5467 index += 1;
5468 tail = index;
5469 }
5470 }
5471 }
5472 if tail == start {
5473 return None;
5474 }
5475
5476 if is_ascii_identifier_char(bytes[tail - 1]) {
5478 let mut word = tail;
5479 while word > start && is_ascii_identifier_char(bytes[word - 1]) {
5480 word -= 1;
5481 }
5482 if ![
5483 b"and".as_slice(),
5484 b"or".as_slice(),
5485 b"not".as_slice(),
5486 b"in".as_slice(),
5487 b"is".as_slice(),
5488 b"if".as_slice(),
5489 b"for".as_slice(),
5490 ]
5491 .iter()
5492 .any(|keyword| &bytes[word..tail] == *keyword)
5493 {
5494 return None;
5495 }
5496 let previous = preceding_keyword(bytes, start, word);
5498 return Some(match (previous, &bytes[word..tail]) {
5499 (Some((at, b"is")), b"not") | (Some((at, b"not")), b"in") => at,
5500 _ => word,
5501 });
5502 }
5503
5504 if !is_dangling_operator_byte(bytes[tail - 1]) {
5505 return None;
5506 }
5507 let mut token = tail;
5508 while token > start && is_dangling_operator_byte(bytes[token - 1]) {
5509 token -= 1;
5510 }
5511
5512 if bytes[token..tail] == *b"." && token > start && bytes[token - 1].is_ascii_digit() {
5514 return None;
5515 }
5516 if bytes[token..tail].iter().all(|byte| *byte == b'.') {
5518 let leftover = (tail - token) % 3;
5519 return (leftover != 0).then(|| tail - leftover);
5520 }
5521 Some(token)
5522}
5523
5524fn invalid_debug_expression_error(
5525 bytes: &[u8],
5526 equals: usize,
5527 end: usize,
5528 prefix: &str,
5529 depth: usize,
5530) -> Option<CpythonDiagnostic> {
5531 let next = equals + 1;
5532 if next >= end {
5533 return None;
5534 }
5535 match bytes[next] {
5538 b'!' => return invalid_conversion_error(bytes, next, end, prefix, depth),
5539 b':' => return invalid_format_spec_error(bytes, next, end, prefix, depth),
5540 b'}' => return None,
5541 _ => {}
5542 }
5543 Some(CpythonDiagnostic::new(
5544 format!("{prefix}: expecting '!', or ':', or '}}'"),
5545 next,
5546 next.saturating_add(1).min(end),
5547 ))
5548}
5549
5550fn invalid_conversion_error(
5551 bytes: &[u8],
5552 bang: usize,
5553 end: usize,
5554 prefix: &str,
5555 depth: usize,
5556) -> Option<CpythonDiagnostic> {
5557 let next = bang + 1;
5558 if next >= end {
5559 return Some(CpythonDiagnostic::new(
5560 format!("{prefix}: expecting '}}'"),
5561 bang,
5562 bang + 1,
5563 ));
5564 }
5565
5566 if bytes[next].is_ascii_whitespace() {
5567 let following = skip_ascii_whitespace(bytes, next, end);
5568 let message = if bytes
5569 .get(following)
5570 .is_some_and(|byte| byte.is_ascii_alphabetic() || *byte == b'_')
5571 {
5572 "conversion type must come right after the exclamation mark"
5573 } else {
5574 "missing conversion character"
5575 };
5576 return Some(CpythonDiagnostic::new(
5577 format!("{prefix}: {message}"),
5578 next,
5579 next + 1,
5580 ));
5581 }
5582
5583 if matches!(bytes[next], b':' | b'}') {
5584 return Some(CpythonDiagnostic::new(
5585 format!("{prefix}: missing conversion character"),
5586 next,
5587 next + 1,
5588 ));
5589 }
5590
5591 if bytes[next] >= 0x80
5594 && let Some(character) = ::core::str::from_utf8(&bytes[next..end])
5595 .ok()
5596 .and_then(|text| text.chars().next())
5597 {
5598 return Some(CpythonDiagnostic::new(
5599 format!(
5600 "{prefix}: invalid conversion character '{character}': expected 's', 'r', or 'a'"
5601 ),
5602 next,
5603 next + character.len_utf8(),
5604 ));
5605 }
5606
5607 if !bytes[next].is_ascii_alphabetic() && bytes[next] != b'_' {
5608 return Some(CpythonDiagnostic::new(
5609 format!("{prefix}: invalid conversion character"),
5610 next,
5611 next + 1,
5612 ));
5613 }
5614
5615 let conversion_end = identifier_end(bytes, next, end);
5616 let conversion = &bytes[next..conversion_end];
5617 if !matches!(conversion, b"s" | b"r" | b"a") {
5618 let conversion = ::core::str::from_utf8(conversion).unwrap_or("");
5619 return Some(CpythonDiagnostic::new(
5620 format!(
5621 "{prefix}: invalid conversion character '{conversion}': expected 's', 'r', or 'a'"
5622 ),
5623 next,
5624 conversion_end,
5625 ));
5626 }
5627
5628 if conversion_end >= end {
5629 return None;
5630 }
5631
5632 match bytes[conversion_end] {
5633 b':' => return invalid_format_spec_error(bytes, conversion_end, end, prefix, depth),
5634 b'}' => return None,
5635 _ => {}
5636 }
5637
5638 Some(CpythonDiagnostic::new(
5639 format!("{prefix}: expecting ':' or '}}'"),
5640 conversion_end,
5641 conversion_end + 1,
5642 ))
5643}
5644
5645fn invalid_format_spec_error(
5646 bytes: &[u8],
5647 colon: usize,
5648 end: usize,
5649 prefix: &str,
5650 depth: usize,
5651) -> Option<CpythonDiagnostic> {
5652 let mut index = colon + 1;
5655 while index < end {
5656 match bytes[index] {
5657 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
5658 b'{' => {
5659 if depth < MAX_REPLACEMENT_FIELD_DEPTH
5660 && let Some(error) =
5661 replacement_field_error(bytes, index, end, prefix, depth + 1)
5662 {
5663 return Some(error);
5664 }
5665 let Some(after) = replacement_field_end(bytes, index, end) else {
5668 break;
5669 };
5670 index = after;
5671 }
5672 b'}' => return None,
5673 _ => index += 1,
5674 }
5675 }
5676 Some(CpythonDiagnostic::new(
5677 format!("{prefix}: expecting '}}', or format specs"),
5678 colon,
5679 colon + 1,
5680 ))
5681}
5682
5683fn replacement_field_closing_brace(bytes: &[u8], mut index: usize, end: usize) -> Option<usize> {
5684 let mut level = 0usize;
5685 while index < end {
5686 match bytes[index] {
5687 b'\'' | b'"' => index = skip_quoted_string(bytes, index),
5688 b'{' => {
5689 level += 1;
5690 index += 1;
5691 }
5692 b'}' if level > 0 => {
5693 level -= 1;
5694 index += 1;
5695 }
5696 b'}' => return Some(index),
5697 _ => index += 1,
5698 }
5699 }
5700 None
5701}
5702
5703fn skip_ascii_whitespace(bytes: &[u8], mut index: usize, end: usize) -> usize {
5704 while index < end && matches!(bytes[index], b' ' | b'\t' | b'\r' | b'\n' | 0x0c) {
5705 index += 1;
5706 }
5707 index
5708}
5709
5710fn string_literal_end_at(bytes: &[u8], index: usize) -> Option<usize> {
5711 match bytes.get(index).copied()? {
5712 b'\'' | b'"' => Some(skip_quoted_string(bytes, index)),
5713 first if first.is_ascii_alphabetic() => {
5714 if matches!(bytes.get(index + 1), Some(b'\'' | b'"')) {
5715 return string_literal_prefix(bytes, index, index + 1)
5716 .then(|| skip_quoted_string(bytes, index + 1));
5717 }
5718 if matches!(bytes.get(index + 2), Some(b'\'' | b'"')) {
5719 return string_literal_prefix(bytes, index, index + 2)
5720 .then(|| skip_quoted_string(bytes, index + 2));
5721 }
5722 None
5723 }
5724 _ => None,
5725 }
5726}
5727
5728fn string_literal_prefix(bytes: &[u8], start: usize, quote: usize) -> bool {
5729 let prefix = &bytes[start..quote];
5730 let valid = matches!(
5731 prefix,
5732 b"b" | b"B"
5733 | b"r"
5734 | b"R"
5735 | b"u"
5736 | b"U"
5737 | b"f"
5738 | b"F"
5739 | b"t"
5740 | b"T"
5741 | b"br"
5742 | b"bR"
5743 | b"Br"
5744 | b"BR"
5745 | b"rb"
5746 | b"rB"
5747 | b"Rb"
5748 | b"RB"
5749 | b"fr"
5750 | b"fR"
5751 | b"Fr"
5752 | b"FR"
5753 | b"rf"
5754 | b"rF"
5755 | b"Rf"
5756 | b"RF"
5757 | b"tr"
5758 | b"tR"
5759 | b"Tr"
5760 | b"TR"
5761 | b"rt"
5762 | b"rT"
5763 | b"Rt"
5764 | b"RT"
5765 );
5766 valid && (start == 0 || !is_ascii_identifier_char(bytes[start - 1]))
5767}
5768
5769fn invalid_expression_error(source: &str) -> Option<CpythonDiagnostic> {
5770 invalid_string_expression_error(source).or_else(|| missing_comma_expression_error(source))
5771}
5772
5773fn invalid_string_expression_error(source: &str) -> Option<CpythonDiagnostic> {
5774 let bytes = source.as_bytes();
5775 let mut index = 0;
5776 while index < bytes.len() {
5777 if let Some(first_string_end) = string_literal_end_at(bytes, index) {
5778 let expr_start = skip_ascii_whitespace(bytes, first_string_end, bytes.len());
5779 if expression_atom_start(bytes, expr_start)
5780 && let Some(expr_end) = adjacent_atom_end(bytes, expr_start)
5781 {
5782 let next = skip_ascii_whitespace(bytes, expr_end, bytes.len());
5783 if string_literal_end_at(bytes, next).is_some() {
5784 return Some(CpythonDiagnostic::new(
5785 "invalid syntax. Is this intended to be part of the string?".to_owned(),
5786 expr_start,
5787 expr_end,
5788 ));
5789 }
5790 }
5791 index = first_string_end;
5792 } else {
5793 index += 1;
5794 }
5795 }
5796 None
5797}
5798
5799fn missing_comma_expression_error(source: &str) -> Option<CpythonDiagnostic> {
5800 let bytes = source.as_bytes();
5801 let mut stack: Vec<u8> = Vec::new();
5802 let mut index = 0;
5803 while index < bytes.len() {
5804 if bytes[index] == b'#' {
5805 while index < bytes.len() && bytes[index] != b'\n' {
5806 index += 1;
5807 }
5808 } else if let Some(string_end) = string_literal_end_at(bytes, index) {
5809 index = string_end;
5810 } else {
5811 match bytes[index] {
5812 b'(' | b'[' | b'{' => {
5813 if bytes[index] == b'[' && opening_bracket_is_class_type_params(bytes, index) {
5814 let Some(close) = matching_delimiter(bytes, index, b']') else {
5815 index += 1;
5816 continue;
5817 };
5818 index = close + 1;
5819 continue;
5820 }
5821 stack.push(bytes[index]);
5822 index += 1;
5823 }
5824 b')' | b']' | b'}' => {
5825 stack.pop();
5826 index += 1;
5827 }
5828 _ if !stack.is_empty() && expression_continuation_keyword(bytes, index) => {
5829 index = identifier_end(bytes, index, bytes.len());
5830 }
5831 byte if !stack.is_empty() && expression_atom_start_byte(byte) => {
5832 if starts_identifier(bytes, index, b"yield") {
5835 index = identifier_end(bytes, index, bytes.len());
5836 let after_yield = skip_ascii_whitespace(bytes, index, bytes.len());
5837 if starts_identifier(bytes, after_yield, b"from") {
5838 index = identifier_end(bytes, after_yield, bytes.len());
5839 }
5840 continue;
5841 }
5842 if starts_identifier(bytes, index, b"await") {
5843 index = identifier_end(bytes, index, bytes.len());
5844 continue;
5845 }
5846 let atom_end = adjacent_atom_end(bytes, index).unwrap_or(index + 1);
5847 let next = skip_ascii_whitespace(bytes, atom_end, bytes.len());
5848 if next > atom_end
5849 && expression_atom_start(bytes, next)
5850 && !expression_continuation_keyword(bytes, next)
5851 {
5852 let second_end = adjacent_atom_end(bytes, next).unwrap_or(next + 1);
5857 return Some(CpythonDiagnostic::new(
5858 "invalid syntax. Perhaps you forgot a comma?".to_owned(),
5859 index,
5860 second_end,
5861 ));
5862 }
5863 index = atom_end;
5864 }
5865 _ => index += 1,
5866 }
5867 }
5868 }
5869 None
5870}
5871
5872fn opening_bracket_is_class_type_params(bytes: &[u8], bracket: usize) -> bool {
5873 let mut cursor = bracket;
5874 while cursor > 0 && matches!(bytes[cursor - 1], b' ' | b'\t' | b'\x0c') {
5875 cursor -= 1;
5876 }
5877 while cursor > 0
5878 && bytes
5879 .get(cursor - 1)
5880 .is_some_and(|byte| *byte >= 0x80 || is_ascii_identifier_char(*byte))
5881 {
5882 cursor -= 1;
5883 }
5884 while cursor > 0 && matches!(bytes[cursor - 1], b' ' | b'\t' | b'\x0c') {
5885 cursor -= 1;
5886 }
5887 cursor >= 5 && starts_identifier(bytes, cursor - 5, b"class")
5888}
5889
5890fn expression_continuation_keyword(bytes: &[u8], index: usize) -> bool {
5891 [
5892 b"and".as_slice(),
5893 b"else".as_slice(),
5894 b"for".as_slice(),
5895 b"if".as_slice(),
5896 b"in".as_slice(),
5897 b"is".as_slice(),
5898 b"not".as_slice(),
5899 b"or".as_slice(),
5900 ]
5901 .iter()
5902 .any(|keyword| starts_identifier(bytes, index, keyword))
5903}
5904
5905fn expression_atom_start(bytes: &[u8], index: usize) -> bool {
5906 bytes
5907 .get(index)
5908 .is_some_and(|byte| expression_atom_start_byte(*byte))
5909 || string_literal_end_at(bytes, index).is_some()
5910}
5911
5912fn expression_atom_start_byte(byte: u8) -> bool {
5913 byte >= 0x80
5914 || byte == b'_'
5915 || byte.is_ascii_alphabetic()
5916 || byte.is_ascii_digit()
5917 || matches!(byte, b'\'' | b'"' | b'(' | b'[' | b'{')
5918}
5919
5920fn adjacent_atom_end(bytes: &[u8], index: usize) -> Option<usize> {
5921 if let Some(string_end) = string_literal_end_at(bytes, index) {
5922 return Some(string_end);
5923 }
5924 match bytes.get(index).copied()? {
5925 byte if byte >= 0x80 || byte == b'_' || byte.is_ascii_alphabetic() => {
5926 Some(identifier_end(bytes, index, bytes.len()))
5927 }
5928 byte if byte.is_ascii_digit() => {
5929 let mut end = index + 1;
5930 while end < bytes.len() && (bytes[end].is_ascii_alphanumeric() || bytes[end] == b'_') {
5931 end += 1;
5932 }
5933 Some(end)
5934 }
5935 b'(' | b'[' | b'{' => Some(index + 1),
5936 _ => None,
5937 }
5938}
5939
5940struct OpenDelimiter {
5941 position: usize,
5942 opener: u8,
5943 in_format_spec: bool,
5944}
5945
5946fn unclosed_replacement_field_error(
5952 bytes: &[u8],
5953 quote_start: usize,
5954 content_start: usize,
5955 content_end: usize,
5956) -> Option<CpythonDiagnostic> {
5957 interpolated_string_prefix(bytes, quote_start)?;
5958
5959 let mut open: Vec<OpenDelimiter> = Vec::new();
5963 let mut index = content_start;
5964 while index < content_end {
5965 let inside_expression = matches!(
5966 open.last(),
5967 Some(OpenDelimiter {
5968 opener: b'{',
5969 in_format_spec: false,
5970 ..
5971 })
5972 );
5973 match bytes[index] {
5974 b'{' if open.is_empty() && bytes.get(index + 1) == Some(&b'{') => index += 2,
5976 b'}' if open.is_empty() && bytes.get(index + 1) == Some(&b'}') => index += 2,
5977 b'{' => {
5978 open.push(OpenDelimiter {
5979 position: index,
5980 opener: b'{',
5981 in_format_spec: false,
5982 });
5983 index += 1;
5984 }
5985 b'}' => {
5986 open.pop();
5987 index += 1;
5988 }
5989 byte @ (b'(' | b'[') if !open.is_empty() => {
5990 open.push(OpenDelimiter {
5991 position: index,
5992 opener: byte,
5993 in_format_spec: false,
5994 });
5995 index += 1;
5996 }
5997 b')' | b']' if !open.is_empty() => {
5998 open.pop();
5999 index += 1;
6000 }
6001 b':' if inside_expression => {
6004 open.last_mut().expect("a brace is open").in_format_spec = true;
6005 index += 1;
6006 }
6007 b'\'' | b'"' if !open.is_empty() => {
6008 index = skip_quoted_string(bytes, index).min(content_end);
6009 }
6010 b'#' if inside_expression => {
6011 index = skip_replacement_field_comment(bytes, index, content_end);
6012 }
6013 _ => index += 1,
6014 }
6015 }
6016
6017 let &OpenDelimiter {
6020 position,
6021 opener,
6022 in_format_spec,
6023 } = open.last()?;
6024 (!in_format_spec).then(|| {
6025 CpythonDiagnostic::new(
6026 format!("'{}' was never closed", opener as char),
6027 position,
6028 position + 1,
6029 )
6030 .with_unclosed_bracket()
6031 })
6032}
6033
6034fn unterminated_string_message(
6035 detected_line: usize,
6036 triple: bool,
6037 has_escaped_quote: bool,
6038 prefix: Option<&str>,
6039) -> String {
6040 let kind = prefix.unwrap_or("string");
6042 if triple {
6043 format!("unterminated triple-quoted {kind} literal (detected at line {detected_line})")
6044 } else if has_escaped_quote && prefix.is_none() {
6048 format!(
6049 "unterminated {kind} literal (detected at line {detected_line}); perhaps you escaped the end quote?"
6050 )
6051 } else {
6052 format!("unterminated {kind} literal (detected at line {detected_line})")
6053 }
6054}
6055
6056fn expected_opening_bracket(closing: char) -> char {
6057 match closing {
6058 ')' => '(',
6059 ']' => '[',
6060 '}' => '{',
6061 _ => unreachable!(),
6062 }
6063}
6064
6065#[derive(Clone)]
6068struct BracketError {
6069 diagnostic: CpythonDiagnostic,
6070 unclosed: bool,
6071}
6072
6073fn bracket_syntax_error(source: &str) -> Option<BracketError> {
6074 let mut stack: Vec<(char, usize, usize)> = Vec::new();
6075 let mut in_string = false;
6076 let mut string_quote = '\0';
6077 let mut triple_quote = false;
6078 let mut escape_next = false;
6079 let mut is_raw_string = false;
6080 let mut line = 1usize;
6081
6082 let chars: Vec<(usize, char)> = source.char_indices().collect();
6083 let mut index = 0;
6084 while index < chars.len() {
6085 let (byte_offset, ch) = chars[index];
6086
6087 if ch == '\n' {
6088 line += 1;
6089 }
6090
6091 if escape_next {
6092 escape_next = false;
6093 index += 1;
6094 continue;
6095 }
6096
6097 if in_string {
6098 if ch == '\\' && !is_raw_string {
6099 escape_next = true;
6100 } else if triple_quote {
6101 if ch == string_quote
6102 && index + 2 < chars.len()
6103 && chars[index + 1].1 == string_quote
6104 && chars[index + 2].1 == string_quote
6105 {
6106 in_string = false;
6107 index += 3;
6108 continue;
6109 }
6110 } else if ch == string_quote {
6111 in_string = false;
6112 }
6113 index += 1;
6114 continue;
6115 }
6116
6117 if ch == '#' {
6118 while index < chars.len() && chars[index].1 != '\n' {
6119 index += 1;
6120 }
6121 continue;
6122 }
6123
6124 if ch == '\\' {
6125 match chars.get(index + 1).map(|(_, next)| *next) {
6126 Some('\n' | '\r') => {
6127 escape_next = true;
6128 index += 1;
6129 continue;
6130 }
6131 Some(_) => {
6132 index += 2;
6133 continue;
6134 }
6135 None => {
6136 index += 1;
6137 continue;
6138 }
6139 }
6140 }
6141
6142 if ch == '\'' || ch == '"' {
6143 is_raw_string = false;
6144 for look_back in 1..=2.min(index) {
6145 let prev = chars[index - look_back].1;
6146 if matches!(prev, 'r' | 'R') {
6147 is_raw_string = true;
6148 break;
6149 }
6150 if !matches!(prev, 'b' | 'B' | 'f' | 'F' | 'u' | 'U') {
6151 break;
6152 }
6153 }
6154 string_quote = ch;
6155 if index + 2 < chars.len() && chars[index + 1].1 == ch && chars[index + 2].1 == ch {
6156 triple_quote = true;
6157 in_string = true;
6158 index += 3;
6159 continue;
6160 }
6161 triple_quote = false;
6162 in_string = true;
6163 index += 1;
6164 continue;
6165 }
6166
6167 match ch {
6168 '(' | '[' | '{' => stack.push((ch, byte_offset, line)),
6169 ')' | ']' | '}' => {
6170 let expected = expected_opening_bracket(ch);
6171 let Some(&(opening, _, opening_line)) = stack.last() else {
6172 return Some(BracketError {
6173 diagnostic: CpythonDiagnostic::new(
6174 format!("unmatched '{ch}'"),
6175 byte_offset,
6176 byte_offset,
6177 ),
6178 unclosed: false,
6179 });
6180 };
6181 if opening == expected {
6182 stack.pop();
6183 } else {
6184 let suffix = if opening_line != line {
6185 format!(" on line {opening_line}")
6186 } else {
6187 String::new()
6188 };
6189 return Some(BracketError {
6190 diagnostic: CpythonDiagnostic::new(
6191 format!(
6192 "closing parenthesis '{ch}' does not match opening parenthesis '{opening}'{suffix}"
6193 ),
6194 byte_offset,
6195 byte_offset,
6196 ),
6197 unclosed: false,
6198 });
6199 }
6200 }
6201 _ => {}
6202 }
6203
6204 index += 1;
6205 }
6206
6207 stack.last().map(|(opening, byte_offset, _)| BracketError {
6208 diagnostic: CpythonDiagnostic::new(
6209 format!("'{opening}' was never closed"),
6210 *byte_offset,
6211 *byte_offset,
6212 ),
6213 unclosed: true,
6214 })
6215}
6216
6217fn is_legacy_statement_expression_start(byte: u8) -> bool {
6218 byte >= 0x80
6219 || byte == b'_'
6220 || byte.is_ascii_alphabetic()
6221 || byte.is_ascii_digit()
6222 || matches!(byte, b'\'' | b'"' | b'{' | b'[')
6223}
6224
6225fn legacy_statement_container_has_invalid_attribute(bytes: &[u8], start: usize) -> bool {
6226 let Some(&opening) = bytes.get(start) else {
6227 return false;
6228 };
6229 if !matches!(opening, b'{' | b'[') {
6230 return false;
6231 }
6232
6233 let mut index = start;
6234 let mut level = 0usize;
6235 while index < bytes.len() {
6236 match bytes[index] {
6237 b'#' => {
6238 while index < bytes.len() && bytes[index] != b'\n' {
6239 index += 1;
6240 }
6241 }
6242 b'\n' | b';' if level == 0 => return false,
6243 b'\'' | b'"' => {
6244 index = skip_quoted_string(bytes, index);
6245 }
6246 b'(' | b'[' | b'{' => {
6247 level += 1;
6248 index += 1;
6249 }
6250 b')' | b']' | b'}' => {
6251 level = level.saturating_sub(1);
6252 index += 1;
6253 if level == 0 {
6254 return false;
6255 }
6256 }
6257 b'.' => {
6258 let mut cursor = index + 1;
6259 while matches!(bytes.get(cursor), Some(b' ' | b'\t' | b'\x0c')) {
6260 cursor += 1;
6261 }
6262 if matches!(bytes.get(cursor), Some(b')' | b']' | b'}')) {
6263 return true;
6264 }
6265 index += 1;
6266 }
6267 _ => index += 1,
6268 }
6269 }
6270 false
6271}
6272
6273fn invalid_legacy_statement_error(source: &str) -> Option<CpythonDiagnostic> {
6274 let bytes = source.as_bytes();
6275 let mut index = 0;
6276 while index < bytes.len() {
6277 match bytes[index] {
6278 b'#' => {
6279 while index < bytes.len() && bytes[index] != b'\n' {
6280 index += 1;
6281 }
6282 }
6283 b'\'' | b'"' => {
6284 index = skip_quoted_string(bytes, index);
6285 }
6286 b'p' | b'e' => {
6287 let keyword = if starts_identifier(bytes, index, b"print") {
6288 Some("print")
6289 } else if starts_identifier(bytes, index, b"exec") {
6290 Some("exec")
6291 } else {
6292 None
6293 };
6294 let Some(keyword) = keyword else {
6295 index += 1;
6296 continue;
6297 };
6298 let after_keyword = index + keyword.len();
6299 if !matches!(bytes.get(after_keyword), Some(b' ' | b'\t' | b'\x0c')) {
6300 index = after_keyword;
6301 continue;
6302 }
6303 let mut cursor = after_keyword;
6304 while matches!(bytes.get(cursor), Some(b' ' | b'\t' | b'\x0c')) {
6305 cursor += 1;
6306 }
6307 if legacy_statement_container_has_invalid_attribute(bytes, cursor) {
6308 index = after_keyword;
6309 continue;
6310 }
6311 if bytes.get(cursor).is_some_and(|byte| {
6312 *byte != b'(' && is_legacy_statement_expression_start(*byte)
6313 }) {
6314 return Some(CpythonDiagnostic::new(
6315 format!(
6316 "Missing parentheses in call to '{keyword}'. Did you mean {keyword}(...)?"
6317 ),
6318 index,
6319 after_keyword,
6320 ));
6321 }
6322 index = after_keyword;
6323 }
6324 _ => index += 1,
6325 }
6326 }
6327 None
6328}
6329
6330#[must_use]
6336pub fn long_decimal_integer_literal_error(
6337 source_file: &SourceFile,
6338 tokens: &Tokens,
6339 max_str_digits: usize,
6340) -> Option<CompileError> {
6341 if max_str_digits == 0 {
6342 return None;
6343 }
6344 tokens.iter().find_map(|token| {
6345 if token.kind() != TokenKind::Int {
6346 return None;
6347 }
6348 let literal = source_file.source_text().slice(token.range());
6349 if literal
6350 .as_bytes()
6351 .get(..2)
6352 .is_some_and(|prefix| matches!(prefix, b"0x" | b"0X" | b"0o" | b"0O" | b"0b" | b"0B"))
6353 {
6354 return None;
6355 }
6356 let digits = literal.bytes().filter(u8::is_ascii_digit).count();
6357 (digits > max_str_digits).then(|| {
6358 let start = token.range().start().to_usize();
6359 CompileError::from_source_error(source_file, CpythonDiagnostic::new(format!(
6360 "Exceeds the limit ({max_str_digits} digits) for integer string conversion: value has {digits} digits; use sys.set_int_max_str_digits() to increase the limit - Consider hexadecimal for huge integer literals to avoid decimal conversion limits."
6361 ), start, start))
6362 })
6363 })
6364}
6365
6366fn invalid_parenthesized_import_star_error(source: &str) -> Option<CpythonDiagnostic> {
6367 let bytes = source.as_bytes();
6368 let mut index = 0;
6369 while index < bytes.len() {
6370 match bytes[index] {
6371 b'#' => {
6372 while index < bytes.len() && bytes[index] != b'\n' {
6373 index += 1;
6374 }
6375 }
6376 b'\'' | b'"' => {
6377 index = skip_quoted_string(bytes, index);
6378 }
6379 b'f' if starts_identifier(bytes, index, b"from") => {
6380 let mut cursor = index + 4;
6381 while cursor < bytes.len() && !matches!(bytes[cursor], b'\n' | b';') {
6382 if starts_identifier(bytes, cursor, b"import") {
6383 cursor += 6;
6384 while matches!(bytes.get(cursor), Some(b' ' | b'\t' | b'\r')) {
6385 cursor += 1;
6386 }
6387 if bytes.get(cursor) == Some(&b'(') {
6388 cursor += 1;
6389 while cursor < bytes.len()
6390 && !matches!(bytes[cursor], b')' | b'\n' | b';')
6391 {
6392 if bytes[cursor] == b'*' {
6393 return Some(CpythonDiagnostic::new(
6394 "invalid syntax".to_owned(),
6395 cursor,
6396 cursor + 1,
6397 ));
6398 }
6399 cursor += 1;
6400 }
6401 }
6402 break;
6403 }
6404 cursor += 1;
6405 }
6406 index = cursor;
6407 }
6408 _ => index += 1,
6409 }
6410 }
6411 None
6412}
6413
6414fn prefix_letters_before_quote(bytes: &[u8], quote_index: usize) -> Option<(usize, Vec<u8>)> {
6415 let mut letters = Vec::new();
6416 let mut index = quote_index;
6417 while index > 0 {
6418 let byte = bytes[index - 1];
6419 if !matches!(
6420 byte,
6421 b'r' | b'R' | b'b' | b'B' | b'u' | b'U' | b'f' | b'F' | b't' | b'T'
6422 ) {
6423 break;
6424 }
6425 letters.push(byte.to_ascii_lowercase());
6426 index -= 1;
6427 }
6428 if letters.is_empty() {
6429 return None;
6430 }
6431 if index > 0 && (bytes[index - 1] == b'_' || bytes[index - 1].is_ascii_alphabetic()) {
6432 return None;
6433 }
6434 letters.reverse();
6435 Some((index, letters))
6436}
6437
6438fn incompatible_string_prefix_error(source: &str) -> Option<CpythonDiagnostic> {
6439 let bytes = source.as_bytes();
6440 let mut index = 0;
6441 while index < bytes.len() {
6442 match bytes[index] {
6443 b'#' => {
6444 while index < bytes.len() && bytes[index] != b'\n' {
6445 index += 1;
6446 }
6447 }
6448 b'\'' | b'"' => {
6449 if let Some((start, letters)) = prefix_letters_before_quote(bytes, index)
6450 && let Some(message) = incompatible_prefix_message(&letters)
6451 {
6452 return Some(CpythonDiagnostic::new(message, start, start + 1));
6453 }
6454 index = skip_string_token(bytes, index);
6455 }
6456 _ => index += 1,
6457 }
6458 }
6459 None
6460}
6461
6462fn incompatible_prefix_message(letters: &[u8]) -> Option<String> {
6463 if letters.len() < 2 {
6464 return None;
6465 }
6466 let mut seen_r = false;
6467 let mut seen_b = false;
6468 let mut seen_u = false;
6469 let mut seen_f = false;
6470 let mut seen_t = false;
6471 for &letter in letters {
6472 match letter {
6473 b'r' if seen_r => return None,
6474 b'b' if seen_b => return None,
6475 b'u' if seen_u => return None,
6476 b'f' if seen_f => return None,
6477 b't' if seen_t => return None,
6478 b'r' => seen_r = true,
6479 b'b' => seen_b = true,
6480 b'u' => seen_u = true,
6481 b'f' => seen_f = true,
6482 b't' => seen_t = true,
6483 _ => {}
6484 }
6485 }
6486 let pair = if seen_u && seen_b {
6487 ("u", "b")
6488 } else if seen_u && seen_r {
6489 ("u", "r")
6490 } else if seen_u && seen_f {
6491 ("u", "f")
6492 } else if seen_u && seen_t {
6493 ("u", "t")
6494 } else if seen_b && seen_f {
6495 ("b", "f")
6496 } else if seen_b && seen_t {
6497 ("b", "t")
6498 } else if seen_f && seen_t {
6499 ("f", "t")
6500 } else {
6501 return None;
6502 };
6503 Some(format!(
6504 "'{}' and '{}' prefixes are incompatible",
6505 pair.0, pair.1
6506 ))
6507}
6508
6509fn malformed_unicode_n_escape_error(source: &str) -> Option<CpythonDiagnostic> {
6510 let bytes = source.as_bytes();
6511 let mut index = 0;
6512 while index < bytes.len() {
6513 match bytes[index] {
6514 b'#' => {
6515 while index < bytes.len() && bytes[index] != b'\n' {
6516 index += 1;
6517 }
6518 }
6519 b'\'' | b'"' => {
6520 let quote_index = index;
6521 let interpolated = interpolated_string_prefix(bytes, quote_index).is_some();
6522 let Some((content_start, content_end)) =
6523 quoted_string_content_range(bytes, quote_index, bytes[quote_index])
6524 else {
6525 index = skip_quoted_string(bytes, quote_index);
6526 continue;
6527 };
6528 if let Some(error) = malformed_unicode_n_in_content(
6529 bytes,
6530 content_start,
6531 content_end,
6532 quote_index,
6533 interpolated,
6534 ) {
6535 return Some(error);
6536 }
6537 index = skip_string_token(bytes, quote_index);
6538 }
6539 _ => index += 1,
6540 }
6541 }
6542 None
6543}
6544
6545fn malformed_unicode_n_in_content(
6546 bytes: &[u8],
6547 start: usize,
6548 end: usize,
6549 quote_index: usize,
6550 interpolated: bool,
6551) -> Option<CpythonDiagnostic> {
6552 let mut index = start;
6553 let mut brace_depth = 0usize;
6554 while index + 1 < end {
6555 if interpolated && brace_depth == 0 && bytes[index] == b'{' {
6556 if bytes.get(index + 1) == Some(&b'{') {
6557 index += 2;
6558 continue;
6559 }
6560 brace_depth = 1;
6561 index += 1;
6562 continue;
6563 }
6564 if interpolated && brace_depth > 0 {
6565 match bytes[index] {
6566 b'\'' | b'"' => index = skip_string_token(bytes, index).max(index + 1),
6567 b'{' => {
6568 brace_depth += 1;
6569 index += 1;
6570 }
6571 b'}' => {
6572 brace_depth -= 1;
6573 index += 1;
6574 }
6575 _ => index += 1,
6576 }
6577 continue;
6578 }
6579 if bytes[index] == b'\\' && bytes[index + 1] == b'N' {
6580 let escape_start = index;
6581 if bytes.get(index + 2) == Some(&b'{') {
6582 let mut look = index + 3;
6583 while look < end && bytes[look] != b'}' {
6584 look += 1;
6585 }
6586 if look >= end {
6587 return Some(unicode_n_diagnostic(escape_start, end, start, quote_index));
6588 }
6589 index = look + 1;
6590 continue;
6591 }
6592 return Some(unicode_n_diagnostic(
6593 escape_start,
6594 escape_start + 2,
6595 start,
6596 quote_index,
6597 ));
6598 }
6599 if interpolated && bytes[index] == b'}' && bytes.get(index + 1) == Some(&b'}') {
6600 index += 2;
6601 continue;
6602 }
6603 index += 1;
6604 }
6605 None
6606}
6607
6608fn unicode_n_diagnostic(
6609 escape_start: usize,
6610 escape_end: usize,
6611 content_start: usize,
6612 quote_index: usize,
6613) -> CpythonDiagnostic {
6614 let start = escape_start - content_start;
6615 let end = escape_end.saturating_sub(content_start).saturating_sub(1);
6616 CpythonDiagnostic::new(
6617 format!(
6618 "(unicode error) 'unicodeescape' codec can't decode bytes in position {start}-{end}: malformed \\N character escape"
6619 ),
6620 quote_index,
6621 quote_index + 1,
6622 )
6623}
6624
6625fn too_many_nested_interpolated_strings(source: &str) -> Option<CpythonDiagnostic> {
6626 too_many_nested_interpolated_strings_in(source.as_bytes(), 0, source.len(), 0)
6627}
6628
6629fn too_many_nested_interpolated_strings_in(
6630 bytes: &[u8],
6631 mut index: usize,
6632 end: usize,
6633 depth: usize,
6634) -> Option<CpythonDiagnostic> {
6635 while index < end {
6636 match bytes[index] {
6637 b'#' => {
6638 while index < end && bytes[index] != b'\n' {
6639 index += 1;
6640 }
6641 }
6642 b'\'' | b'"' => {
6643 if interpolated_string_prefix(bytes, index).is_some() {
6644 if depth + 1 >= MAXFSTRINGLEVEL {
6645 return Some(CpythonDiagnostic::new(
6646 "too many nested f-strings or t-strings".to_owned(),
6647 index,
6648 index + 1,
6649 ));
6650 }
6651 if let Some((content_start, content_end)) =
6652 interpolated_string_content_range(bytes, index)
6653 && let Some(error) = too_many_nested_interpolated_strings_in(
6654 bytes,
6655 content_start,
6656 content_end,
6657 depth + 1,
6658 )
6659 {
6660 return Some(error);
6661 }
6662 }
6663 index = skip_string_token(bytes, index).max(index + 1);
6664 }
6665 _ => index += 1,
6666 }
6667 }
6668 None
6669}
6670
6671fn too_many_nested_parentheses_error(source: &str) -> Option<CpythonDiagnostic> {
6672 const MAXLEVEL: usize = 200;
6673
6674 let bytes = source.as_bytes();
6675 let mut index = 0;
6676 let mut level = 0usize;
6677 while index < bytes.len() {
6678 match bytes[index] {
6679 b'#' => {
6680 while index < bytes.len() && bytes[index] != b'\n' {
6681 index += 1;
6682 }
6683 }
6684 b'\'' | b'"' => {
6685 index = skip_quoted_string(bytes, index);
6686 }
6687 b'(' | b'[' | b'{' => {
6688 if level >= MAXLEVEL {
6689 return Some(CpythonDiagnostic::new(
6690 "too many nested parentheses".to_owned(),
6691 index,
6692 index + 1,
6693 ));
6694 }
6695 level += 1;
6696 index += 1;
6697 }
6698 b')' | b']' | b'}' => {
6699 level = level.saturating_sub(1);
6700 index += 1;
6701 }
6702 _ => index += 1,
6703 }
6704 }
6705 None
6706}
6707
6708fn invalid_unparenthesized_yield_after_comma_error(source: &str) -> Option<CpythonDiagnostic> {
6709 let bytes = source.as_bytes();
6710 let mut index = 0;
6711 while index < bytes.len() {
6712 match bytes[index] {
6713 b'#' => {
6714 while index < bytes.len() && bytes[index] != b'\n' {
6715 index += 1;
6716 }
6717 }
6718 b'\'' | b'"' => {
6719 index = skip_quoted_string(bytes, index);
6720 }
6721 b',' => {
6722 let mut cursor = index + 1;
6723 while matches!(bytes.get(cursor), Some(b' ' | b'\t' | b'\x0c')) {
6724 cursor += 1;
6725 }
6726 if starts_identifier(bytes, cursor, b"yield") {
6727 return Some(CpythonDiagnostic::new(
6728 "invalid syntax".to_owned(),
6729 cursor,
6730 cursor + 5,
6731 ));
6732 }
6733 index += 1;
6734 }
6735 _ => index += 1,
6736 }
6737 }
6738 None
6739}
6740
6741const fn is_ascii_tokenizer_whitespace(c: char) -> bool {
6743 matches!(c, ' ' | '\t' | '\n' | '\r' | '\x0c')
6744}
6745
6746#[doc(hidden)]
6748#[must_use]
6749pub fn is_blank_python_source(source: &str) -> bool {
6750 source.lines().all(|line| {
6751 let trimmed = line.trim_matches(is_ascii_tokenizer_whitespace);
6752 trimmed.is_empty() || trimmed.starts_with('#')
6753 })
6754}
6755
6756fn single_mode_blank_source_error(source_file: &SourceFile) -> Option<CompileError> {
6757 let source = source_file.source_text();
6758 if !is_blank_python_source(source) {
6759 return None;
6760 }
6761 let has_indent_only_line = source.lines().any(|line| {
6762 line.trim_matches(is_ascii_tokenizer_whitespace).is_empty()
6763 && line.chars().any(|c| matches!(c, ' ' | '\t'))
6764 });
6765 if has_indent_only_line {
6766 let (location, end_location) =
6767 source_locations(source_file, TextSize::new(0), TextSize::new(0));
6768 return Some(CompileError::Parse(ParseError {
6769 error: parser::ParseErrorType::UnexpectedIndentation,
6770 raw_location: ruff_text_size::TextRange::new(TextSize::new(0), TextSize::new(0)),
6771 location,
6772 end_location,
6773 source_path: source_file.name().to_owned(),
6774 is_unclosed_bracket: false,
6775 is_unclosed_string: false,
6776 }));
6777 }
6778 Some(CompileError::from_source_error(
6779 source_file,
6780 CpythonDiagnostic::new("invalid syntax".to_owned(), 0, 0),
6781 ))
6782}
6783
6784#[doc(hidden)]
6789#[must_use]
6790pub fn leading_byte_order_mark_error(source_file: &SourceFile) -> Option<CompileError> {
6791 source_file.source_text().starts_with('\u{feff}').then(|| {
6792 CompileError::from_source_error(
6793 source_file,
6794 CpythonDiagnostic::new("invalid non-printable character U+FEFF".to_owned(), 0, 0),
6795 )
6796 })
6797}
6798
6799pub fn pre_parse_source_error(source_file: &SourceFile) -> Result<(), CompileError> {
6804 match too_many_nested_parentheses_error(source_file.source_text()) {
6805 Some(error) => Err(CompileError::from_source_error(source_file, error)),
6806 None => Ok(()),
6807 }
6808}
6809
6810fn post_parse_source_error(
6811 source_file: &SourceFile,
6812 tokens: &Tokens,
6813 opts: &CompileOpts,
6814) -> Option<CompileError> {
6815 if let Some(error) = leading_byte_order_mark_error(source_file) {
6816 return Some(error);
6817 }
6818 if let Some(error) = too_many_nested_interpolated_strings(source_file.source_text()) {
6819 return Some(CompileError::from_source_error(source_file, error));
6820 }
6821 if let Some(error) =
6822 long_decimal_integer_literal_error(source_file, tokens, opts.int_max_str_digits)
6823 {
6824 return Some(error);
6825 }
6826 invalid_call_argument_error(source_file.source_text())
6827 .or_else(|| invalid_match_mapping_rest_wildcard_error(source_file.source_text()))
6828 .or_else(|| invalid_match_as_target_error(source_file.source_text()))
6829 .or_else(|| invalid_unparenthesized_yield_after_comma_error(source_file.source_text()))
6830 .or_else(|| invalid_parenthesized_import_star_error(source_file.source_text()))
6831 .map(|error| CompileError::from_source_error(source_file, error))
6832}
6833
6834fn is_compound_stmt(stmt: &ast::Stmt) -> bool {
6835 matches!(
6836 stmt,
6837 ast::Stmt::FunctionDef(_)
6838 | ast::Stmt::ClassDef(_)
6839 | ast::Stmt::If(_)
6840 | ast::Stmt::For(_)
6841 | ast::Stmt::While(_)
6842 | ast::Stmt::With(_)
6843 | ast::Stmt::Try(_)
6844 | ast::Stmt::Match(_)
6845 )
6846}
6847
6848pub fn too_deeply_nested_error(
6855 ast: &ast::Mod,
6856 source_file: &SourceFile,
6857 limit: usize,
6858) -> Result<(), CompileError> {
6859 use ast::visitor::Visitor;
6860
6861 struct DepthChecker {
6862 depth: usize,
6863 limit: usize,
6864 too_deep: Option<ruff_text_size::TextRange>,
6865 }
6866
6867 impl DepthChecker {
6868 fn descend(&mut self, range: ruff_text_size::TextRange, walk: impl FnOnce(&mut Self)) {
6869 if self.too_deep.is_some() {
6870 return;
6871 }
6872 if self.depth >= self.limit {
6873 self.too_deep = Some(range);
6874 return;
6875 }
6876 self.depth += 1;
6877 walk(self);
6878 self.depth -= 1;
6879 }
6880 }
6881
6882 impl<'a> Visitor<'a> for DepthChecker {
6883 fn visit_stmt(&mut self, stmt: &'a ast::Stmt) {
6884 self.descend(stmt.range(), |checker| {
6885 ast::visitor::walk_stmt(checker, stmt);
6886 });
6887 }
6888
6889 fn visit_expr(&mut self, expr: &'a ast::Expr) {
6890 self.descend(expr.range(), |checker| {
6891 ast::visitor::walk_expr(checker, expr);
6892 });
6893 }
6894
6895 fn visit_pattern(&mut self, pattern: &'a ast::Pattern) {
6896 self.descend(pattern.range(), |checker| {
6897 ast::visitor::walk_pattern(checker, pattern);
6898 });
6899 }
6900 }
6901
6902 let mut checker = DepthChecker {
6903 depth: 0,
6904 limit,
6905 too_deep: None,
6906 };
6907 match ast {
6908 ast::Mod::Module(module) => checker.visit_body(&module.body),
6909 ast::Mod::Expression(expression) => checker.visit_expr(&expression.body),
6910 }
6911
6912 let Some(range) = checker.too_deep else {
6913 return Ok(());
6914 };
6915 let (location, end_location) = source_locations(source_file, range.start(), range.end());
6916 Err(CompileError::Codegen(codegen::error::CodegenError {
6917 location: Some(location),
6918 end_location: Some(end_location),
6919 error: codegen::error::CodegenErrorType::RecursionError,
6920 source_path: source_file.name().to_owned(),
6921 }))
6922}
6923
6924#[doc(hidden)]
6931#[must_use]
6932pub fn unsupported_grammar_error(ast: &ast::Mod, source_file: &SourceFile) -> Option<CompileError> {
6933 use ast::visitor::Visitor;
6934
6935 const MAX_FORMAT_SPEC_DEPTH: usize = 2;
6937
6938 struct Checker<'a> {
6939 source_file: &'a SourceFile,
6940 error: Option<CompileError>,
6941 }
6942
6943 impl Checker<'_> {
6944 fn fail(&mut self, message: &str, range: ruff_text_size::TextRange) {
6945 self.error = Some(CompileError::from_source_error(
6946 self.source_file,
6947 CpythonDiagnostic::new(
6948 message.to_owned(),
6949 range.start().to_usize(),
6950 range.end().to_usize(),
6951 ),
6952 ));
6953 }
6954
6955 fn check_format_specs(
6956 &mut self,
6957 kind: &str,
6958 elements: &ast::InterpolatedStringElements,
6959 depth: usize,
6960 ) {
6961 for element in elements.interpolations() {
6962 let Some(format_spec) = &element.format_spec else {
6963 continue;
6964 };
6965 if depth == MAX_FORMAT_SPEC_DEPTH {
6966 self.fail(
6967 &alloc::format!("{kind}: expressions nested too deeply"),
6968 format_spec.range,
6969 );
6970 return;
6971 }
6972 self.check_format_specs(kind, &format_spec.elements, depth + 1);
6973 if self.error.is_some() {
6974 return;
6975 }
6976 }
6977 }
6978 }
6979
6980 impl<'a> Visitor<'a> for Checker<'_> {
6981 fn visit_stmt(&mut self, stmt: &'a ast::Stmt) {
6982 if self.error.is_some() {
6983 return;
6984 }
6985 if let ast::Stmt::ClassDef(class_def) = stmt
6986 && let Some(arguments) = &class_def.arguments
6987 && let [ast::Expr::Generator(generator)] = &arguments.args[..]
6988 && !generator.parenthesized
6989 {
6990 let range = generator
6991 .generators
6992 .first()
6993 .map_or(generator.range, |comprehension| comprehension.range);
6994 self.fail("invalid syntax", range);
6995 return;
6996 }
6997 ast::visitor::walk_stmt(self, stmt);
6998 }
6999
7000 fn visit_expr(&mut self, expr: &'a ast::Expr) {
7001 if self.error.is_some() {
7002 return;
7003 }
7004 match expr {
7007 ast::Expr::FString(fstring) => {
7008 for part in &fstring.value {
7009 if let ast::FStringPart::FString(part) = part {
7010 self.check_format_specs("f-string", &part.elements, 0);
7011 }
7012 }
7013 }
7014 ast::Expr::TString(tstring) => {
7015 for part in &tstring.value {
7016 self.check_format_specs("t-string", &part.elements, 0);
7017 }
7018 }
7019 _ => {}
7020 }
7021 if self.error.is_some() {
7022 return;
7023 }
7024 ast::visitor::walk_expr(self, expr);
7025 }
7026 }
7027
7028 let mut checker = Checker {
7029 source_file,
7030 error: None,
7031 };
7032 match ast {
7033 ast::Mod::Module(module) => checker.visit_body(&module.body),
7034 ast::Mod::Expression(expression) => checker.visit_expr(&expression.body),
7035 }
7036 checker.error
7037}
7038
7039fn single_mode_body_error(body: &[ast::Stmt], source_file: &SourceFile) -> Option<CompileError> {
7040 let first = body.first()?;
7041 let source_code = source_file.to_source_code();
7042 let first_start = source_code.source_location(first.range().start(), PositionEncoding::Utf8);
7043 let first_end = source_code.source_location(first.range().end(), PositionEncoding::Utf8);
7044
7045 if body.iter().skip(1).any(|stmt| {
7046 source_code
7047 .source_location(stmt.range().start(), PositionEncoding::Utf8)
7048 .line
7049 > first_start.line
7050 }) {
7051 return Some(CompileError::from_source_error(
7052 source_file,
7053 CpythonDiagnostic::new(
7054 "multiple statements found while compiling a single statement".to_owned(),
7055 first.range().end().to_usize(),
7056 first.range().end().to_usize(),
7057 ),
7058 ));
7059 }
7060
7061 if is_compound_stmt(first)
7062 && first_start.line == first_end.line
7063 && !ends_with_line_break(source_file.source_text())
7064 {
7065 return Some(CompileError::from_source_error(
7066 source_file,
7067 CpythonDiagnostic::new(
7068 "invalid syntax".to_owned(),
7069 first.range().start().to_usize(),
7070 first.range().start().to_usize(),
7071 ),
7072 ));
7073 }
7074 None
7075}
7076
7077fn single_mode_source_error(ast: &ast::Mod, source_file: &SourceFile) -> Option<CompileError> {
7078 let ast::Mod::Module(module) = ast else {
7079 return None;
7080 };
7081 single_mode_body_error(&module.body, source_file)
7082}
7083
7084fn ends_with_line_break(source: &str) -> bool {
7085 source.ends_with('\n') || source.ends_with('\r')
7086}
7087
7088fn ends_with_implied_dedent(source: &str) -> bool {
7089 let mut lexer = parser::lexer::lex(source, parser::Mode::Module);
7090 let mut last_kind = TokenKind::EndOfFile;
7091 loop {
7092 let kind = lexer.next_token();
7093 if kind.is_eof() {
7094 break;
7095 }
7096 last_kind = kind;
7097 }
7098 matches!(last_kind, TokenKind::Dedent)
7099}
7100
7101#[must_use]
7106pub fn dont_imply_dedent_source_error(source_file: &SourceFile) -> Option<CompileError> {
7107 let source = source_file.source_text();
7108 if ends_with_line_break(source) || !ends_with_implied_dedent(source) {
7109 return None;
7110 }
7111 let eof = source.len();
7112 Some(CompileError::from_source_error(
7113 source_file,
7114 CpythonDiagnostic::new("incomplete input".to_owned(), eof, eof),
7115 ))
7116}
7117
7118fn find_unclosed_bracket(source: &str) -> Option<(char, usize)> {
7121 let mut stack: Vec<(char, usize)> = Vec::new();
7122 let mut in_string = false;
7123 let mut string_quote = '\0';
7124 let mut triple_quote = false;
7125 let mut escape_next = false;
7126 let mut is_raw_string = false;
7127
7128 let chars: Vec<(usize, char)> = source.char_indices().collect();
7129 let mut i = 0;
7130
7131 while i < chars.len() {
7132 let (byte_offset, ch) = chars[i];
7133
7134 if escape_next {
7135 escape_next = false;
7136 i += 1;
7137 continue;
7138 }
7139
7140 if in_string {
7141 if ch == '\\' && !is_raw_string {
7142 escape_next = true;
7143 } else if triple_quote {
7144 if ch == string_quote
7145 && i + 2 < chars.len()
7146 && chars[i + 1].1 == string_quote
7147 && chars[i + 2].1 == string_quote
7148 {
7149 in_string = false;
7150 i += 3;
7151 continue;
7152 }
7153 } else if ch == string_quote {
7154 in_string = false;
7155 }
7156 i += 1;
7157 continue;
7158 }
7159
7160 if ch == '#' {
7162 while i < chars.len() && chars[i].1 != '\n' {
7164 i += 1;
7165 }
7166 continue;
7167 }
7168
7169 if ch == '\'' || ch == '"' {
7171 is_raw_string = false;
7173 for look_back in 1..=2.min(i) {
7174 let prev = chars[i - look_back].1;
7175 if matches!(prev, 'r' | 'R') {
7176 is_raw_string = true;
7177 break;
7178 }
7179 if !matches!(prev, 'b' | 'B' | 'f' | 'F' | 'u' | 'U') {
7180 break;
7181 }
7182 }
7183 string_quote = ch;
7184 if i + 2 < chars.len() && chars[i + 1].1 == ch && chars[i + 2].1 == ch {
7185 triple_quote = true;
7186 in_string = true;
7187 i += 3;
7188 continue;
7189 }
7190 triple_quote = false;
7191 in_string = true;
7192 i += 1;
7193 continue;
7194 }
7195
7196 match ch {
7197 '(' | '[' | '{' => stack.push((ch, byte_offset)),
7198 ')' | ']' | '}' => {
7199 let expected = match ch {
7200 ')' => '(',
7201 ']' => '[',
7202 '}' => '{',
7203 _ => unreachable!(),
7204 };
7205 if stack.last().is_some_and(|&(open, _)| open == expected) {
7206 stack.pop();
7207 }
7208 }
7209 _ => {}
7210 }
7211
7212 i += 1;
7213 }
7214
7215 stack.last().copied()
7216}
7217
7218pub fn compile(
7220 source: &str,
7221 mode: Mode,
7222 source_path: &str,
7223 opts: CompileOpts,
7224) -> Result<CodeObject, CompileError> {
7225 #[cfg(windows)]
7228 let source = source.replace("\r\n", "\n");
7229 #[cfg(windows)]
7230 let source = source.as_str();
7231
7232 let source_file = SourceFileBuilder::new(source_path, source).finish();
7233 _compile(source_file, mode, opts)
7234 }
7250
7251fn _compile(
7252 source_file: SourceFile,
7253 mode: Mode,
7254 opts: CompileOpts,
7255) -> Result<CodeObject, CompileError> {
7256 _compile_with_syntax_warning_handler(source_file, mode, opts, None)
7257}
7258
7259fn _compile_with_syntax_warning_handler<'a>(
7260 source_file: SourceFile,
7261 mode: Mode,
7262 opts: CompileOpts,
7263 syntax_warning_handler: Option<&'a mut compile::SyntaxWarningHandler<'a>>,
7264) -> Result<CodeObject, CompileError> {
7265 let parser_mode = match mode {
7266 Mode::Exec => parser::Mode::Module,
7267 Mode::Eval => parser::Mode::Expression,
7268 Mode::Single | Mode::BlockExpr => parser::Mode::Module,
7271 };
7272 let parser_options = parser::ParseOptions::from(parser_mode);
7273 let barry_source = prepare_barry_as_flufl_source(
7274 source_file.source_text(),
7275 parser_options.clone(),
7276 opts.future_features
7277 .contains(core::bytecode::CodeFlags::FUTURE_BARRY_AS_BDFL),
7278 );
7279 pre_parse_source_error(&source_file)?;
7280 let parsed = parser::parse(barry_source.source(), parser_options);
7281 if let Some(error) = barry_source.diagnostic(parsed.as_ref().err(), &source_file) {
7282 return Err(error);
7283 }
7284 let parsed =
7285 parsed.map_err(|err| CompileError::from_ruff_parse_error(err, &source_file, mode))?;
7286 if matches!(mode, Mode::Single)
7287 && let Some(error) = single_mode_blank_source_error(&source_file)
7288 {
7289 return Err(error);
7290 }
7291 if opts.dont_imply_dedent
7292 && matches!(mode, Mode::Single)
7293 && let Some(error) = dont_imply_dedent_source_error(&source_file)
7294 {
7295 return Err(error);
7296 }
7297 if let Some(error) = post_parse_source_error(&source_file, parsed.tokens(), &opts) {
7298 return Err(error);
7299 }
7300 let ast = parsed.into_syntax();
7301 too_deeply_nested_error(&ast, &source_file, opts.recursion_limit)?;
7302 if let Some(error) = unsupported_grammar_error(&ast, &source_file) {
7303 return Err(error);
7304 }
7305 let single_mode_error = matches!(mode, Mode::Single)
7306 .then(|| single_mode_source_error(&ast, &source_file))
7307 .flatten();
7308 let code = compile::compile_top_with_syntax_warning_handler(
7309 ast,
7310 source_file,
7311 mode,
7312 opts,
7313 syntax_warning_handler,
7314 )
7315 .map_err(CompileError::from)?;
7316 if let Some(error) = single_mode_error {
7317 return Err(error);
7318 }
7319 Ok(code)
7320}
7321
7322#[doc(hidden)]
7323pub struct BarrySource<'a> {
7324 source: Cow<'a, str>,
7325 not_equal: Option<ruff_text_size::TextRange>,
7326 legacy_not_equal: Vec<ruff_text_size::TextRange>,
7327}
7328
7329impl BarrySource<'_> {
7330 #[must_use]
7331 pub fn source(&self) -> &str {
7332 &self.source
7333 }
7334
7335 #[must_use]
7336 pub fn not_equal_before(
7337 &self,
7338 parse_error: Option<&parser::ParseError>,
7339 ) -> Option<ruff_text_size::TextRange> {
7340 self.not_equal.filter(|range| {
7341 parse_error.is_none_or(|error| {
7342 let diagnostic_start = if matches!(
7343 &error.error,
7344 parser::ParseErrorType::Lexical(parser::LexicalErrorType::Eof)
7345 ) {
7346 find_unclosed_bracket(&self.source).map_or_else(
7347 || error.location.start(),
7348 |(_, offset)| TextSize::new(offset as u32),
7349 )
7350 } else {
7351 error.location.start()
7352 };
7353 range.start() <= diagnostic_start
7354 })
7355 })
7356 }
7357
7358 #[must_use]
7364 pub fn invalid_legacy_operator(
7365 &self,
7366 parse_error: &parser::ParseError,
7367 ) -> Option<ruff_text_size::TextRange> {
7368 let location = parse_error.location.start();
7369 self.legacy_not_equal
7370 .iter()
7371 .copied()
7372 .find(|range| range.contains(location) || range.start() == location)
7373 .filter(|range| self.outranks_unclosed_bracket(*range))
7374 }
7375
7376 fn outranks_unclosed_bracket(&self, range: ruff_text_size::TextRange) -> bool {
7379 find_unclosed_bracket(&self.source)
7380 .is_none_or(|(_, offset)| range.start() <= TextSize::new(offset as u32))
7381 }
7382
7383 #[must_use]
7387 pub fn diagnostic(
7388 &self,
7389 parse_error: Option<&parser::ParseError>,
7390 source_file: &SourceFile,
7391 ) -> Option<CompileError> {
7392 if let Some(range) = parse_error.and_then(|error| self.invalid_legacy_operator(error)) {
7393 return Some(barry_as_flufl_invalid_legacy_operator_error(
7394 source_file,
7395 range,
7396 ));
7397 }
7398 self.not_equal_before(parse_error)
7399 .map(|range| barry_as_flufl_not_equal_error(source_file, range))
7400 }
7401}
7402
7403#[doc(hidden)]
7404#[must_use]
7405pub fn barry_as_flufl_not_equal_error(
7406 source_file: &SourceFile,
7407 range: ruff_text_size::TextRange,
7408) -> CompileError {
7409 CompileError::from_source_error(
7410 source_file,
7411 CpythonDiagnostic::new(
7412 "with Barry as BDFL, use '<>' instead of '!='".to_owned(),
7413 range.start().to_usize(),
7414 range.end().to_usize(),
7415 ),
7416 )
7417}
7418
7419#[doc(hidden)]
7420#[must_use]
7421pub fn barry_as_flufl_invalid_legacy_operator_error(
7422 source_file: &SourceFile,
7423 range: ruff_text_size::TextRange,
7424) -> CompileError {
7425 CompileError::from_source_error(
7426 source_file,
7427 CpythonDiagnostic::new(
7428 "invalid syntax".to_owned(),
7429 range.start().to_usize(),
7430 range.end().to_usize(),
7431 ),
7432 )
7433}
7434
7435fn textual_legacy_not_equal(source: &str) -> Vec<ruff_text_size::TextRange> {
7439 source
7440 .match_indices("<>")
7441 .map(|(offset, matched)| {
7442 ruff_text_size::TextRange::at(
7443 TextSize::new(offset as u32),
7444 TextSize::new(matched.len() as u32),
7445 )
7446 })
7447 .collect()
7448}
7449
7450#[doc(hidden)]
7451pub fn prepare_barry_as_flufl_source(
7452 source: &str,
7453 parser_options: parser::ParseOptions,
7454 inherited: bool,
7455) -> BarrySource<'_> {
7456 let scanned = (inherited || source.contains("barry_as_FLUFL"))
7457 .then(|| parser::parse_unchecked(source, parser_options));
7458 let enabled = scanned.as_ref().is_some_and(|scanned| {
7459 inherited
7460 || codegen::preprocess::future_features(scanned.syntax())
7461 .contains(core::bytecode::CodeFlags::FUTURE_BARRY_AS_BDFL)
7462 });
7463 let Some(scanned) = scanned.filter(|_| enabled) else {
7464 return BarrySource {
7465 source: Cow::Borrowed(source),
7466 not_equal: None,
7467 legacy_not_equal: textual_legacy_not_equal(source),
7468 };
7469 };
7470
7471 let not_equal = scanned
7472 .tokens()
7473 .iter()
7474 .find(|token| token.kind() == TokenKind::NotEqual)
7475 .map(Ranged::range);
7476 let replacements = scanned
7477 .tokens()
7478 .windows(2)
7479 .filter_map(|tokens| {
7480 let [less, greater] = tokens else {
7481 return None;
7482 };
7483 (less.kind() == TokenKind::Less
7484 && greater.kind() == TokenKind::Greater
7485 && less.end() == greater.start())
7486 .then(|| ruff_text_size::TextRange::new(less.start(), greater.end()))
7487 })
7488 .collect::<Vec<_>>();
7489
7490 let source = if replacements.is_empty() {
7491 Cow::Borrowed(source)
7492 } else {
7493 let mut rewritten = source.to_owned();
7494 for range in replacements.iter().rev() {
7495 rewritten.replace_range(range.start().to_usize()..range.end().to_usize(), "!=");
7496 }
7497 Cow::Owned(rewritten)
7498 };
7499 BarrySource {
7500 source,
7501 not_equal,
7502 legacy_not_equal: replacements,
7503 }
7504}
7505
7506pub fn compile_with_syntax_warning_handler<'a>(
7507 source: &str,
7508 mode: Mode,
7509 source_path: &str,
7510 opts: CompileOpts,
7511 syntax_warning_handler: &'a mut compile::SyntaxWarningHandler<'a>,
7512) -> Result<CodeObject, CompileError> {
7513 let source = source.replace("\r\n", "\n");
7514 #[cfg(windows)]
7515 let source = source.as_str();
7516
7517 let source_file = SourceFileBuilder::new(source_path, source).finish();
7518 _compile_with_syntax_warning_handler(source_file, mode, opts, Some(syntax_warning_handler))
7519}
7520
7521pub fn compile_symtable(
7522 source: &str,
7523 mode: Mode,
7524 source_path: &str,
7525) -> Result<symboltable::SymbolTable, CompileError> {
7526 let source_file = SourceFileBuilder::new(source_path, source).finish();
7527 _compile_symtable(source_file, mode)
7528}
7529
7530fn symtable_preprocess_module(
7532 module: &mut ast::ModModule,
7533 source_file: &SourceFile,
7534) -> Result<(), CompileError> {
7535 let future_features = codegen::preprocess::checked_future_features_in_body(&module.body)
7536 .map_err(|error| future_feature_error(error, source_file))?;
7537 let future_annotations =
7538 future_features.contains(core::bytecode::CodeFlags::FUTURE_ANNOTATIONS);
7539 codegen::preprocess::preprocess_statements(&mut module.body, 0, future_annotations, true);
7542 Ok(())
7543}
7544
7545fn future_feature_error(
7546 error: codegen::preprocess::FutureFeatureError,
7547 source_file: &SourceFile,
7548) -> CompileError {
7549 let source_code = source_file.to_source_code();
7550 let location = source_code.source_location(error.range.start(), PositionEncoding::Utf8);
7551 let end_location = source_code.source_location(error.range.end(), PositionEncoding::Utf8);
7552 let error = match error.kind {
7553 codegen::preprocess::FutureFeatureErrorKind::InvalidFeature(feature) => {
7554 codegen::error::CodegenErrorType::InvalidFutureFeature(feature)
7555 }
7556 codegen::preprocess::FutureFeatureErrorKind::InvalidBraces => {
7557 codegen::error::CodegenErrorType::InvalidFutureBraces
7558 }
7559 };
7560 codegen::error::CodegenError {
7561 location: Some(location),
7562 end_location: Some(end_location),
7563 error,
7564 source_path: source_file.name().to_owned(),
7565 }
7566 .into()
7567}
7568
7569pub fn _compile_symtable(
7570 source_file: SourceFile,
7571 mode: Mode,
7572) -> Result<symboltable::SymbolTable, CompileError> {
7573 let parser_mode = match mode {
7574 Mode::Exec | Mode::Single | Mode::BlockExpr => parser::Mode::Module,
7575 Mode::Eval => parser::Mode::Expression,
7576 };
7577 let parser_options = parser::ParseOptions::from(parser_mode);
7578 let barry_source =
7579 prepare_barry_as_flufl_source(source_file.source_text(), parser_options.clone(), false);
7580 let res = match mode {
7581 Mode::Exec | Mode::Single | Mode::BlockExpr => {
7582 pre_parse_source_error(&source_file)?;
7583 let parsed = ruff_python_parser::parse(barry_source.source(), parser_options);
7584 if let Some(error) = barry_source.diagnostic(parsed.as_ref().err(), &source_file) {
7585 return Err(error);
7586 }
7587 let ast =
7588 parsed.map_err(|e| CompileError::from_ruff_parse_error(e, &source_file, mode))?;
7589 if let Some(error) =
7590 post_parse_source_error(&source_file, ast.tokens(), &CompileOpts::default())
7591 {
7592 return Err(error);
7593 }
7594 let ast = ast.into_syntax();
7595 too_deeply_nested_error(&ast, &source_file, CompileOpts::default().recursion_limit)?;
7596 if let Some(error) = unsupported_grammar_error(&ast, &source_file) {
7597 return Err(error);
7598 }
7599 let ast = ast.expect_module();
7600 if matches!(mode, Mode::Single)
7601 && let Some(error) = single_mode_body_error(&ast.body, &source_file)
7602 {
7603 return Err(error);
7604 }
7605 let mut ast = ast;
7606 symtable_preprocess_module(&mut ast, &source_file)?;
7607 symboltable::SymbolTable::scan_program(&ast, source_file.clone())
7608 }
7609 Mode::Eval => {
7610 pre_parse_source_error(&source_file)?;
7611 let parsed = ruff_python_parser::parse(barry_source.source(), parser_options);
7612 if let Some(error) = barry_source.diagnostic(parsed.as_ref().err(), &source_file) {
7613 return Err(error);
7614 }
7615 let ast =
7616 parsed.map_err(|e| CompileError::from_ruff_parse_error(e, &source_file, mode))?;
7617 if let Some(error) =
7618 post_parse_source_error(&source_file, ast.tokens(), &CompileOpts::default())
7619 {
7620 return Err(error);
7621 }
7622 let ast = ast.into_syntax();
7623 too_deeply_nested_error(&ast, &source_file, CompileOpts::default().recursion_limit)?;
7624 if let Some(error) = unsupported_grammar_error(&ast, &source_file) {
7625 return Err(error);
7626 }
7627 let mut ast = ast;
7628 codegen::preprocess::preprocess_mod(&mut ast, 0, false, true);
7629 symboltable::SymbolTable::scan_expr(&ast.expect_expression(), source_file.clone())
7630 }
7631 };
7632 res.map_err(|e| e.into_codegen_error(source_file.name().to_owned()).into())
7633}
7634
7635#[cfg(test)]
7636mod tests {
7637 use super::*;
7638
7639 #[test]
7640 fn basic_compile() {
7641 let code = "x = 'abc'";
7642 let compiled = compile(code, Mode::Single, "<>", CompileOpts::default());
7643 dbg!(compiled.expect("compile error"));
7644 }
7645
7646 #[test]
7647 fn empty_parameter_default_is_reported_before_later_params() {
7648 let err = compile(
7649 "def foo(a=1,d=,c):\n pass\n",
7650 Mode::Exec,
7651 "<params>",
7652 CompileOpts::default(),
7653 )
7654 .expect_err("empty default should fail");
7655 assert!(
7656 err.to_string()
7657 .contains("expected default value expression"),
7658 "got {err}"
7659 );
7660 }
7661
7662 #[test]
7663 fn too_many_nested_fstrings_match_tokenizer_limit() {
7664 fn nested(n: usize) -> String {
7665 if n == 0 {
7666 return "1+1".to_owned();
7667 }
7668 format!("f\"{{{}}}\"", nested(n - 1))
7669 }
7670
7671 compile(&nested(149), Mode::Eval, "<nest>", CompileOpts::default())
7672 .expect("149 nested f-strings should compile");
7673 let err = compile(&nested(150), Mode::Eval, "<nest>", CompileOpts::default())
7674 .expect_err("150 nested f-strings should fail");
7675 assert!(
7676 err.to_string()
7677 .contains("too many nested f-strings or t-strings"),
7678 "got {err}"
7679 );
7680 }
7681
7682 #[test]
7683 fn interpolated_string_diagnostics_match_cpython() {
7684 for (source, expected) in [
7685 ("f'{'", "f-string: expecting '}'"),
7686 ("t'{'", "t-string: expecting '}'"),
7687 (
7688 "f'{1=}{;'",
7689 "f-string: expecting a valid expression after '{'",
7690 ),
7691 (
7692 "t'{x;y}'",
7693 "t-string: expecting '=', or '!', or ':', or '}'",
7694 ),
7695 ("t'{x!s:'", "t-string: expecting '}', or format specs"),
7696 ("t'{x=!}'", "t-string: missing conversion character"),
7697 (
7698 "t'{x:{;}}'",
7699 "t-string: expecting a valid expression after '{'",
7700 ),
7701 ("f'{1#}'", "'{' was never closed"),
7702 ("t'{", "'{' was never closed"),
7703 ("f'{a", "'{' was never closed"),
7707 ("f'{a!r", "'{' was never closed"),
7708 ("f'{a=", "'{' was never closed"),
7709 ("f'{ {1:2}", "'{' was never closed"),
7710 ("f'{d[1:2]", "'{' was never closed"),
7711 ("f'{(lambda x: x)", "'{' was never closed"),
7712 ("f'{a:{b:{c", "'{' was never closed"),
7713 ("f'{a:", "unterminated f-string literal"),
7714 ("f'{a:>5", "unterminated f-string literal"),
7715 ("f'{a!r:", "unterminated f-string literal"),
7716 ("f'{a}{b:", "unterminated f-string literal"),
7717 ("f'{a:{b}c", "unterminated f-string literal"),
7718 ("t'{a:>5", "unterminated t-string literal"),
7719 ("f'''{a:>5", "unterminated triple-quoted f-string literal"),
7720 ("f'{a[", "'[' was never closed"),
7722 ("f'{(a", "'(' was never closed"),
7723 ("f'{)#}'", "f-string: unmatched ')'"),
7724 (
7725 "f'{a[4)}'",
7726 "closing parenthesis ')' does not match opening parenthesis '['",
7727 ),
7728 ("t'", "unterminated t-string literal (detected at line 1)"),
7729 (
7730 "t'''",
7731 "unterminated triple-quoted t-string literal (detected at line 1)",
7732 ),
7733 (
7734 "t\"x\" b\"y\"",
7735 "Cannot mix t-string literals with string or bytes literals",
7736 ),
7737 (
7738 "b\"x\" t\"y\"",
7739 "Cannot mix t-string literals with string or bytes literals",
7740 ),
7741 (
7744 "\"a\" b\"b\" t\"c\"",
7745 "cannot mix bytes and nonbytes literals",
7746 ),
7747 (
7749 "f'{a==}'",
7750 "f-string: expecting '=', or '!', or ':', or '}'",
7751 ),
7752 (
7753 "f'{a and}'",
7754 "f-string: expecting '=', or '!', or ':', or '}'",
7755 ),
7756 (
7757 "f'{a is not}'",
7758 "f-string: expecting '=', or '!', or ':', or '}'",
7759 ),
7760 (
7761 "f'{a.b.}'",
7762 "f-string: expecting '=', or '!', or ':', or '}'",
7763 ),
7764 ("t'{a~}'", "t-string: expecting '=', or '!', or ':', or '}'"),
7765 (
7768 "f'{==a}'",
7769 "f-string: expecting a valid expression after '{'",
7770 ),
7771 (
7772 "f'{!=a}'",
7773 "f-string: expecting a valid expression after '{'",
7774 ),
7775 (
7776 "t'{==a}'",
7777 "t-string: expecting a valid expression after '{'",
7778 ),
7779 ("f'{=a}'", "f-string: valid expression required before '='"),
7780 ("f'{!}'", "f-string: valid expression required before '!'"),
7781 (
7782 "f'{lambda x:x}'",
7783 "f-string: lambda expressions are not allowed without parentheses",
7784 ),
7785 (
7786 "f'{1, lambda:x}'",
7787 "f-string: lambda expressions are not allowed without parentheses",
7788 ),
7789 (
7790 "f'{+ lambda:None}'",
7791 "f-string: expecting a valid expression after '{'",
7792 ),
7793 ("fu''", "'u' and 'f' prefixes are incompatible"),
7794 ("fb''", "'b' and 'f' prefixes are incompatible"),
7795 ("ufr''", "'u' and 'r' prefixes are incompatible"),
7796 (
7797 "(]\nbu'x'",
7798 "closing parenthesis ']' does not match opening parenthesis '('",
7799 ),
7800 ("0x\nbu'x'", "invalid hexadecimal literal"),
7801 ("(0x", "invalid hexadecimal literal"),
7802 ("print x; 0x", "invalid hexadecimal literal"),
7803 ("exec x; 0x", "invalid hexadecimal literal"),
7804 (
7805 "print x; 0x1",
7806 "Missing parentheses in call to 'print'. Did you mean print(...)?",
7807 ),
7808 (
7809 "print x; (",
7810 "Missing parentheses in call to 'print'. Did you mean print(...)?",
7811 ),
7812 ("print x; )", "unmatched ')'"),
7813 (
7814 "( '\\N'",
7815 "(unicode error) 'unicodeescape' codec can't decode bytes in position 0-1: malformed \\N character escape",
7816 ),
7817 ("(print x", "'(' was never closed"),
7818 (
7819 "(print x; '\\N'",
7820 "Missing parentheses in call to 'print'. Did you mean print(...)?",
7821 ),
7822 ("f'{x'; '", "f-string: expecting '}'"),
7823 ("f'{x'; 0x", "f-string: expecting '}'"),
7824 (
7825 "print x; f'{x'; 0x",
7826 "Missing parentheses in call to 'print'. Did you mean print(...)?",
7827 ),
7828 (
7829 concat!("f'{1:", "d\n}'"),
7830 "f-string: newlines are not allowed in format specifiers",
7831 ),
7832 ("f'{\n}'", "f-string: valid expression required before '}'"),
7833 (
7834 "f'''\n{\n# only a comment\n}'''",
7835 "f-string: valid expression required before '}'",
7836 ),
7837 ("{\\'a\\'}", "unexpected character after line continuation"),
7838 ("\"\\\n\"(1 for c in I,\\\n\\", "'(' was never closed"),
7839 (
7840 r"'\N'",
7841 "(unicode error) 'unicodeescape' codec can't decode bytes in position 0-1: malformed \\N character escape",
7842 ),
7843 (
7844 r"f'\N{'",
7845 "(unicode error) 'unicodeescape' codec can't decode bytes in position 0-2: malformed \\N character escape",
7846 ),
7847 ] {
7848 let err = compile(source, Mode::Eval, "<interp>", CompileOpts::default())
7849 .expect_err("should not compile");
7850 assert!(
7851 err.to_string().contains(expected),
7852 "{source:?}: expected {expected:?}, got {err}"
7853 );
7854 }
7855 }
7856
7857 #[test]
7858 fn missing_indent_outranks_print_missing_parentheses() {
7859 for (source, expected) in [
7860 (
7861 "if True:\nprint \"No indent\"",
7862 "expected an indented block after 'if' statement on line 1",
7863 ),
7864 (
7865 "print \"old style\"",
7866 "Missing parentheses in call to 'print'. Did you mean print(...)?",
7867 ),
7868 ] {
7869 let err = compile(source, Mode::Exec, "<fragment>", CompileOpts::default())
7870 .expect_err("should not compile");
7871 assert!(
7872 err.to_string().contains(expected),
7873 "{source:?}: expected {expected:?}, got {err}"
7874 );
7875 }
7876 }
7877
7878 #[test]
7879 fn unclosed_fstring_field_keeps_the_unclosed_bracket_flag() {
7880 let err = compile("f'{", Mode::Eval, "<interp>", CompileOpts::default())
7881 .expect_err("should not compile");
7882 let crate::CompileError::Parse(parse) = err else {
7883 panic!("expected a parse error, got {err}");
7884 };
7885 assert!(
7886 parse.is_unclosed_bracket,
7887 "unclosed f-string field must stay incomplete, got {parse}"
7888 );
7889 assert!(
7890 parse.to_string().contains("'{' was never closed"),
7891 "got {parse}"
7892 );
7893 }
7894
7895 #[test]
7896 fn interpolated_literals_do_not_take_the_escaped_quote_hint() {
7897 for (source, expected) in [
7901 (
7902 r"f'\'",
7903 "unterminated f-string literal (detected at line 1)",
7904 ),
7905 (
7906 r"t'\'",
7907 "unterminated t-string literal (detected at line 1)",
7908 ),
7909 ] {
7910 let err = compile(source, Mode::Eval, "<escaped>", CompileOpts::default())
7911 .expect_err("should not compile");
7912 assert_eq!(err.to_string(), expected, "{source:?}");
7913 }
7914 let err = compile(r"'\'", Mode::Eval, "<escaped>", CompileOpts::default())
7916 .expect_err("should not compile");
7917 assert_eq!(
7918 err.to_string(),
7919 "unterminated string literal (detected at line 1); perhaps you escaped the end quote?"
7920 );
7921 }
7922
7923 #[test]
7924 fn a_field_bracket_mismatch_names_the_opening_line() {
7925 for (source, expected) in [
7928 (
7929 "x = f\"\"\"{a[\n4)}\"\"\"\n",
7930 "closing parenthesis ')' does not match opening parenthesis '[' on line 1",
7931 ),
7932 (
7933 "x = f\"\"\"{a(\n\n4]}\"\"\"\n",
7934 "closing parenthesis ']' does not match opening parenthesis '(' on line 1",
7935 ),
7936 (
7937 "x = f\"{a[4)}\"\n",
7938 "closing parenthesis ')' does not match opening parenthesis '['",
7939 ),
7940 ] {
7941 let err = compile(source, Mode::Exec, "<paren>", CompileOpts::default())
7942 .expect_err("should not compile");
7943 assert_eq!(err.to_string(), expected, "{source:?}");
7944 }
7945 }
7946
7947 #[test]
7948 fn a_dangling_operator_needs_the_field_to_close() {
7949 for (source, expected) in [
7954 ("f'{a;'", "f-string: expecting '=', or '!', or ':', or '}'"),
7955 ("f'{a$'", "f-string: expecting '=', or '!', or ':', or '}'"),
7956 ("f'{a?'", "f-string: expecting '=', or '!', or ':', or '}'"),
7957 ("f'{a and'", "f-string: expecting '}'"),
7958 ("f'{a+'", "f-string: expecting '}'"),
7959 ("f'{a=='", "f-string: expecting '}'"),
7960 ("f'{a.b.'", "f-string: expecting '}'"),
7961 ("f'{a is not'", "f-string: expecting '}'"),
7962 ("t'{a and'", "t-string: expecting '}'"),
7963 (
7965 "f'{a and}'",
7966 "f-string: expecting '=', or '!', or ':', or '}'",
7967 ),
7968 (
7969 "f'{a==}'",
7970 "f-string: expecting '=', or '!', or ':', or '}'",
7971 ),
7972 ] {
7973 let err = compile(source, Mode::Eval, "<dangling>", CompileOpts::default())
7974 .expect_err("should not compile");
7975 assert!(
7976 err.to_string().contains(expected),
7977 "{source:?}: expected {expected:?}, got {err}"
7978 );
7979 }
7980 }
7981
7982 #[test]
7983 fn deeply_nested_format_specs_stay_linear() {
7984 for depth in [32usize, 200] {
7988 let source = format!("x = f\"{}{}\"\n$", "{1:".repeat(depth), "}".repeat(depth));
7989 compile(&source, Mode::Exec, "<nested>", CompileOpts::default())
7990 .expect_err("the trailing `$` is a syntax error");
7991 }
7992 }
7993
7994 #[test]
7995 #[expect(
7996 clippy::literal_string_with_formatting_args,
7997 reason = "these are Python format specs, not Rust format args"
7998 )]
7999 fn valid_interpolated_literals_do_not_shadow_a_later_syntax_error() {
8000 for source in [
8004 "x = f\"{1:(}\"\n$",
8005 "x = f\"{1:[}\"\n$",
8006 "x = f\"{1:#x}\"\n$",
8007 "x = f\"{1!r:#>5}\"\n$",
8008 "x = t\"{1:(}\"\n$",
8009 "x = f\"\"\"{1 # (\n}\"\"\"\n$",
8010 "x = f\"\"\"{1 # {\n}\"\"\"\n$",
8011 "x = f\"\"\"{1 # ]\n}\"\"\"\n$",
8012 "x = f\"\"\"{1 # a\\ b\n}\"\"\"\n$",
8013 "x = f\"\"\"{a # +\n}\"\"\"\n$",
8015 "x = f\"\"\"{a # and\n}\"\"\"\n$",
8016 "x = f\"\"\"{a # ==\n}\"\"\"\n$",
8017 "x = t\"\"\"{a # .\n}\"\"\"\n$",
8018 "x = f\"\"\"{1:{a # +\n}}\"\"\"\n$",
8019 "x = f\"\"\"{a # +\n + b}\"\"\"\n$",
8020 "x = f\"{-.5}\"\n$",
8022 "x = f\"{+.5}\"\n$",
8023 "x = f\"{...}\"\n$",
8025 "x = f\"{.5}\"\n$",
8026 "x = t\"{...}\"\n$",
8027 "x = f\"{a+b}\"\n$",
8029 "x = f\"{a.b.c}\"\n$",
8030 "x = f\"{1.}\"\n$",
8031 "x = f\"{a==b}\"\n$",
8033 "x = f\"{a!=b}\"\n$",
8034 "x = f\"{a<=b}\"\n$",
8035 "x = f\"{a>=b}\"\n$",
8036 "x = t\"{a==b}\"\n$",
8037 "x = f\"{a:{b==c}}\"\n$",
8038 ] {
8039 let err = compile(source, Mode::Exec, "<interp>", CompileOpts::default())
8040 .expect_err("the trailing `$` is a syntax error");
8041 assert_eq!(
8042 err.python_location().0,
8043 source.lines().count(),
8044 "{source:?} reported the wrong line: {err}"
8045 );
8046 }
8047 }
8048
8049 #[test]
8050 fn dont_imply_dedent_requires_terminating_newline() {
8051 let code = "if True:\n pass";
8052
8053 let opts = CompileOpts {
8054 dont_imply_dedent: true,
8055 ..CompileOpts::default()
8056 };
8057 let err = compile(code, Mode::Single, "<>", opts.clone()).expect_err("compile succeeded");
8058 assert_eq!(err.to_string(), "incomplete input");
8059
8060 compile("if True:\n pass\n", Mode::Single, "<>", opts).expect("compile error");
8061 compile(code, Mode::Single, "<>", CompileOpts::default()).expect("compile error");
8062 }
8063
8064 #[test]
8065 fn barry_as_flufl_rewrites_legacy_not_equal_after_future_import() {
8066 let code = compile(
8067 "from __future__ import barry_as_FLUFL\nresult = 2 <> 3\n",
8068 Mode::Exec,
8069 "<barry>",
8070 CompileOpts::default(),
8071 )
8072 .expect("Barry comparison should compile");
8073 assert!(
8074 code.flags
8075 .contains(core::bytecode::CodeFlags::FUTURE_BARRY_AS_BDFL)
8076 );
8077 }
8078
8079 #[test]
8080 fn inherited_barry_as_flufl_rewrites_legacy_not_equal() {
8081 let opts = CompileOpts {
8082 future_features: core::bytecode::CodeFlags::FUTURE_BARRY_AS_BDFL,
8083 ..CompileOpts::default()
8084 };
8085 compile("2 <> 3", Mode::Single, "<barry>", opts)
8086 .expect("inherited Barry comparison should compile");
8087 }
8088
8089 #[test]
8090 fn barry_as_flufl_rejects_modern_not_equal() {
8091 let err = compile(
8092 "from __future__ import barry_as_FLUFL\n2 != 3\n",
8093 Mode::Exec,
8094 "<barry>",
8095 CompileOpts::default(),
8096 )
8097 .expect_err("Barry mode should reject !=");
8098 assert_eq!(
8099 err.to_string(),
8100 "with Barry as BDFL, use '<>' instead of '!='"
8101 );
8102 assert_eq!(err.python_location(), (2, 3));
8103 }
8104
8105 #[test]
8106 fn fstring_adjacent_atoms_are_a_missing_comma() {
8107 let err = compile("f'{6 0}'", Mode::Exec, "<fragment>", CompileOpts::default())
8108 .expect_err("adjacent atoms in an f-string field are a syntax error");
8109 assert_eq!(
8110 err.to_string(),
8111 "invalid syntax. Perhaps you forgot a comma?"
8112 );
8113 assert_eq!(err.python_location(), (1, 4));
8114 assert_eq!(err.python_end_location(), Some((1, 7)));
8115
8116 compile(
8117 "f'{not x}'",
8118 Mode::Exec,
8119 "<fragment>",
8120 CompileOpts::default(),
8121 )
8122 .expect("unary not is a prefix, not two atoms");
8123 let err = compile(
8124 "f'{a and}'",
8125 Mode::Exec,
8126 "<fragment>",
8127 CompileOpts::default(),
8128 )
8129 .expect_err("a dangling 'and' is an f-string separator error");
8130 assert_eq!(
8131 err.to_string(),
8132 "f-string: expecting '=', or '!', or ':', or '}'"
8133 );
8134 }
8135
8136 #[test]
8137 fn missing_comma_diagnostic_spans_the_whole_second_atom() {
8138 let span = |source: &str| {
8141 let err = compile(source, Mode::Eval, "<comma>", CompileOpts::default())
8142 .expect_err("two adjacent atoms are a syntax error");
8143 assert_eq!(
8144 err.to_string(),
8145 "invalid syntax. Perhaps you forgot a comma?"
8146 );
8147 (
8148 err.python_location().1,
8149 err.python_end_location().unwrap().1,
8150 )
8151 };
8152
8153 assert_eq!(span("(a b)"), (2, 5));
8155 assert_eq!(span("(a bb)"), (2, 6));
8157 assert_eq!(span("(a bbb)"), (2, 7));
8158 assert_eq!(span("(1 22)"), (2, 6));
8159 assert_eq!(span("(a \u{3b2})"), (2, 5));
8162 assert_eq!(span("(a \u{3b2}\u{3b2})"), (2, 6));
8163 assert_eq!(span("(\u{3b1}\u{3b1} \u{3b2})"), (2, 6));
8164 assert_eq!(span("(\u{3b1} b)"), (2, 5));
8166 assert_eq!(span("[\u{3b1} \u{3b2}]"), (2, 5));
8168 }
8169
8170 #[test]
8171 fn parenthesized_yield_assignment_uses_invalid_target_message() {
8172 let err = compile(
8173 "def f(): (yield bar) = y\n",
8174 Mode::Exec,
8175 "<yield>",
8176 CompileOpts::default(),
8177 )
8178 .expect_err("parenthesized yield is not an assignment target");
8179 assert_eq!(
8180 err.to_string(),
8181 "cannot assign to yield expression here. Maybe you meant '==' instead of '='?"
8182 );
8183 }
8184
8185 #[test]
8186 fn parenthesized_yield_augassign_uses_illegal_expression_message() {
8187 let err = compile(
8188 "def f(): (yield bar) += y\n",
8189 Mode::Exec,
8190 "<yield>",
8191 CompileOpts::default(),
8192 )
8193 .expect_err("parenthesized yield is not an augmented assignment target");
8194 assert_eq!(
8195 err.to_string(),
8196 "'yield expression' is an illegal expression for augmented assignment"
8197 );
8198 }
8199
8200 #[test]
8201 fn kwarg_unparenthesized_genexp_uses_eq_or_walrus_message() {
8202 let err = compile(
8203 "dict(a = i for i in range(10))\n",
8204 Mode::Exec,
8205 "<kwarg>",
8206 CompileOpts::default(),
8207 )
8208 .expect_err("unparenthesized genexp after '=' is invalid");
8209 assert_eq!(
8210 err.to_string(),
8211 "invalid syntax. Maybe you meant '==' or ':=' instead of '='?"
8212 );
8213 }
8214
8215 #[test]
8216 fn eval_fstring_assignment_keeps_invalid_syntax() {
8217 for source in ["f'' = 3", "f'{0}' = x", "f'{x}' = x"] {
8218 let err = compile(source, Mode::Eval, "<eval>", CompileOpts::default())
8219 .expect_err("assignment is invalid in eval");
8220 assert_eq!(err.to_string(), "invalid syntax", "{source}");
8221 }
8222 let err = compile("f'' = 3", Mode::Exec, "<exec>", CompileOpts::default())
8223 .expect_err("f-string is not an assignment target");
8224 assert_eq!(
8225 err.to_string(),
8226 "cannot assign to f-string expression here. Maybe you meant '==' instead of '='?"
8227 );
8228 }
8229
8230 #[test]
8231 fn if_assignment_uses_eq_or_walrus_message() {
8232 let err = compile(
8233 "if x = 3: pass\n",
8234 Mode::Exec,
8235 "<if>",
8236 CompileOpts::default(),
8237 )
8238 .expect_err("assignment in if condition is invalid");
8239 assert_eq!(
8240 err.to_string(),
8241 "invalid syntax. Maybe you meant '==' or ':=' instead of '='?"
8242 );
8243 }
8244
8245 #[test]
8246 fn parenthesized_if_assignment_uses_eq_or_walrus_message() {
8247 for source in [
8248 "if (x = 3): pass\n",
8249 "if ((x = 3)): pass\n",
8250 "if (x = 3) and y: pass\n",
8251 ] {
8252 let err = compile(source, Mode::Exec, "<if>", CompileOpts::default())
8253 .expect_err("parenthesized assignment in if condition is invalid");
8254 assert_eq!(
8255 err.to_string(),
8256 "invalid syntax. Maybe you meant '==' or ':=' instead of '='?"
8257 );
8258 }
8259 }
8260
8261 #[test]
8262 fn earlier_syntax_error_is_not_replaced_by_later_condition() {
8263 let err = compile(
8264 "@@@\nif x = 3: pass\n",
8265 Mode::Exec,
8266 "<if>",
8267 CompileOpts::default(),
8268 )
8269 .expect_err("the first invalid token is the syntax error");
8270 assert_eq!(err.to_string(), "invalid syntax");
8271 assert_eq!(err.python_location().0, 1);
8272 }
8273
8274 #[test]
8275 fn if_attribute_assignment_uses_invalid_target_hint() {
8276 let err = compile(
8277 "if x.a = 3: pass\n",
8278 Mode::Exec,
8279 "<if>",
8280 CompileOpts::default(),
8281 )
8282 .expect_err("attribute assignment in if condition is invalid");
8283 assert_eq!(
8284 err.to_string(),
8285 "cannot assign to attribute here. Maybe you meant '==' instead of '='?"
8286 );
8287 }
8288
8289 #[test]
8290 fn parenthesized_yield_from_assignment_uses_invalid_target_message() {
8291 let err = compile(
8292 "def f(): (yield from value) = target\n",
8293 Mode::Exec,
8294 "<yield>",
8295 CompileOpts::default(),
8296 )
8297 .expect_err("parenthesized yield from is not an assignment target");
8298 assert_eq!(
8299 err.to_string(),
8300 "cannot assign to yield expression here. Maybe you meant '==' instead of '='?"
8301 );
8302 }
8303
8304 #[test]
8305 fn set_display_assignment_uses_invalid_target_hint() {
8306 let err = compile(
8307 "{1, 2, 3} = 42\n",
8308 Mode::Exec,
8309 "<set>",
8310 CompileOpts::default(),
8311 )
8312 .expect_err("set display is not an assignment target");
8313 assert_eq!(
8314 err.to_string(),
8315 "cannot assign to set display here. Maybe you meant '==' instead of '='?"
8316 );
8317 }
8318
8319 #[test]
8320 fn obsolete_not_equal_diagnostic_spans_the_whole_operator() {
8321 let err = compile("2 <> 3\n", Mode::Exec, "<obsolete>", CompileOpts::default())
8322 .expect_err("'<>' outside Barry mode is a syntax error");
8323 assert_eq!(err.to_string(), "invalid syntax");
8324 assert_eq!(err.python_location(), (1, 3));
8325 assert_eq!(err.python_end_location(), Some((1, 5)));
8326
8327 let err = compile("2 <;\n", Mode::Exec, "<obsolete>", CompileOpts::default())
8330 .expect_err("'<;' is a syntax error");
8331 assert_eq!(err.to_string(), "invalid syntax");
8332 assert_eq!(err.python_location(), (1, 4));
8333 assert_eq!(err.python_end_location(), Some((1, 5)));
8334
8335 let err = compile("<>\n", Mode::Exec, "<obsolete>", CompileOpts::default())
8338 .expect_err("a bare '<>' is a syntax error");
8339 assert_eq!(err.to_string(), "invalid syntax");
8340 assert_eq!(err.python_location(), (1, 1));
8341 assert_eq!(err.python_end_location(), Some((1, 3)));
8342
8343 let err = compile(
8345 "(\n2 <> 3",
8346 Mode::Exec,
8347 "<obsolete>",
8348 CompileOpts::default(),
8349 )
8350 .expect_err("the bracket is never closed");
8351 assert_eq!(err.to_string(), "'(' was never closed");
8352 assert_eq!(err.python_location(), (1, 1));
8353 }
8354
8355 #[test]
8356 fn barry_as_flufl_does_not_rewrite_strings_or_comments() {
8357 compile(
8358 "from __future__ import barry_as_FLUFL\nx = '<>'\n# <>\n",
8359 Mode::Exec,
8360 "<barry>",
8361 CompileOpts::default(),
8362 )
8363 .expect("Barry markers in strings and comments should stay untouched");
8364 }
8365
8366 #[test]
8367 fn syntax_error_before_barry_not_equal_takes_precedence() {
8368 let err = compile(
8369 "from __future__ import barry_as_FLUFL\n<>\n2 != 3\n",
8370 Mode::Exec,
8371 "<barry>",
8372 CompileOpts::default(),
8373 )
8374 .expect_err("the earlier invalid comparison should fail");
8375 assert_eq!(err.to_string(), "invalid syntax");
8376 assert_eq!(err.python_location(), (2, 1));
8377 }
8378
8379 #[test]
8380 fn unclosed_bracket_before_barry_not_equal_takes_precedence() {
8381 let err = compile(
8382 "from __future__ import barry_as_FLUFL\n(\n2 != 3",
8383 Mode::Exec,
8384 "<barry>",
8385 CompileOpts::default(),
8386 )
8387 .expect_err("the earlier unclosed bracket should fail");
8388 assert_eq!(err.to_string(), "'(' was never closed");
8389 assert_eq!(err.python_location(), (2, 1));
8390 }
8391
8392 #[test]
8393 fn compile_phello() {
8394 let code = r#"
8395initialized = True
8396def main():
8397 print("Hello world!")
8398if __name__ == '__main__':
8399 main()
8400"#;
8401 let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8402 dbg!(compiled.expect("compile error"));
8403 }
8404
8405 #[test]
8406 fn compile_if_elif_else() {
8407 let code = r#"
8408if False:
8409 pass
8410elif False:
8411 pass
8412elif False:
8413 pass
8414else:
8415 pass
8416"#;
8417 let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8418 dbg!(compiled.expect("compile error"));
8419 }
8420
8421 #[test]
8422 fn compile_lambda() {
8423 let code = r#"
8424lambda: 'a'
8425"#;
8426 let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8427 dbg!(compiled.expect("compile error"));
8428 }
8429
8430 #[test]
8431 fn compile_lambda2() {
8432 let code = r#"
8433(lambda x: f'hello, {x}')('world}')
8434"#;
8435 let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8436 dbg!(compiled.expect("compile error"));
8437 }
8438
8439 #[test]
8440 fn compile_lambda3() {
8441 let code = r#"
8442def g():
8443 pass
8444def f():
8445 if False:
8446 return lambda x: g(x)
8447 elif False:
8448 return g
8449 else:
8450 return g
8451"#;
8452 let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8453 dbg!(compiled.expect("compile error"));
8454 }
8455
8456 #[test]
8457 fn compile_call_arg_lambda_default() {
8458 let code = "signature((lambda a=10: a))";
8459 let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8460 dbg!(compiled.expect("compile error"));
8461 }
8462
8463 #[test]
8464 fn compile_generic_function_parameter_default() {
8465 let code = "def __repr__[T: str](self, default: T = '') -> str: pass";
8466 let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8467 dbg!(compiled.expect("compile error"));
8468 }
8469
8470 #[test]
8471 fn compile_int() {
8472 let code = r#"
8473a = 0xFF
8474"#;
8475 let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8476 dbg!(compiled.expect("compile error"));
8477 }
8478
8479 #[test]
8480 fn compile_bigint() {
8481 let code = r#"
8482a = 0xFFFFFFFFFFFFFFFFFFFFFFFF
8483"#;
8484 let compiled = compile(code, Mode::Exec, "<>", CompileOpts::default());
8485 dbg!(compiled.expect("compile error"));
8486 }
8487
8488 #[test]
8489 fn compile_fstring() {
8490 let code1 = r#"
8491assert f"1" == '1'
8492 "#;
8493 let compiled = compile(code1, Mode::Exec, "<>", CompileOpts::default());
8494 dbg!(compiled.expect("compile error"));
8495
8496 let code2 = r#"
8497assert f"{1}" == '1'
8498 "#;
8499 let compiled = compile(code2, Mode::Exec, "<>", CompileOpts::default());
8500 dbg!(compiled.expect("compile error"));
8501 let code3 = r#"
8502assert f"{1+1}" == '2'
8503 "#;
8504 let compiled = compile(code3, Mode::Exec, "<>", CompileOpts::default());
8505 dbg!(compiled.expect("compile error"));
8506
8507 let code4 = r#"
8508assert f"{{{(lambda: f'{1}')}" == '{1'
8509 "#;
8510 let compiled = compile(code4, Mode::Exec, "<>", CompileOpts::default());
8511 dbg!(compiled.expect("compile error"));
8512
8513 let code5 = r#"
8514assert f"a{1}" == 'a1'
8515 "#;
8516 let compiled = compile(code5, Mode::Exec, "<>", CompileOpts::default());
8517 dbg!(compiled.expect("compile error"));
8518
8519 let code6 = r#"
8520assert f"{{{(lambda x: f'hello, {x}')('world}')}" == '{hello, world}'
8521 "#;
8522 let compiled = compile(code6, Mode::Exec, "<>", CompileOpts::default());
8523 dbg!(compiled.expect("compile error"));
8524 }
8525
8526 #[test]
8527 fn simple_enum() {
8528 let code = r#"
8529import enum
8530@enum._simple_enum(enum.IntFlag, boundary=enum.KEEP)
8531class RegexFlag:
8532 NOFLAG = 0
8533 DEBUG = 1
8534print(RegexFlag.NOFLAG & RegexFlag.DEBUG)
8535"#;
8536 let compiled = compile(code, Mode::Exec, "<string>", CompileOpts::default());
8537 dbg!(compiled.expect("compile error"));
8538 }
8539}