Skip to main content

rustpython_vm/vm/
compile.rs

1//! Python code compilation functions.
2//!
3//! For code execution functions, see python_run.rs
4
5use core::fmt;
6
7use crate::{
8    AsObject, PyObjectRef, PyRef, PyResult, VirtualMachine,
9    builtins::{PyBaseExceptionRef, PyCode},
10    compiler::{self, CompileError, CompileOpts},
11    vm::compile_mode::{CompileStart, CompilerFlags, compile_future_features_from_flags},
12};
13
14#[derive(Debug)]
15pub enum VmCompileError {
16    Compile(CompileError),
17    Warning(CompileWarningError),
18}
19
20#[derive(Debug)]
21pub struct CompileWarningError {
22    exception: PyBaseExceptionRef,
23    filename: String,
24    lineno: usize,
25    offset: usize,
26    replacement: Option<SyntaxErrorReplacement>,
27}
28
29#[derive(Debug)]
30struct SyntaxErrorReplacement {
31    message: String,
32    end_offset: usize,
33}
34
35impl From<CompileError> for VmCompileError {
36    fn from(err: CompileError) -> Self {
37        Self::Compile(err)
38    }
39}
40
41impl fmt::Display for VmCompileError {
42    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
43        match self {
44            Self::Compile(err) => err.fmt(f),
45            Self::Warning(_) => f.write_str("compiler warning raised as an exception"),
46        }
47    }
48}
49
50impl VmCompileError {
51    pub fn into_pyexception(self, vm: &VirtualMachine, source: Option<&str>) -> PyBaseExceptionRef {
52        self.into_pyexception_maybe_incomplete(vm, source, false)
53    }
54
55    pub fn into_pyexception_maybe_incomplete(
56        self,
57        vm: &VirtualMachine,
58        source: Option<&str>,
59        allow_incomplete: bool,
60    ) -> PyBaseExceptionRef {
61        match self {
62            Self::Compile(err) => {
63                vm.new_syntax_error_maybe_incomplete(&err, source, allow_incomplete)
64            }
65            Self::Warning(err) => err.into_pyexception(vm, source),
66        }
67    }
68}
69
70impl CompileWarningError {
71    fn into_pyexception(self, vm: &VirtualMachine, source: Option<&str>) -> PyBaseExceptionRef {
72        if !self
73            .exception
74            .fast_isinstance(vm.ctx.exceptions.syntax_warning)
75        {
76            return self.exception;
77        }
78        let (message, end_offset) = if let Some(replacement) = self.replacement {
79            (replacement.message, Some(replacement.end_offset))
80        } else {
81            let Ok(message) = self.exception.as_object().str(vm) else {
82                return self.exception;
83            };
84            (message.to_string_lossy().into_owned(), None)
85        };
86        let syntax_error =
87            vm.new_exception_msg(vm.ctx.exceptions.syntax_error.to_owned(), message.into());
88        syntax_error
89            .as_object()
90            .set_attr("lineno", vm.ctx.new_int(self.lineno), vm)
91            .unwrap();
92        syntax_error
93            .as_object()
94            .set_attr("offset", vm.ctx.new_int(self.offset), vm)
95            .unwrap();
96        if let Some(end_offset) = end_offset {
97            syntax_error
98                .as_object()
99                .set_attr("end_lineno", vm.ctx.new_int(self.lineno), vm)
100                .unwrap();
101            syntax_error
102                .as_object()
103                .set_attr("end_offset", vm.ctx.new_int(end_offset), vm)
104                .unwrap();
105        }
106        syntax_error
107            .as_object()
108            .set_attr("filename", vm.ctx.new_str(self.filename), vm)
109            .unwrap();
110        let text = source
111            .and_then(|source| source.split('\n').nth(self.lineno.saturating_sub(1)))
112            .map_or_else(
113                || vm.ctx.none(),
114                |line| {
115                    vm.ctx
116                        .new_str(format!("{}\n", line.trim_end_matches('\r')))
117                        .into()
118                },
119            );
120        syntax_error.as_object().set_attr("text", text, vm).unwrap();
121        syntax_error
122    }
123}
124
125impl VirtualMachine {
126    #[cfg(feature = "parser")]
127    fn detect_source_encoding(source: &[u8]) -> Option<String> {
128        fn find_encoding_in_line(line: &[u8]) -> Option<String> {
129            let hash_pos = line.iter().position(|&b| b == b'#')?;
130            if !line[..hash_pos]
131                .iter()
132                .all(|&b| b == b' ' || b == b'\t' || b == b'\x0c' || b == b'\r')
133            {
134                return None;
135            }
136            let after_hash = &line[hash_pos..];
137            let coding_pos = after_hash.windows(6).position(|w| w == b"coding")?;
138            let after_coding = &after_hash[coding_pos + 6..];
139            let rest = if after_coding.first() == Some(&b':') || after_coding.first() == Some(&b'=')
140            {
141                &after_coding[1..]
142            } else {
143                return None;
144            };
145            let name: String = rest
146                .iter()
147                .copied()
148                .skip_while(|&b| b == b' ' || b == b'\t')
149                .take_while(|&b| b.is_ascii_alphanumeric() || b == b'-' || b == b'_' || b == b'.')
150                .map(|b| b as char)
151                .collect();
152            (!name.is_empty()).then(|| VirtualMachine::normalize_source_encoding(&name))
153        }
154
155        let mut lines = source.splitn(3, |&b| b == b'\n');
156        if let Some(first) = lines.next() {
157            let first = first.strip_prefix(b"\xef\xbb\xbf").unwrap_or(first);
158            if let Some(enc) = find_encoding_in_line(first) {
159                return Some(enc);
160            }
161            let trimmed = first
162                .iter()
163                .skip_while(|&&b| b == b' ' || b == b'\t' || b == b'\x0c' || b == b'\r')
164                .copied()
165                .collect::<Vec<_>>();
166            if !trimmed.is_empty() && trimmed[0] != b'#' {
167                return None;
168            }
169        }
170        lines.next().and_then(find_encoding_in_line)
171    }
172
173    #[cfg(feature = "parser")]
174    fn normalize_source_encoding(name: &str) -> String {
175        let mut normalized = String::with_capacity(name.len().min(12));
176        for ch in name.chars().take(12) {
177            if ch == '_' {
178                normalized.push('-');
179            } else {
180                normalized.push(ch.to_ascii_lowercase());
181            }
182        }
183
184        if normalized == "utf-8" || normalized.starts_with("utf-8-") {
185            "utf-8".to_owned()
186        } else if normalized == "latin-1"
187            || normalized == "iso-8859-1"
188            || normalized == "iso-latin-1"
189            || normalized.starts_with("latin-1-")
190            || normalized.starts_with("iso-8859-1-")
191            || normalized.starts_with("iso-latin-1-")
192        {
193            "iso-8859-1".to_owned()
194        } else {
195            name.to_owned()
196        }
197    }
198
199    #[cfg(feature = "parser")]
200    fn is_utf8_encoding(name: &str) -> bool {
201        name == "utf-8"
202    }
203
204    /// Load the source line for a compiler SyntaxError. Tokenizer errors keep
205    /// the in-memory line; compiler errors read the file named by the
206    /// exception, and get None when that file cannot be opened.
207    #[cfg(any(feature = "parser", feature = "compiler"))]
208    pub(crate) fn program_text(&self, filename: &str, lineno: usize) -> Option<String> {
209        if lineno == 0 {
210            return None;
211        }
212        #[cfg(feature = "host_env")]
213        {
214            let buf = crate::host_env::fs::read(filename).ok()?;
215            let decoded = {
216                #[cfg(feature = "parser")]
217                {
218                    let encoding = Self::detect_source_encoding(&buf);
219                    if encoding.as_deref().is_none_or(Self::is_utf8_encoding) {
220                        String::from_utf8_lossy(&buf).into_owned()
221                    } else if encoding.as_deref() == Some("iso-8859-1") {
222                        buf.iter().copied().map(char::from).collect()
223                    } else {
224                        let name = encoding.as_deref()?;
225                        let bytes = self.ctx.new_bytes(buf);
226                        self.state
227                            .codec_registry
228                            .decode_text(bytes.into(), name, None, self)
229                            .ok()?
230                            .to_string_lossy()
231                            .into_owned()
232                    }
233                }
234                #[cfg(not(feature = "parser"))]
235                {
236                    String::from_utf8_lossy(&buf).into_owned()
237                }
238            };
239            let mut remaining = decoded.as_str();
240            if remaining.starts_with('\u{feff}') {
241                remaining = &remaining['\u{feff}'.len_utf8()..];
242            }
243            remaining
244                .split_inclusive('\n')
245                .nth(lineno.checked_sub(1)?)
246                .map(str::to_owned)
247        }
248        #[cfg(not(feature = "host_env"))]
249        {
250            let _ = filename;
251            None
252        }
253    }
254
255    #[cfg(feature = "parser")]
256    fn new_non_utf8_syntax_error(
257        &self,
258        filename: &str,
259        src: &[u8],
260        error_at: usize,
261    ) -> PyBaseExceptionRef {
262        let bad_byte = src[error_at];
263        let line_start = src[..error_at]
264            .iter()
265            .rposition(|&b| b == b'\n')
266            .map_or(0, |i| i + 1);
267        let lineno = src[..error_at].iter().filter(|&&b| b == b'\n').count() + 1;
268        let offset =
269            core::str::from_utf8(&src[line_start..error_at]).map_or(1, |s| s.chars().count() + 1);
270        let line_end = src[line_start..]
271            .iter()
272            .position(|&b| b == b'\n')
273            .map_or(src.len(), |i| line_start + i);
274        let text = String::from_utf8_lossy(&src[line_start..line_end]).into_owned();
275        let in_file = if filename.starts_with('<') {
276            String::new()
277        } else {
278            format!(" in file {filename}")
279        };
280        let msg = format!(
281            "Non-UTF-8 code starting with '\\x{bad_byte:02x}'{in_file} \
282             on line {lineno}, but no encoding declared; \
283             see https://peps.python.org/pep-0263/ for details"
284        );
285        let location = self.ctx.new_tuple(vec![
286            self.ctx.new_str(filename).into(),
287            self.ctx.new_int(lineno).into(),
288            self.ctx.new_int(offset).into(),
289            self.ctx.new_str(text).into(),
290            self.ctx.new_int(lineno).into(),
291            self.ctx.new_int(offset).into(),
292        ]);
293        self.invoke_exception(
294            self.ctx.exceptions.syntax_error,
295            vec![self.ctx.new_str(msg.as_str()).into(), location.into()],
296        )
297        .unwrap_or_else(|_| {
298            self.new_exception_msg(self.ctx.exceptions.syntax_error.to_owned(), msg.into())
299        })
300    }
301
302    #[cfg(feature = "parser")]
303    pub(crate) fn decode_source_bytes(
304        &self,
305        source: &[u8],
306        filename: &str,
307        ignore_cookie: bool,
308    ) -> PyResult<String> {
309        // `ignore_cookie` marks source that was handed over as text. A byte
310        // order mark belongs to the byte encoding, so text keeps whatever it
311        // was written with and the parser rejects a stray U+FEFF.
312        let has_bom = !ignore_cookie && source.starts_with(b"\xef\xbb\xbf");
313        let encoding = if ignore_cookie {
314            None
315        } else {
316            Self::detect_source_encoding(source)
317        };
318        let is_utf8 = encoding.as_deref().is_none_or(Self::is_utf8_encoding);
319        if has_bom && !is_utf8 {
320            let enc = encoding.as_deref().unwrap_or("utf-8");
321            let after_bom = &source[3..];
322            let line_end = after_bom
323                .iter()
324                .position(|&b| b == b'\n')
325                .unwrap_or(after_bom.len());
326            let text = String::from_utf8_lossy(&after_bom[..line_end]).into_owned();
327            let end_offset = text.chars().count() as i32;
328            let msg = format!("encoding problem: {enc} with BOM");
329            let location = self.ctx.new_tuple(vec![
330                self.ctx.new_str(filename).into(),
331                self.ctx.new_int(1).into(),
332                self.ctx.new_int(0).into(),
333                self.ctx.new_str(text).into(),
334                self.ctx.new_int(1).into(),
335                self.ctx.new_int(end_offset).into(),
336            ]);
337            return Err(self
338                .invoke_exception(
339                    self.ctx.exceptions.syntax_error,
340                    vec![self.ctx.new_str(msg.as_str()).into(), location.into()],
341                )
342                .unwrap_or_else(|_| {
343                    self.new_exception_msg(self.ctx.exceptions.syntax_error.to_owned(), msg.into())
344                }));
345        }
346
347        if is_utf8 {
348            let src = if has_bom { &source[3..] } else { source };
349            match core::str::from_utf8(src) {
350                Ok(s) => Ok(s.to_owned()),
351                Err(e) => Err(self.new_non_utf8_syntax_error(filename, src, e.valid_up_to())),
352            }
353        } else {
354            let encoding = encoding.as_deref().unwrap();
355            let bytes = self.ctx.new_bytes(source.to_vec());
356            let decoded = self
357                .state
358                .codec_registry
359                .decode_text(bytes.into(), encoding, None, self)
360                .map_err(|exc| {
361                    if exc.fast_isinstance(self.ctx.exceptions.lookup_error) {
362                        self.new_exception_msg(
363                            self.ctx.exceptions.syntax_error.to_owned(),
364                            format!("unknown encoding for '{filename}': {encoding}").into(),
365                        )
366                    } else {
367                        exc
368                    }
369                })?;
370            Ok(decoded.to_string_lossy().into_owned())
371        }
372    }
373
374    #[cfg(feature = "parser")]
375    pub fn compile_string_object_with_flags(
376        &self,
377        source: &[u8],
378        filename: &str,
379        start: i32,
380        flags: i32,
381        feature_version: i32,
382        optimize: i32,
383    ) -> PyResult<PyObjectRef> {
384        use crate::convert::ToPyException;
385        use crate::stdlib::_ast;
386
387        let cf = CompilerFlags::from_bits_retain(flags);
388        let Some(start) = CompileStart::from_i32(start) else {
389            return Err(self.new_system_error("Invalid start argument passed to Py_CompileString"));
390        };
391        let source =
392            self.decode_source_bytes(source, filename, cf.contains(CompilerFlags::IGNORE_COOKIE))?;
393        let source = source.as_str();
394        let optimize = match optimize {
395            -1 => self.state.config.settings.optimize.min(2),
396            0..=2 => optimize as u8,
397            _ => return Err(self.new_value_error("compile(): invalid optimize value")),
398        };
399        let allow_incomplete = cf.contains(CompilerFlags::ALLOW_INCOMPLETE_INPUT);
400        let type_comments = cf.contains(CompilerFlags::TYPE_COMMENTS);
401        let dont_imply_dedent = cf.contains(CompilerFlags::DONT_IMPLY_DEDENT);
402        let is_ast_only = cf.contains(CompilerFlags::ONLY_AST);
403        let optimized_ast = cf.contains(CompilerFlags::OPTIMIZED_AST);
404        let future_features = compile_future_features_from_flags(flags);
405        let target_version = if is_ast_only {
406            Some(ruff_python_ast::PythonVersion {
407                major: 3,
408                minor: u8::try_from(feature_version).unwrap_or(crate::version::MINOR as u8),
409            })
410        } else {
411            None
412        };
413
414        if is_ast_only {
415            if start == CompileStart::FuncType {
416                return _ast::parse_func_type(self, source, filename, optimize, target_version)
417                    .map_err(|e| (e, Some(source), allow_incomplete).to_pyexception(self));
418            }
419            let (parser_mode, interactive) = match start {
420                CompileStart::Single => (ruff_python_parser::Mode::Module, true),
421                CompileStart::File => (ruff_python_parser::Mode::Module, false),
422                CompileStart::Eval => (ruff_python_parser::Mode::Expression, false),
423                CompileStart::FuncType => unreachable!(),
424            };
425            let parsed = _ast::parse(
426                self,
427                source,
428                filename,
429                parser_mode,
430                optimize,
431                target_version,
432                type_comments,
433                optimized_ast,
434                interactive,
435                future_features,
436                dont_imply_dedent,
437            )
438            .map_err(|e| (e, Some(source), allow_incomplete).to_pyexception(self))?;
439            if start == CompileStart::Single {
440                return _ast::wrap_interactive(self, &parsed);
441            }
442            return Ok(parsed);
443        }
444
445        if type_comments {
446            let parser_mode = match start {
447                CompileStart::Single | CompileStart::File => ruff_python_parser::Mode::Module,
448                CompileStart::Eval => ruff_python_parser::Mode::Expression,
449                CompileStart::FuncType => ruff_python_parser::Mode::Module,
450            };
451            _ast::parse(
452                self,
453                source,
454                filename,
455                parser_mode,
456                optimize,
457                None,
458                type_comments,
459                false,
460                start == CompileStart::Single,
461                future_features,
462                dont_imply_dedent,
463            )
464            .map_err(|e| (e, Some(source), allow_incomplete).to_pyexception(self))?;
465        }
466
467        let mode = match start {
468            CompileStart::Single => compiler::Mode::Single,
469            CompileStart::File => compiler::Mode::Exec,
470            CompileStart::Eval => compiler::Mode::Eval,
471            CompileStart::FuncType => compiler::Mode::BlockExpr,
472        };
473        let mut opts = self.compile_opts();
474        opts.optimize = optimize;
475        opts.allow_top_level_await = cf.contains(CompilerFlags::ALLOW_TOP_LEVEL_AWAIT);
476        opts.future_features = future_features;
477        opts.dont_imply_dedent = dont_imply_dedent;
478        let code = self
479            .compile_with_opts(source, mode, filename, opts)
480            .map_err(|err| {
481                err.into_pyexception_maybe_incomplete(self, Some(source), allow_incomplete)
482            })?;
483        Ok(code.into())
484    }
485
486    pub fn compile(
487        &self,
488        source: &str,
489        mode: compiler::Mode,
490        source_path: impl Into<String>,
491    ) -> Result<PyRef<PyCode>, VmCompileError> {
492        self.compile_with_opts(source, mode, source_path, self.compile_opts())
493    }
494
495    pub fn compile_with_opts(
496        &self,
497        source: &str,
498        mode: compiler::Mode,
499        source_path: impl Into<String>,
500        opts: CompileOpts,
501    ) -> Result<PyRef<PyCode>, VmCompileError> {
502        let source_path = source_path.into();
503        #[cfg(feature = "parser")]
504        {
505            self.emit_tokenizer_syntax_warnings(source, &source_path)
506                .map_err(VmCompileError::Warning)?;
507            self.emit_string_escape_warnings(source, &source_path)
508                .map_err(VmCompileError::Warning)?;
509        }
510        #[cfg(feature = "parser")]
511        let code = {
512            // A warning the filter escalates to an exception is stashed here so
513            // its precise category survives; codegen only sees an abort marker.
514            let escalated: core::cell::Cell<Option<CompileWarningError>> =
515                core::cell::Cell::new(None);
516            let mut syntax_warning_handler = |location, message| {
517                escape_warnings::warn_syntax_at_location(&source_path, location, message, self)
518                    .map_err(|warning| {
519                        escalated.set(Some(warning));
520                        // Recovered below via `escalated`, so this is never surfaced.
521                        compiler::codegen::error::CodegenError {
522                            location: Some(location),
523                            end_location: None,
524                            error: compiler::codegen::error::CodegenErrorType::SyntaxError(
525                                String::new(),
526                            ),
527                            source_path: source_path.clone(),
528                        }
529                    })
530            };
531            let result = compiler::compile_with_syntax_warning_handler(
532                source,
533                mode,
534                &source_path,
535                opts,
536                &mut syntax_warning_handler,
537            );
538            match escalated.take() {
539                Some(warning) => return Err(VmCompileError::Warning(warning)),
540                None => result,
541            }
542        };
543        #[cfg(not(feature = "parser"))]
544        let code = compiler::compile(source, mode, &source_path, opts);
545        let code = code
546            .map(|code| PyCode::new_ref_from_bytecode(self, code))
547            .map_err(VmCompileError::Compile)?;
548        Ok(code)
549    }
550}
551
552/// Scan source for invalid escape sequences in all string literals and emit
553/// SyntaxWarning.
554///
555/// Corresponds to:
556/// - `warn_invalid_escape_sequence()` in `Parser/string_parser.c`
557/// - `_PyTokenizer_warn_invalid_escape_sequence()` in `Parser/tokenizer/helpers.c`
558#[cfg(feature = "parser")]
559mod escape_warnings {
560    use super::*;
561    use crate::warn;
562    use ruff_python_ast::{self as ast, visitor::Visitor};
563    use ruff_text_size::TextRange;
564
565    /// Calculate 1-indexed line number at byte offset in source.
566    fn line_number_at(source: &str, offset: usize) -> usize {
567        source[..offset.min(source.len())]
568            .bytes()
569            .filter(|&b| b == b'\n')
570            .count()
571            + 1
572    }
573
574    fn line_offset_at(source: &str, offset: usize) -> (usize, usize) {
575        let offset = offset.min(source.len());
576        let prefix = &source[..offset];
577        let lineno = prefix.bytes().filter(|&b| b == b'\n').count() + 1;
578        let line_start = prefix.rfind('\n').map_or(0, |index| index + 1);
579        let column = source[line_start..offset].chars().count() + 1;
580        (lineno, column)
581    }
582
583    fn compile_warning_error(
584        exception: PyBaseExceptionRef,
585        source: &str,
586        filename: &str,
587        offset: usize,
588    ) -> CompileWarningError {
589        let (lineno, offset) = line_offset_at(source, offset);
590        CompileWarningError {
591            exception,
592            filename: filename.to_owned(),
593            lineno,
594            offset,
595            replacement: None,
596        }
597    }
598
599    /// Get content bounds (start, end byte offsets) of a quoted string literal,
600    /// excluding prefix characters and quote delimiters.
601    fn content_bounds(source: &str, range: TextRange) -> Option<(usize, usize)> {
602        let s = range.start().to_usize();
603        let e = range.end().to_usize();
604        if s >= e || e > source.len() {
605            return None;
606        }
607        let bytes = &source.as_bytes()[s..e];
608        // Skip prefix (u, b, r, etc.) to find the first quote character.
609        let qi = bytes.iter().position(|&c| c == b'\'' || c == b'"')?;
610        let qc = bytes[qi];
611        let ql = if bytes.get(qi + 1) == Some(&qc) && bytes.get(qi + 2) == Some(&qc) {
612            3
613        } else {
614            1
615        };
616        let cs = s + qi + ql;
617        let ce = e.checked_sub(ql)?;
618        if cs <= ce { Some((cs, ce)) } else { None }
619    }
620
621    enum InvalidEscape {
622        Char { ch: char, offset: usize },
623        Octal { digits: [u8; 3], offset: usize },
624    }
625
626    impl InvalidEscape {
627        fn offset(&self) -> usize {
628            match self {
629                Self::Char { offset, .. } | Self::Octal { offset, .. } => *offset,
630            }
631        }
632
633        fn octal_text(digits: [u8; 3]) -> [char; 3] {
634            [digits[0] as char, digits[1] as char, digits[2] as char]
635        }
636
637        fn warning_message(&self) -> String {
638            match self {
639                Self::Char { ch, .. } => format!(
640                    "\"\\{ch}\" is an invalid escape sequence. \
641                     Such sequences will not work in the future. \
642                     Did you mean \"\\\\{ch}\"? A raw string is also an option."
643                ),
644                Self::Octal { digits, .. } => {
645                    let [a, b, c] = Self::octal_text(*digits);
646                    format!(
647                        "\"\\{a}{b}{c}\" is an invalid octal escape sequence. \
648                         Such sequences will not work in the future. \
649                         Did you mean \"\\\\{a}{b}{c}\"? A raw string is also an option."
650                    )
651                }
652            }
653        }
654
655        fn syntax_error_message(&self) -> String {
656            match self {
657                Self::Char { ch, .. } => format!(
658                    "\"\\{ch}\" is an invalid escape sequence. \
659                     Did you mean \"\\\\{ch}\"? A raw string is also an option."
660                ),
661                Self::Octal { digits, .. } => {
662                    let [a, b, c] = Self::octal_text(*digits);
663                    format!(
664                        "\"\\{a}{b}{c}\" is an invalid octal escape sequence. \
665                         Did you mean \"\\\\{a}{b}{c}\"? A raw string is also an option."
666                    )
667                }
668            }
669        }
670    }
671
672    /// Scan `source[start..end]` for the first invalid escape sequence.
673    ///
674    /// When `is_bytes` is true, `\u`, `\U`, and `\N` are treated as invalid
675    /// (bytes literals only support byte-oriented escapes).
676    ///
677    /// Only reports the **first** invalid escape per string literal, matching
678    /// `_PyUnicode_DecodeUnicodeEscapeInternal2` which stores only the first
679    /// `first_invalid_escape_char`.
680    fn first_invalid_escape(
681        source: &str,
682        start: usize,
683        end: usize,
684        is_bytes: bool,
685    ) -> Option<InvalidEscape> {
686        let raw = &source[start..end];
687        let mut chars = raw.char_indices().peekable();
688        while let Some((i, ch)) = chars.next() {
689            if ch != '\\' {
690                continue;
691            }
692            let Some((_, next)) = chars.next() else {
693                break;
694            };
695            let offset = start + i;
696            let valid = match next {
697                '\\' | '\'' | '"' | 'a' | 'b' | 'f' | 'n' | 'r' | 't' | 'v' => true,
698                '\n' => true,
699                '\r' => {
700                    if matches!(chars.peek(), Some(&(_, '\n'))) {
701                        chars.next();
702                    }
703                    true
704                }
705                '0'..='7' => {
706                    let mut digits = [next as u8, 0, 0];
707                    let mut len = 1;
708                    for _ in 0..2 {
709                        if matches!(chars.peek(), Some(&(_, '0'..='7'))) {
710                            let (_, digit) = chars.next().unwrap();
711                            digits[len] = digit as u8;
712                            len += 1;
713                        } else {
714                            break;
715                        }
716                    }
717                    if len == 3 {
718                        let value = ((digits[0] - b'0') as u16) * 64
719                            + ((digits[1] - b'0') as u16) * 8
720                            + (digits[2] - b'0') as u16;
721                        if value > 0o377 {
722                            return Some(InvalidEscape::Octal { digits, offset });
723                        }
724                    }
725                    true
726                }
727                'x' | 'u' | 'U' => {
728                    // \u and \U are only valid in string literals, not bytes
729                    if is_bytes && next != 'x' {
730                        false
731                    } else {
732                        let count = match next {
733                            'x' => 2,
734                            'u' => 4,
735                            'U' => 8,
736                            _ => unreachable!(),
737                        };
738                        for _ in 0..count {
739                            if chars.peek().is_some_and(|&(_, c)| c.is_ascii_hexdigit()) {
740                                chars.next();
741                            } else {
742                                break;
743                            }
744                        }
745                        true
746                    }
747                }
748                'N' => {
749                    // \N{name} is only valid in string literals, not bytes
750                    if is_bytes {
751                        false
752                    } else {
753                        if matches!(chars.peek(), Some(&(_, '{'))) {
754                            chars.next();
755                            for (_, c) in chars.by_ref() {
756                                if c == '}' {
757                                    break;
758                                }
759                            }
760                        }
761                        true
762                    }
763                }
764                _ => false,
765            };
766            if !valid {
767                return Some(InvalidEscape::Char { ch: next, offset });
768            }
769        }
770        None
771    }
772
773    /// Emit `SyntaxWarning` for an invalid escape sequence.
774    ///
775    /// `warn_invalid_escape_sequence()` in `Parser/string_parser.c`
776    fn warn_invalid_escape_sequence(
777        source: &str,
778        escape: InvalidEscape,
779        filename: &str,
780        vm: &VirtualMachine,
781    ) -> Result<(), CompileWarningError> {
782        let (lineno, column) = line_offset_at(source, escape.offset());
783        let warning = escape.warning_message();
784        let syntax_error = escape.syntax_error_message();
785        let fname = vm.ctx.new_str(filename);
786        warn::warn_explicit(
787            Some(vm.ctx.exceptions.syntax_warning.to_owned()),
788            vm.ctx.new_str(warning).into(),
789            fname,
790            lineno,
791            None,
792            vm.ctx.none(),
793            None,
794            None,
795            vm,
796        )
797        .map_err(|exception| CompileWarningError {
798            exception,
799            filename: filename.to_owned(),
800            lineno,
801            offset: column,
802            replacement: Some(SyntaxErrorReplacement {
803                message: syntax_error,
804                end_offset: column + 2,
805            }),
806        })
807    }
808
809    fn warn_syntax_at_offset(
810        source: &str,
811        filename: &str,
812        offset: usize,
813        message: String,
814        vm: &VirtualMachine,
815    ) -> Result<(), CompileWarningError> {
816        let lineno = line_number_at(source, offset);
817        let fname = vm.ctx.new_str(filename);
818        let message = vm.ctx.new_str(message);
819        warn::warn_explicit(
820            Some(vm.ctx.exceptions.syntax_warning.to_owned()),
821            message.into(),
822            fname,
823            lineno,
824            None,
825            vm.ctx.none(),
826            None,
827            None,
828            vm,
829        )
830        .map_err(|err| compile_warning_error(err, source, filename, offset))
831    }
832
833    pub(super) fn warn_syntax_at_location(
834        filename: &str,
835        location: compiler::core::SourceLocation,
836        message: String,
837        vm: &VirtualMachine,
838    ) -> Result<(), CompileWarningError> {
839        let fname = vm.ctx.new_str(filename);
840        let message = vm.ctx.new_str(message);
841        warn::warn_explicit(
842            Some(vm.ctx.exceptions.syntax_warning.to_owned()),
843            message.into(),
844            fname,
845            location.line.get(),
846            None,
847            vm.ctx.none(),
848            None,
849            None,
850            vm,
851        )
852        .map_err(|exception| CompileWarningError {
853            exception,
854            filename: filename.to_owned(),
855            lineno: location.line.get(),
856            offset: location.character_offset.get(),
857            replacement: None,
858        })
859    }
860
861    fn is_ascii_identifier_char(byte: u8) -> bool {
862        byte == b'_' || byte.is_ascii_alphanumeric()
863    }
864
865    fn numeric_keyword_suffix(rest: &[u8]) -> bool {
866        rest.starts_with(b"and")
867            || rest.starts_with(b"else")
868            || rest.starts_with(b"for")
869            || rest.starts_with(b"if")
870            || rest.starts_with(b"in")
871            || rest.starts_with(b"is")
872            || rest.starts_with(b"or")
873            || rest.starts_with(b"not")
874    }
875
876    fn consume_decimal_digits(bytes: &[u8], mut index: usize) -> usize {
877        while index < bytes.len() {
878            match bytes[index] {
879                b'0'..=b'9' => index += 1,
880                b'_' if bytes
881                    .get(index + 1)
882                    .is_some_and(|byte| byte.is_ascii_digit()) =>
883                {
884                    index += 2;
885                }
886                _ => break,
887            }
888        }
889        index
890    }
891
892    fn consume_radix_digits(
893        bytes: &[u8],
894        mut index: usize,
895        is_digit: impl Fn(u8) -> bool,
896    ) -> usize {
897        while index < bytes.len() {
898            if is_digit(bytes[index]) {
899                index += 1;
900            } else if bytes.get(index) == Some(&b'_')
901                && bytes.get(index + 1).is_some_and(|&byte| is_digit(byte))
902            {
903                index += 2;
904            } else {
905                break;
906            }
907        }
908        index
909    }
910
911    fn number_literal_end(bytes: &[u8], start: usize) -> Option<(&'static str, usize)> {
912        if bytes.get(start) == Some(&b'.') {
913            if !bytes
914                .get(start + 1)
915                .is_some_and(|byte| byte.is_ascii_digit())
916            {
917                return None;
918            }
919            let mut index = consume_decimal_digits(bytes, start + 1);
920            index = consume_exponent(bytes, index);
921            if matches!(bytes.get(index), Some(b'j' | b'J')) {
922                return Some(("imaginary", index + 1));
923            }
924            return Some(("decimal", index));
925        }
926
927        if !bytes.get(start).is_some_and(|byte| byte.is_ascii_digit()) {
928            return None;
929        }
930
931        if bytes.get(start) == Some(&b'0') {
932            match bytes.get(start + 1) {
933                Some(b'x' | b'X') => {
934                    let end =
935                        consume_radix_digits(bytes, start + 2, |byte| byte.is_ascii_hexdigit());
936                    return Some(("hexadecimal", end));
937                }
938                Some(b'o' | b'O') => {
939                    let end =
940                        consume_radix_digits(bytes, start + 2, |byte| matches!(byte, b'0'..=b'7'));
941                    return Some(("octal", end));
942                }
943                Some(b'b' | b'B') => {
944                    let end =
945                        consume_radix_digits(bytes, start + 2, |byte| matches!(byte, b'0' | b'1'));
946                    return Some(("binary", end));
947                }
948                _ => {}
949            }
950        }
951
952        let mut index = consume_decimal_digits(bytes, start);
953        if bytes.get(index) == Some(&b'.') {
954            index = consume_decimal_digits(bytes, index + 1);
955        }
956        index = consume_exponent(bytes, index);
957        if matches!(bytes.get(index), Some(b'j' | b'J')) {
958            return Some(("imaginary", index + 1));
959        }
960        Some(("decimal", index))
961    }
962
963    fn consume_exponent(bytes: &[u8], index: usize) -> usize {
964        if !matches!(bytes.get(index), Some(b'e' | b'E')) {
965            return index;
966        }
967        let mut cursor = index + 1;
968        if matches!(bytes.get(cursor), Some(b'+' | b'-')) {
969            cursor += 1;
970        }
971        if bytes.get(cursor).is_some_and(|byte| byte.is_ascii_digit()) {
972            consume_decimal_digits(bytes, cursor)
973        } else {
974            index
975        }
976    }
977
978    fn skip_quoted_string(bytes: &[u8], mut index: usize) -> usize {
979        let quote = bytes[index];
980        let triple = bytes.get(index + 1) == Some(&quote) && bytes.get(index + 2) == Some(&quote);
981        let quote_len = if triple { 3 } else { 1 };
982        index += quote_len;
983        while index < bytes.len() {
984            if bytes[index] == b'\\' {
985                index = (index + 2).min(bytes.len());
986            } else if triple
987                && bytes.get(index) == Some(&quote)
988                && bytes.get(index + 1) == Some(&quote)
989                && bytes.get(index + 2) == Some(&quote)
990            {
991                return index + 3;
992            } else if !triple && bytes[index] == quote {
993                return index + 1;
994            } else {
995                index += 1;
996            }
997        }
998        index
999    }
1000
1001    #[derive(Clone, Copy)]
1002    enum StringPrefix {
1003        None,
1004        Invalid,
1005        Valid {
1006            is_raw: bool,
1007            is_bytes: bool,
1008            is_interpolated: bool,
1009        },
1010    }
1011
1012    fn is_string_prefix_letter(byte: u8) -> bool {
1013        matches!(
1014            byte,
1015            b'r' | b'R' | b'b' | b'B' | b'u' | b'U' | b'f' | b'F' | b't' | b'T'
1016        )
1017    }
1018
1019    fn prefix_continues_identifier(byte: u8) -> bool {
1020        byte == b'_' || byte.is_ascii_alphabetic() || byte >= 0x80
1021    }
1022
1023    fn two_char_prefix(first: u8, second: u8) -> Option<(bool, bool, bool)> {
1024        match [first.to_ascii_lowercase(), second.to_ascii_lowercase()] {
1025            [b'r', b'f' | b't'] | [b'f' | b't', b'r'] => Some((true, false, true)),
1026            [b'r', b'b'] | [b'b', b'r'] => Some((true, true, false)),
1027            _ => None,
1028        }
1029    }
1030
1031    /// Prefix letters immediately before a quote, only when they form their own token.
1032    fn string_prefix(bytes: &[u8], quote_index: usize) -> StringPrefix {
1033        let mut taken = [0u8; 2];
1034        let mut count = 0;
1035        let mut index = quote_index;
1036        while index > 0 && count < 2 {
1037            let byte = bytes[index - 1];
1038            if !is_string_prefix_letter(byte) {
1039                break;
1040            }
1041            taken[count] = byte;
1042            count += 1;
1043            index -= 1;
1044        }
1045        if count == 0 {
1046            return StringPrefix::None;
1047        }
1048        if index > 0 && prefix_continues_identifier(bytes[index - 1]) {
1049            return StringPrefix::None;
1050        }
1051        if count == 2 {
1052            return match two_char_prefix(taken[1], taken[0]) {
1053                Some((is_raw, is_bytes, is_interpolated)) => StringPrefix::Valid {
1054                    is_raw,
1055                    is_bytes,
1056                    is_interpolated,
1057                },
1058                None => StringPrefix::Invalid,
1059            };
1060        }
1061        let (is_raw, is_bytes, is_interpolated) = match taken[0] {
1062            b'r' | b'R' => (true, false, false),
1063            b'b' | b'B' => (false, true, false),
1064            b'f' | b'F' | b't' | b'T' => (false, false, true),
1065            _ => (false, false, false),
1066        };
1067        StringPrefix::Valid {
1068            is_raw,
1069            is_bytes,
1070            is_interpolated,
1071        }
1072    }
1073
1074    fn text_range_from_bounds(start: usize, end: usize) -> TextRange {
1075        TextRange::new(
1076            ruff_text_size::TextSize::try_from(start).unwrap_or_default(),
1077            ruff_text_size::TextSize::try_from(end).unwrap_or_default(),
1078        )
1079    }
1080
1081    fn warn_quoted_literal(
1082        source: &str,
1083        quote_index: usize,
1084        end: usize,
1085        is_bytes: bool,
1086        filename: &str,
1087        vm: &VirtualMachine,
1088    ) -> Result<(), CompileWarningError> {
1089        if let Some((start, content_end)) =
1090            content_bounds(source, text_range_from_bounds(quote_index, end))
1091            && let Some(escape) = first_invalid_escape(source, start, content_end, is_bytes)
1092        {
1093            warn_invalid_escape_sequence(source, escape, filename, vm)?;
1094        }
1095        Ok(())
1096    }
1097
1098    fn warn_fstring_literal_part(
1099        source: &str,
1100        start: usize,
1101        end: usize,
1102        filename: &str,
1103        vm: &VirtualMachine,
1104    ) -> Result<(), CompileWarningError> {
1105        if start >= end || end > source.len() {
1106            return Ok(());
1107        }
1108        if let Some(escape) = first_invalid_escape(source, start, end, false) {
1109            return warn_invalid_escape_sequence(source, escape, filename, vm);
1110        }
1111        let trailing_bs = source.as_bytes()[start..end]
1112            .iter()
1113            .rev()
1114            .take_while(|&&byte| byte == b'\\')
1115            .count();
1116        if trailing_bs % 2 == 1
1117            && let Some(&after) = source.as_bytes().get(end)
1118            && (after == b'{' || after == b'}')
1119        {
1120            warn_invalid_escape_sequence(
1121                source,
1122                InvalidEscape::Char {
1123                    ch: after as char,
1124                    offset: end - 1,
1125                },
1126                filename,
1127                vm,
1128            )?;
1129        }
1130        Ok(())
1131    }
1132
1133    fn find_interpolation_end(bytes: &[u8], open: usize, limit: usize) -> usize {
1134        let mut index = open + 1;
1135        let mut depth: u32 = 1;
1136        let mut paren: u32 = 0;
1137        let mut bracket: u32 = 0;
1138        while index < limit {
1139            match bytes[index] {
1140                b'#' if paren == 0 && bracket == 0 && depth == 1 => {
1141                    while index < limit && bytes[index] != b'\n' {
1142                        index += 1;
1143                    }
1144                }
1145                b'\'' | b'"' => {
1146                    index = skip_quoted_string(bytes, index).max(index + 1);
1147                }
1148                b'(' => {
1149                    paren += 1;
1150                    index += 1;
1151                }
1152                b')' => {
1153                    paren = paren.saturating_sub(1);
1154                    index += 1;
1155                }
1156                b'[' => {
1157                    bracket += 1;
1158                    index += 1;
1159                }
1160                b']' => {
1161                    bracket = bracket.saturating_sub(1);
1162                    index += 1;
1163                }
1164                b'{' => {
1165                    depth += 1;
1166                    index += 1;
1167                }
1168                b'}' => {
1169                    depth -= 1;
1170                    index += 1;
1171                    if depth == 0 {
1172                        return index;
1173                    }
1174                }
1175                b'\\' => index = (index + 2).min(limit),
1176                _ => index += 1,
1177            }
1178        }
1179        limit
1180    }
1181
1182    fn scan_interpolated_content(
1183        source: &str,
1184        start: usize,
1185        end: usize,
1186        is_raw: bool,
1187        filename: &str,
1188        vm: &VirtualMachine,
1189    ) -> Result<(), CompileWarningError> {
1190        let bytes = source.as_bytes();
1191        let mut index = start;
1192        let mut literal_start = start;
1193        while index < end {
1194            match bytes[index] {
1195                b'{' if bytes.get(index + 1) == Some(&b'{') => {
1196                    index += 2;
1197                }
1198                b'{' => {
1199                    if !is_raw {
1200                        warn_fstring_literal_part(source, literal_start, index, filename, vm)?;
1201                    }
1202                    let close = find_interpolation_end(bytes, index, end);
1203                    let body_end = if close > index + 1 { close - 1 } else { close };
1204                    emit_string_escape_warnings_in_range(
1205                        source,
1206                        index + 1,
1207                        body_end,
1208                        Some(is_raw),
1209                        filename,
1210                        vm,
1211                    )?;
1212                    index = close.max(index + 1);
1213                    literal_start = index;
1214                }
1215                b'}' if bytes.get(index + 1) == Some(&b'}') => {
1216                    index += 2;
1217                }
1218                b'\\' => index = (index + 2).min(end),
1219                _ => index += 1,
1220            }
1221        }
1222        if !is_raw {
1223            warn_fstring_literal_part(source, literal_start, end, filename, vm)?;
1224        }
1225        Ok(())
1226    }
1227
1228    /// Scan quoted literals without a successful parse, matching the tokenizer
1229    /// path that warns before the parser rejects the rest of the source.
1230    fn emit_string_escape_warnings_unparsed(
1231        source: &str,
1232        filename: &str,
1233        vm: &VirtualMachine,
1234    ) -> Result<(), CompileWarningError> {
1235        emit_string_escape_warnings_in_range(source, 0, source.len(), None, filename, vm)
1236    }
1237
1238    fn emit_string_escape_warnings_in_range(
1239        source: &str,
1240        start: usize,
1241        end: usize,
1242        format_spec_raw: Option<bool>,
1243        filename: &str,
1244        vm: &VirtualMachine,
1245    ) -> Result<(), CompileWarningError> {
1246        let bytes = source.as_bytes();
1247        let end = end.min(bytes.len());
1248        let mut index = start;
1249        let mut paren: u32 = 0;
1250        let mut bracket: u32 = 0;
1251        let mut brace: u32 = 0;
1252        while index < end {
1253            match bytes[index] {
1254                b'#' => {
1255                    while index < end && bytes[index] != b'\n' {
1256                        index += 1;
1257                    }
1258                }
1259                b'\'' | b'"' => {
1260                    let prefix = string_prefix(bytes, index);
1261                    let quote_end = skip_quoted_string(bytes, index).min(bytes.len());
1262                    match prefix {
1263                        StringPrefix::Valid {
1264                            is_raw,
1265                            is_bytes,
1266                            is_interpolated,
1267                        } => {
1268                            if is_interpolated {
1269                                if let Some((content_start, content_end)) =
1270                                    content_bounds(source, text_range_from_bounds(index, quote_end))
1271                                {
1272                                    scan_interpolated_content(
1273                                        source,
1274                                        content_start,
1275                                        content_end.min(end),
1276                                        is_raw,
1277                                        filename,
1278                                        vm,
1279                                    )?;
1280                                }
1281                            } else if !is_raw {
1282                                warn_quoted_literal(
1283                                    source, index, quote_end, is_bytes, filename, vm,
1284                                )?;
1285                            }
1286                        }
1287                        StringPrefix::None => {
1288                            warn_quoted_literal(source, index, quote_end, false, filename, vm)?;
1289                        }
1290                        StringPrefix::Invalid => {}
1291                    }
1292                    index = quote_end.max(index + 1);
1293                }
1294                b':' if let Some(is_raw) = format_spec_raw
1295                    && paren == 0
1296                    && bracket == 0
1297                    && brace == 0 =>
1298                {
1299                    scan_interpolated_content(source, index + 1, end, is_raw, filename, vm)?;
1300                    return Ok(());
1301                }
1302                b'(' if format_spec_raw.is_some() => {
1303                    paren += 1;
1304                    index += 1;
1305                }
1306                b')' if format_spec_raw.is_some() => {
1307                    paren = paren.saturating_sub(1);
1308                    index += 1;
1309                }
1310                b'[' if format_spec_raw.is_some() => {
1311                    bracket += 1;
1312                    index += 1;
1313                }
1314                b']' if format_spec_raw.is_some() => {
1315                    bracket = bracket.saturating_sub(1);
1316                    index += 1;
1317                }
1318                b'{' if format_spec_raw.is_some() => {
1319                    brace += 1;
1320                    index += 1;
1321                }
1322                b'}' if format_spec_raw.is_some() => {
1323                    brace = brace.saturating_sub(1);
1324                    index += 1;
1325                }
1326                _ => index += 1,
1327            }
1328        }
1329        Ok(())
1330    }
1331
1332    fn emit_numeric_literal_warnings(
1333        source: &str,
1334        filename: &str,
1335        vm: &VirtualMachine,
1336    ) -> Result<(), CompileWarningError> {
1337        let bytes = source.as_bytes();
1338        let mut index = 0;
1339        while index < bytes.len() {
1340            match bytes[index] {
1341                b'#' => {
1342                    while index < bytes.len() && bytes[index] != b'\n' {
1343                        index += 1;
1344                    }
1345                }
1346                b'\'' | b'"' => {
1347                    index = skip_quoted_string(bytes, index);
1348                }
1349                byte if byte >= 0x80 || byte == b'_' || byte.is_ascii_alphabetic() => {
1350                    index += 1;
1351                    while index < bytes.len()
1352                        && (bytes[index] >= 0x80 || is_ascii_identifier_char(bytes[index]))
1353                    {
1354                        index += 1;
1355                    }
1356                }
1357                b'.' | b'0'..=b'9' => {
1358                    let Some((kind, end)) = number_literal_end(bytes, index) else {
1359                        index += 1;
1360                        continue;
1361                    };
1362                    if end > index && numeric_keyword_suffix(&bytes[end..]) {
1363                        warn_syntax_at_offset(
1364                            source,
1365                            filename,
1366                            index,
1367                            format!("invalid {kind} literal"),
1368                            vm,
1369                        )?;
1370                    }
1371                    index = end.max(index + 1);
1372                }
1373                _ => index += 1,
1374            }
1375        }
1376        Ok(())
1377    }
1378
1379    struct EscapeWarningVisitor<'a> {
1380        source: &'a str,
1381        filename: &'a str,
1382        vm: &'a VirtualMachine,
1383        error: Option<CompileWarningError>,
1384        /// This pass runs before the compile that rejects an over-nested
1385        /// tree, so it has to stop itself.
1386        depth: usize,
1387        depth_limit: usize,
1388    }
1389
1390    impl<'a> EscapeWarningVisitor<'a> {
1391        fn record_warning(&mut self, result: Result<(), CompileWarningError>) {
1392            if self.error.is_none()
1393                && let Err(err) = result
1394            {
1395                self.error = Some(err);
1396            }
1397        }
1398
1399        /// Check a quoted string/bytes literal for invalid escapes.
1400        /// The range must include the prefix and quote delimiters.
1401        fn check_quoted_literal(&mut self, range: TextRange, is_bytes: bool) {
1402            if let Some((start, end)) = content_bounds(self.source, range)
1403                && let Some(escape) = first_invalid_escape(self.source, start, end, is_bytes)
1404            {
1405                let result =
1406                    warn_invalid_escape_sequence(self.source, escape, self.filename, self.vm);
1407                self.record_warning(result);
1408            }
1409        }
1410
1411        /// Check an f-string literal element for invalid escapes.
1412        /// The range covers content only (no prefix/quotes).
1413        ///
1414        /// Also handles `\{` / `\}` at the literal–interpolation boundary,
1415        /// equivalent to `_PyTokenizer_warn_invalid_escape_sequence` handling
1416        /// `FSTRING_MIDDLE` / `FSTRING_END` tokens.
1417        fn check_fstring_literal(&mut self, range: TextRange) {
1418            let start = range.start().to_usize();
1419            let end = range.end().to_usize();
1420            if start >= end || end > self.source.len() {
1421                return;
1422            }
1423            if let Some(escape) = first_invalid_escape(self.source, start, end, false) {
1424                let result =
1425                    warn_invalid_escape_sequence(self.source, escape, self.filename, self.vm);
1426                self.record_warning(result);
1427                return;
1428            }
1429            // In CPython, _PyTokenizer_warn_invalid_escape_sequence handles
1430            // `\{` and `\}` for FSTRING_MIDDLE/FSTRING_END tokens.  Ruff
1431            // splits the literal element before the interpolation delimiter,
1432            // so the `\` sits at the end of the literal range and the `{`/`}`
1433            // sits just after it.  Only warn when the number of trailing
1434            // backslashes is odd (an even count means they are all escaped).
1435            let trailing_bs = self.source.as_bytes()[start..end]
1436                .iter()
1437                .rev()
1438                .take_while(|&&b| b == b'\\')
1439                .count();
1440            if trailing_bs % 2 == 1
1441                && let Some(&after) = self.source.as_bytes().get(end)
1442                && (after == b'{' || after == b'}')
1443            {
1444                let result = warn_invalid_escape_sequence(
1445                    self.source,
1446                    InvalidEscape::Char {
1447                        ch: after as char,
1448                        offset: end - 1,
1449                    },
1450                    self.filename,
1451                    self.vm,
1452                );
1453                self.record_warning(result);
1454            }
1455        }
1456
1457        /// Visit f-string elements, checking literals and recursing into
1458        /// interpolation expressions and format specs.
1459        fn visit_fstring_elements(&mut self, elements: &'a ast::InterpolatedStringElements) {
1460            for element in elements {
1461                if self.error.is_some() {
1462                    return;
1463                }
1464                match element {
1465                    ast::InterpolatedStringElement::Literal(lit) => {
1466                        self.check_fstring_literal(lit.range);
1467                    }
1468                    ast::InterpolatedStringElement::Interpolation(interp) => {
1469                        self.visit_expr(&interp.expression);
1470                        if let Some(spec) = &interp.format_spec {
1471                            self.visit_fstring_elements(&spec.elements);
1472                        }
1473                    }
1474                }
1475            }
1476        }
1477    }
1478
1479    impl<'a> Visitor<'a> for EscapeWarningVisitor<'a> {
1480        fn visit_expr(&mut self, expr: &'a ast::Expr) {
1481            if self.error.is_some() {
1482                return;
1483            }
1484            match expr {
1485                // Regular string literals — decode_unicode_with_escapes path
1486                ast::Expr::StringLiteral(string) => {
1487                    for part in string.value.as_slice() {
1488                        if !matches!(
1489                            part.flags.prefix(),
1490                            ast::str_prefix::StringLiteralPrefix::Raw { .. }
1491                        ) {
1492                            self.check_quoted_literal(part.range, false);
1493                        }
1494                    }
1495                }
1496                // Byte string literals — decode_bytes_with_escapes path
1497                ast::Expr::BytesLiteral(bytes) => {
1498                    for part in bytes.value.as_slice() {
1499                        if !matches!(
1500                            part.flags.prefix(),
1501                            ast::str_prefix::ByteStringPrefix::Raw { .. }
1502                        ) {
1503                            self.check_quoted_literal(part.range, true);
1504                        }
1505                    }
1506                }
1507                // F-string literals — tokenizer + string_parser paths
1508                ast::Expr::FString(fstring_expr) => {
1509                    for part in fstring_expr.value.as_slice() {
1510                        match part {
1511                            ast::FStringPart::Literal(string_lit) => {
1512                                // Plain string part in f-string concatenation
1513                                if !matches!(
1514                                    string_lit.flags.prefix(),
1515                                    ast::str_prefix::StringLiteralPrefix::Raw { .. }
1516                                ) {
1517                                    self.check_quoted_literal(string_lit.range, false);
1518                                }
1519                            }
1520                            ast::FStringPart::FString(fstring) => {
1521                                if matches!(
1522                                    fstring.flags.prefix(),
1523                                    ast::str_prefix::FStringPrefix::Raw { .. }
1524                                ) {
1525                                    continue;
1526                                }
1527                                self.visit_fstring_elements(&fstring.elements);
1528                            }
1529                        }
1530                    }
1531                }
1532                _ => {
1533                    if self.depth < self.depth_limit {
1534                        self.depth += 1;
1535                        ast::visitor::walk_expr(self, expr);
1536                        self.depth -= 1;
1537                    }
1538                }
1539            }
1540        }
1541    }
1542
1543    impl VirtualMachine {
1544        /// Emit tokenizer-level SyntaxWarnings raised before
1545        /// code generation.
1546        pub(super) fn emit_tokenizer_syntax_warnings(
1547            &self,
1548            source: &str,
1549            filename: &str,
1550        ) -> Result<(), CompileWarningError> {
1551            emit_numeric_literal_warnings(source, filename, self)
1552        }
1553
1554        /// Walk all string literals in `source` and emit `SyntaxWarning` for
1555        /// each that contains an invalid escape sequence.
1556        pub(super) fn emit_string_escape_warnings(
1557            &self,
1558            source: &str,
1559            filename: &str,
1560        ) -> Result<(), CompileWarningError> {
1561            // The compile that follows rejects this source; parsing it here
1562            // would build a tree that exhausts the stack when dropped.
1563            let source_file = compiler::core::SourceFileBuilder::new(filename, source).finish();
1564            if compiler::pre_parse_source_error(&source_file).is_err() {
1565                return Ok(());
1566            }
1567            let Ok(parsed) =
1568                ruff_python_parser::parse(source, ruff_python_parser::Mode::Module.into())
1569            else {
1570                return emit_string_escape_warnings_unparsed(source, filename, self);
1571            };
1572            let ast = parsed.into_syntax();
1573            let mut visitor = EscapeWarningVisitor {
1574                source,
1575                filename,
1576                vm: self,
1577                error: None,
1578                depth: 0,
1579                depth_limit: compiler::CompileOpts::default().recursion_limit,
1580            };
1581            match &ast {
1582                ast::Mod::Module(module) => {
1583                    for stmt in &module.body {
1584                        visitor.visit_stmt(stmt);
1585                    }
1586                }
1587                ast::Mod::Expression(expr) => {
1588                    visitor.visit_expr(&expr.body);
1589                }
1590            }
1591            visitor.error.map_or(Ok(()), Err)
1592        }
1593    }
1594
1595    #[cfg(test)]
1596    mod tests {
1597        use super::*;
1598        use crate::{Interpreter, builtins::PyTuple};
1599
1600        fn install_syntax_warning_error_filter(vm: &VirtualMachine) {
1601            let error_filter = PyTuple::new_ref(
1602                vec![
1603                    vm.ctx.new_str("error").into(),
1604                    vm.ctx.none(),
1605                    vm.ctx.exceptions.syntax_warning.as_object().to_owned(),
1606                    vm.ctx.none(),
1607                    vm.ctx.new_int(0).into(),
1608                ],
1609                &vm.ctx,
1610            );
1611            vm.state
1612                .warnings
1613                .filters
1614                .borrow_vec_mut()
1615                .insert(0, error_filter.into());
1616            vm.state.warnings.filters_mutated();
1617        }
1618
1619        fn first_compiler_warning(source: &str) -> String {
1620            Interpreter::without_stdlib(Default::default()).enter(|vm| {
1621                install_syntax_warning_error_filter(vm);
1622                let err = vm
1623                    .compile(source, compiler::Mode::Exec, "<test>")
1624                    .expect_err("expected compiler SyntaxWarning");
1625                let exception = err.into_pyexception(vm, Some(source));
1626                exception
1627                    .as_object()
1628                    .str(vm)
1629                    .expect("warning message should stringify")
1630                    .as_wtf8()
1631                    .to_string()
1632            })
1633        }
1634
1635        fn compile_error_message(source: &str) -> String {
1636            Interpreter::without_stdlib(Default::default()).enter(|vm| {
1637                install_syntax_warning_error_filter(vm);
1638                let err = match vm.compile(source, compiler::Mode::Exec, "<test>") {
1639                    Ok(_) => panic!("expected compile error"),
1640                    Err(err) => err,
1641                };
1642                err.into_pyexception(vm, Some(source))
1643                    .as_object()
1644                    .str(vm)
1645                    .expect("compile error should stringify")
1646                    .as_wtf8()
1647                    .to_string()
1648            })
1649        }
1650
1651        #[test]
1652        fn ast_only_compile_honors_barry_as_flufl() {
1653            Interpreter::without_stdlib(Default::default()).enter(|vm| {
1654                let flags = CompilerFlags::ONLY_AST.bits();
1655                vm.compile_string_object_with_flags(
1656                    b"from __future__ import barry_as_FLUFL\n2 <> 3\n",
1657                    "<test>",
1658                    CompileStart::File.as_i32(),
1659                    flags,
1660                    -1,
1661                    -1,
1662                )
1663                .expect("PyCF_ONLY_AST should accept <> in Barry mode");
1664
1665                let err = vm
1666                    .compile_string_object_with_flags(
1667                        b"from __future__ import barry_as_FLUFL\n2 != 3\n",
1668                        "<test>",
1669                        CompileStart::File.as_i32(),
1670                        flags,
1671                        -1,
1672                        -1,
1673                    )
1674                    .expect_err("PyCF_ONLY_AST should reject != in Barry mode");
1675                assert!(
1676                    err.as_object()
1677                        .str(vm)
1678                        .unwrap()
1679                        .as_wtf8()
1680                        .to_string()
1681                        .contains("with Barry as BDFL")
1682                );
1683            });
1684        }
1685
1686        #[test]
1687        fn type_comment_preparse_honors_inherited_barry_as_flufl() {
1688            Interpreter::without_stdlib(Default::default()).enter(|vm| {
1689                let flags = CompilerFlags::TYPE_COMMENTS.bits()
1690                    | crate::bytecode::CodeFlags::FUTURE_BARRY_AS_BDFL.bits() as i32;
1691                vm.compile_string_object_with_flags(
1692                    b"2 <> 3\n",
1693                    "<test>",
1694                    CompileStart::File.as_i32(),
1695                    flags,
1696                    -1,
1697                    -1,
1698                )
1699                .expect("type-comment preparse should accept <> in inherited Barry mode");
1700            });
1701        }
1702
1703        #[test]
1704        fn codegen_caller_warning_precedes_later_return_error() {
1705            let message = compile_error_message("(1)()\nreturn\n");
1706            assert!(
1707                message.contains("'int' object is not callable"),
1708                "expected caller SyntaxWarning first, got {message:?}"
1709            );
1710        }
1711
1712        #[test]
1713        fn symboltable_error_still_precedes_codegen_caller_warning() {
1714            let message = compile_error_message("(1)()\ndef f():\n    from x import *\n");
1715            assert!(
1716                message.contains("import * only allowed at module level"),
1717                "expected symboltable error first, got {message:?}"
1718            );
1719        }
1720
1721        #[test]
1722        fn codegen_compare_warning_precedes_later_return_error() {
1723            let message = compile_error_message("1 is 1\nreturn\n");
1724            assert!(
1725                message.contains("\"is\" with 'int' literal"),
1726                "expected compare SyntaxWarning first, got {message:?}"
1727            );
1728        }
1729
1730        #[test]
1731        fn codegen_assert_warning_precedes_later_return_error() {
1732            let message = compile_error_message("assert (1,)\nreturn\n");
1733            assert!(
1734                message.contains("assertion is always true"),
1735                "expected assert SyntaxWarning first, got {message:?}"
1736            );
1737        }
1738
1739        #[test]
1740        fn codegen_subscript_warning_precedes_later_return_error() {
1741            let message = compile_error_message("(1)[None]\nreturn\n");
1742            assert!(
1743                message.contains("'int' object is not subscriptable"),
1744                "expected subscript SyntaxWarning first, got {message:?}"
1745            );
1746        }
1747
1748        #[test]
1749        fn codegen_index_warning_precedes_later_return_error() {
1750            let message = compile_error_message("'x'[None]\nreturn\n");
1751            assert!(
1752                message.contains("str indices must be integers or slices, not NoneType"),
1753                "expected index SyntaxWarning first, got {message:?}"
1754            );
1755        }
1756
1757        #[test]
1758        fn string_escape_warning_precedes_later_return_error() {
1759            let message = compile_error_message("\"\\z\"\nreturn\n");
1760            assert!(
1761                message.contains("\"\\z\" is an invalid escape sequence"),
1762                "expected invalid escape SyntaxWarning first, got {message:?}"
1763            );
1764        }
1765
1766        #[test]
1767        fn string_escape_warning_precedes_later_symboltable_error() {
1768            let message = compile_error_message("\"\\z\"\ndef f():\n    from x import *\n");
1769            assert!(
1770                message.contains("\"\\z\" is an invalid escape sequence"),
1771                "expected invalid escape SyntaxWarning first, got {message:?}"
1772            );
1773        }
1774
1775        #[test]
1776        fn string_escape_warning_precedes_later_parse_error() {
1777            let message = compile_error_message("'\\e' $\n");
1778            assert!(
1779                message.contains("\"\\e\" is an invalid escape sequence"),
1780                "expected invalid escape before parse error, got {message:?}"
1781            );
1782        }
1783
1784        #[test]
1785        fn invalid_octal_escape_warning_escalates() {
1786            let message = compile_error_message("'''\n\\407'''\n");
1787            assert!(
1788                message.contains("\"\\407\" is an invalid octal escape sequence"),
1789                "expected invalid octal escape, got {message:?}"
1790            );
1791        }
1792
1793        #[test]
1794        fn unparsed_identifier_suffix_is_not_a_raw_prefix() {
1795            let message = compile_error_message("bar'\\z' $\n");
1796            assert!(
1797                message.contains("\"\\z\" is an invalid escape sequence"),
1798                "trailing r in an identifier must not suppress the warning, got {message:?}"
1799            );
1800        }
1801
1802        #[test]
1803        fn unparsed_raw_prefix_after_dot_stays_raw() {
1804            let message = compile_error_message("obj.r'\\z'\n");
1805            assert!(
1806                !message.contains("invalid escape"),
1807                "r after '.' is a raw prefix, got {message:?}"
1808            );
1809        }
1810
1811        #[test]
1812        fn unparsed_incompatible_prefix_skips_escape_scan() {
1813            let message = compile_error_message("ur'\\z'\n");
1814            assert!(
1815                !message.contains("invalid escape"),
1816                "incompatible prefixes should not emit an escape warning, got {message:?}"
1817            );
1818        }
1819
1820        #[test]
1821        fn unparsed_raw_interpolation_does_not_warn() {
1822            let message = compile_error_message("f\"{r'\\z'}\" $\n");
1823            assert!(
1824                !message.contains("invalid escape"),
1825                "raw interpolation must not be scanned as f-string text, got {message:?}"
1826            );
1827        }
1828
1829        #[test]
1830        fn unparsed_fstring_literal_parts_still_warn() {
1831            let message = compile_error_message("f\"pre\\z{r'\\e'}post\\q\" $\n");
1832            assert!(
1833                message.contains("\"\\z\" is an invalid escape sequence"),
1834                "f-string literal parts should still warn, got {message:?}"
1835            );
1836        }
1837
1838        #[test]
1839        fn unparsed_nested_fstring_in_interpolation_warns() {
1840            let message = compile_error_message("f\"{f'\\z'}\" $\n");
1841            assert!(
1842                message.contains("\"\\z\" is an invalid escape sequence"),
1843                "non-raw nested f-string should warn, got {message:?}"
1844            );
1845        }
1846
1847        #[test]
1848        fn unparsed_format_spec_escape_warns() {
1849            let message = compile_error_message("f\"{x:\\z}\" $\n");
1850            assert!(
1851                message.contains("\"\\z\" is an invalid escape sequence"),
1852                "format spec is f-string text, got {message:?}"
1853            );
1854        }
1855
1856        #[test]
1857        fn ast_preprocess_finally_warning_precedes_later_return_error() {
1858            let message = compile_error_message("try:\n    pass\nfinally:\n    return\nreturn\n");
1859            assert!(
1860                message.contains("'return' in a 'finally' block"),
1861                "expected finally SyntaxWarning first, got {message:?}"
1862            );
1863        }
1864
1865        #[test]
1866        fn ast_preprocess_finally_warning_precedes_symboltable_error() {
1867            let message = compile_error_message(
1868                "def f():\n    from x import *\ntry:\n    pass\nfinally:\n    return\n",
1869            );
1870            assert!(
1871                message.contains("'return' in a 'finally' block"),
1872                "expected finally SyntaxWarning first, got {message:?}"
1873            );
1874        }
1875
1876        #[test]
1877        fn compiler_warning_visits_function_decorators_before_defaults_and_body() {
1878            let message = first_compiler_warning(
1879                r#"
1880@(b"decorator")()
1881def f(x=(1)()):
1882    assert (1,)
1883"#,
1884            );
1885            assert!(
1886                message.contains("'bytes' object is not callable"),
1887                "expected decorator warning first, got {message:?}"
1888            );
1889        }
1890
1891        #[test]
1892        fn compiler_warning_visits_function_defaults_before_annotations() {
1893            let message = first_compiler_warning(
1894                r#"
1895def f(x: (1)() = ("default")()):
1896    pass
1897"#,
1898            );
1899            assert!(
1900                message.contains("'str' object is not callable"),
1901                "expected default warning before annotation warning, got {message:?}"
1902            );
1903        }
1904
1905        #[test]
1906        fn compiler_warning_visits_class_decorators_before_body_and_bases() {
1907            let message = first_compiler_warning(
1908                r#"
1909@(b"decorator")()
1910class C((1)()):
1911    assert (1,)
1912"#,
1913            );
1914            assert!(
1915                message.contains("'bytes' object is not callable"),
1916                "expected class decorator warning first, got {message:?}"
1917            );
1918        }
1919
1920        #[test]
1921        fn compiler_warning_visits_class_body_before_bases() {
1922            let message = first_compiler_warning(
1923                r#"
1924class C((1)()):
1925    assert (1,)
1926"#,
1927            );
1928            assert!(
1929                message.contains("assertion is always true"),
1930                "expected class body warning before base warning, got {message:?}"
1931            );
1932        }
1933
1934        #[test]
1935        fn compiler_warning_visits_type_alias_type_params_before_value() {
1936            let message = first_compiler_warning(
1937                r#"
1938type Alias[T: (1)()] = ("value")()
1939"#,
1940            );
1941            assert!(
1942                message.contains("'int' object is not callable"),
1943                "expected type parameter warning before alias value warning, got {message:?}"
1944            );
1945        }
1946    }
1947}