1use core::fmt;
6
7use crate::{
8 AsObject, PyObjectRef, PyRef, PyResult, VirtualMachine,
9 builtins::{PyBaseExceptionRef, PyCode},
10 compiler::{self, CompileError, CompileOpts},
11 vm::compile_mode::{CompileStart, CompilerFlags, compile_future_features_from_flags},
12};
13
14#[derive(Debug)]
15pub enum VmCompileError {
16 Compile(CompileError),
17 Warning(CompileWarningError),
18}
19
20#[derive(Debug)]
21pub struct CompileWarningError {
22 exception: PyBaseExceptionRef,
23 filename: String,
24 lineno: usize,
25 offset: usize,
26 replacement: Option<SyntaxErrorReplacement>,
27}
28
29#[derive(Debug)]
30struct SyntaxErrorReplacement {
31 message: String,
32 end_offset: usize,
33}
34
35impl From<CompileError> for VmCompileError {
36 fn from(err: CompileError) -> Self {
37 Self::Compile(err)
38 }
39}
40
41impl fmt::Display for VmCompileError {
42 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
43 match self {
44 Self::Compile(err) => err.fmt(f),
45 Self::Warning(_) => f.write_str("compiler warning raised as an exception"),
46 }
47 }
48}
49
50impl VmCompileError {
51 pub fn into_pyexception(self, vm: &VirtualMachine, source: Option<&str>) -> PyBaseExceptionRef {
52 self.into_pyexception_maybe_incomplete(vm, source, false)
53 }
54
55 pub fn into_pyexception_maybe_incomplete(
56 self,
57 vm: &VirtualMachine,
58 source: Option<&str>,
59 allow_incomplete: bool,
60 ) -> PyBaseExceptionRef {
61 match self {
62 Self::Compile(err) => {
63 vm.new_syntax_error_maybe_incomplete(&err, source, allow_incomplete)
64 }
65 Self::Warning(err) => err.into_pyexception(vm, source),
66 }
67 }
68}
69
70impl CompileWarningError {
71 fn into_pyexception(self, vm: &VirtualMachine, source: Option<&str>) -> PyBaseExceptionRef {
72 if !self
73 .exception
74 .fast_isinstance(vm.ctx.exceptions.syntax_warning)
75 {
76 return self.exception;
77 }
78 let (message, end_offset) = if let Some(replacement) = self.replacement {
79 (replacement.message, Some(replacement.end_offset))
80 } else {
81 let Ok(message) = self.exception.as_object().str(vm) else {
82 return self.exception;
83 };
84 (message.to_string_lossy().into_owned(), None)
85 };
86 let syntax_error =
87 vm.new_exception_msg(vm.ctx.exceptions.syntax_error.to_owned(), message.into());
88 syntax_error
89 .as_object()
90 .set_attr("lineno", vm.ctx.new_int(self.lineno), vm)
91 .unwrap();
92 syntax_error
93 .as_object()
94 .set_attr("offset", vm.ctx.new_int(self.offset), vm)
95 .unwrap();
96 if let Some(end_offset) = end_offset {
97 syntax_error
98 .as_object()
99 .set_attr("end_lineno", vm.ctx.new_int(self.lineno), vm)
100 .unwrap();
101 syntax_error
102 .as_object()
103 .set_attr("end_offset", vm.ctx.new_int(end_offset), vm)
104 .unwrap();
105 }
106 syntax_error
107 .as_object()
108 .set_attr("filename", vm.ctx.new_str(self.filename), vm)
109 .unwrap();
110 let text = source
111 .and_then(|source| source.split('\n').nth(self.lineno.saturating_sub(1)))
112 .map_or_else(
113 || vm.ctx.none(),
114 |line| {
115 vm.ctx
116 .new_str(format!("{}\n", line.trim_end_matches('\r')))
117 .into()
118 },
119 );
120 syntax_error.as_object().set_attr("text", text, vm).unwrap();
121 syntax_error
122 }
123}
124
125impl VirtualMachine {
126 #[cfg(feature = "parser")]
127 fn detect_source_encoding(source: &[u8]) -> Option<String> {
128 fn find_encoding_in_line(line: &[u8]) -> Option<String> {
129 let hash_pos = line.iter().position(|&b| b == b'#')?;
130 if !line[..hash_pos]
131 .iter()
132 .all(|&b| b == b' ' || b == b'\t' || b == b'\x0c' || b == b'\r')
133 {
134 return None;
135 }
136 let after_hash = &line[hash_pos..];
137 let coding_pos = after_hash.windows(6).position(|w| w == b"coding")?;
138 let after_coding = &after_hash[coding_pos + 6..];
139 let rest = if after_coding.first() == Some(&b':') || after_coding.first() == Some(&b'=')
140 {
141 &after_coding[1..]
142 } else {
143 return None;
144 };
145 let name: String = rest
146 .iter()
147 .copied()
148 .skip_while(|&b| b == b' ' || b == b'\t')
149 .take_while(|&b| b.is_ascii_alphanumeric() || b == b'-' || b == b'_' || b == b'.')
150 .map(|b| b as char)
151 .collect();
152 (!name.is_empty()).then(|| VirtualMachine::normalize_source_encoding(&name))
153 }
154
155 let mut lines = source.splitn(3, |&b| b == b'\n');
156 if let Some(first) = lines.next() {
157 let first = first.strip_prefix(b"\xef\xbb\xbf").unwrap_or(first);
158 if let Some(enc) = find_encoding_in_line(first) {
159 return Some(enc);
160 }
161 let trimmed = first
162 .iter()
163 .skip_while(|&&b| b == b' ' || b == b'\t' || b == b'\x0c' || b == b'\r')
164 .copied()
165 .collect::<Vec<_>>();
166 if !trimmed.is_empty() && trimmed[0] != b'#' {
167 return None;
168 }
169 }
170 lines.next().and_then(find_encoding_in_line)
171 }
172
173 #[cfg(feature = "parser")]
174 fn normalize_source_encoding(name: &str) -> String {
175 let mut normalized = String::with_capacity(name.len().min(12));
176 for ch in name.chars().take(12) {
177 if ch == '_' {
178 normalized.push('-');
179 } else {
180 normalized.push(ch.to_ascii_lowercase());
181 }
182 }
183
184 if normalized == "utf-8" || normalized.starts_with("utf-8-") {
185 "utf-8".to_owned()
186 } else if normalized == "latin-1"
187 || normalized == "iso-8859-1"
188 || normalized == "iso-latin-1"
189 || normalized.starts_with("latin-1-")
190 || normalized.starts_with("iso-8859-1-")
191 || normalized.starts_with("iso-latin-1-")
192 {
193 "iso-8859-1".to_owned()
194 } else {
195 name.to_owned()
196 }
197 }
198
199 #[cfg(feature = "parser")]
200 fn is_utf8_encoding(name: &str) -> bool {
201 name == "utf-8"
202 }
203
204 #[cfg(any(feature = "parser", feature = "compiler"))]
208 pub(crate) fn program_text(&self, filename: &str, lineno: usize) -> Option<String> {
209 if lineno == 0 {
210 return None;
211 }
212 #[cfg(feature = "host_env")]
213 {
214 let buf = crate::host_env::fs::read(filename).ok()?;
215 let decoded = {
216 #[cfg(feature = "parser")]
217 {
218 let encoding = Self::detect_source_encoding(&buf);
219 if encoding.as_deref().is_none_or(Self::is_utf8_encoding) {
220 String::from_utf8_lossy(&buf).into_owned()
221 } else if encoding.as_deref() == Some("iso-8859-1") {
222 buf.iter().copied().map(char::from).collect()
223 } else {
224 let name = encoding.as_deref()?;
225 let bytes = self.ctx.new_bytes(buf);
226 self.state
227 .codec_registry
228 .decode_text(bytes.into(), name, None, self)
229 .ok()?
230 .to_string_lossy()
231 .into_owned()
232 }
233 }
234 #[cfg(not(feature = "parser"))]
235 {
236 String::from_utf8_lossy(&buf).into_owned()
237 }
238 };
239 let mut remaining = decoded.as_str();
240 if remaining.starts_with('\u{feff}') {
241 remaining = &remaining['\u{feff}'.len_utf8()..];
242 }
243 remaining
244 .split_inclusive('\n')
245 .nth(lineno.checked_sub(1)?)
246 .map(str::to_owned)
247 }
248 #[cfg(not(feature = "host_env"))]
249 {
250 let _ = filename;
251 None
252 }
253 }
254
255 #[cfg(feature = "parser")]
256 fn new_non_utf8_syntax_error(
257 &self,
258 filename: &str,
259 src: &[u8],
260 error_at: usize,
261 ) -> PyBaseExceptionRef {
262 let bad_byte = src[error_at];
263 let line_start = src[..error_at]
264 .iter()
265 .rposition(|&b| b == b'\n')
266 .map_or(0, |i| i + 1);
267 let lineno = src[..error_at].iter().filter(|&&b| b == b'\n').count() + 1;
268 let offset =
269 core::str::from_utf8(&src[line_start..error_at]).map_or(1, |s| s.chars().count() + 1);
270 let line_end = src[line_start..]
271 .iter()
272 .position(|&b| b == b'\n')
273 .map_or(src.len(), |i| line_start + i);
274 let text = String::from_utf8_lossy(&src[line_start..line_end]).into_owned();
275 let in_file = if filename.starts_with('<') {
276 String::new()
277 } else {
278 format!(" in file {filename}")
279 };
280 let msg = format!(
281 "Non-UTF-8 code starting with '\\x{bad_byte:02x}'{in_file} \
282 on line {lineno}, but no encoding declared; \
283 see https://peps.python.org/pep-0263/ for details"
284 );
285 let location = self.ctx.new_tuple(vec![
286 self.ctx.new_str(filename).into(),
287 self.ctx.new_int(lineno).into(),
288 self.ctx.new_int(offset).into(),
289 self.ctx.new_str(text).into(),
290 self.ctx.new_int(lineno).into(),
291 self.ctx.new_int(offset).into(),
292 ]);
293 self.invoke_exception(
294 self.ctx.exceptions.syntax_error,
295 vec![self.ctx.new_str(msg.as_str()).into(), location.into()],
296 )
297 .unwrap_or_else(|_| {
298 self.new_exception_msg(self.ctx.exceptions.syntax_error.to_owned(), msg.into())
299 })
300 }
301
302 #[cfg(feature = "parser")]
303 pub(crate) fn decode_source_bytes(
304 &self,
305 source: &[u8],
306 filename: &str,
307 ignore_cookie: bool,
308 ) -> PyResult<String> {
309 let has_bom = !ignore_cookie && source.starts_with(b"\xef\xbb\xbf");
313 let encoding = if ignore_cookie {
314 None
315 } else {
316 Self::detect_source_encoding(source)
317 };
318 let is_utf8 = encoding.as_deref().is_none_or(Self::is_utf8_encoding);
319 if has_bom && !is_utf8 {
320 let enc = encoding.as_deref().unwrap_or("utf-8");
321 let after_bom = &source[3..];
322 let line_end = after_bom
323 .iter()
324 .position(|&b| b == b'\n')
325 .unwrap_or(after_bom.len());
326 let text = String::from_utf8_lossy(&after_bom[..line_end]).into_owned();
327 let end_offset = text.chars().count() as i32;
328 let msg = format!("encoding problem: {enc} with BOM");
329 let location = self.ctx.new_tuple(vec![
330 self.ctx.new_str(filename).into(),
331 self.ctx.new_int(1).into(),
332 self.ctx.new_int(0).into(),
333 self.ctx.new_str(text).into(),
334 self.ctx.new_int(1).into(),
335 self.ctx.new_int(end_offset).into(),
336 ]);
337 return Err(self
338 .invoke_exception(
339 self.ctx.exceptions.syntax_error,
340 vec![self.ctx.new_str(msg.as_str()).into(), location.into()],
341 )
342 .unwrap_or_else(|_| {
343 self.new_exception_msg(self.ctx.exceptions.syntax_error.to_owned(), msg.into())
344 }));
345 }
346
347 if is_utf8 {
348 let src = if has_bom { &source[3..] } else { source };
349 match core::str::from_utf8(src) {
350 Ok(s) => Ok(s.to_owned()),
351 Err(e) => Err(self.new_non_utf8_syntax_error(filename, src, e.valid_up_to())),
352 }
353 } else {
354 let encoding = encoding.as_deref().unwrap();
355 let bytes = self.ctx.new_bytes(source.to_vec());
356 let decoded = self
357 .state
358 .codec_registry
359 .decode_text(bytes.into(), encoding, None, self)
360 .map_err(|exc| {
361 if exc.fast_isinstance(self.ctx.exceptions.lookup_error) {
362 self.new_exception_msg(
363 self.ctx.exceptions.syntax_error.to_owned(),
364 format!("unknown encoding for '{filename}': {encoding}").into(),
365 )
366 } else {
367 exc
368 }
369 })?;
370 Ok(decoded.to_string_lossy().into_owned())
371 }
372 }
373
374 #[cfg(feature = "parser")]
375 pub fn compile_string_object_with_flags(
376 &self,
377 source: &[u8],
378 filename: &str,
379 start: i32,
380 flags: i32,
381 feature_version: i32,
382 optimize: i32,
383 ) -> PyResult<PyObjectRef> {
384 use crate::convert::ToPyException;
385 use crate::stdlib::_ast;
386
387 let cf = CompilerFlags::from_bits_retain(flags);
388 let Some(start) = CompileStart::from_i32(start) else {
389 return Err(self.new_system_error("Invalid start argument passed to Py_CompileString"));
390 };
391 let source =
392 self.decode_source_bytes(source, filename, cf.contains(CompilerFlags::IGNORE_COOKIE))?;
393 let source = source.as_str();
394 let optimize = match optimize {
395 -1 => self.state.config.settings.optimize.min(2),
396 0..=2 => optimize as u8,
397 _ => return Err(self.new_value_error("compile(): invalid optimize value")),
398 };
399 let allow_incomplete = cf.contains(CompilerFlags::ALLOW_INCOMPLETE_INPUT);
400 let type_comments = cf.contains(CompilerFlags::TYPE_COMMENTS);
401 let dont_imply_dedent = cf.contains(CompilerFlags::DONT_IMPLY_DEDENT);
402 let is_ast_only = cf.contains(CompilerFlags::ONLY_AST);
403 let optimized_ast = cf.contains(CompilerFlags::OPTIMIZED_AST);
404 let future_features = compile_future_features_from_flags(flags);
405 let target_version = if is_ast_only {
406 Some(ruff_python_ast::PythonVersion {
407 major: 3,
408 minor: u8::try_from(feature_version).unwrap_or(crate::version::MINOR as u8),
409 })
410 } else {
411 None
412 };
413
414 if is_ast_only {
415 if start == CompileStart::FuncType {
416 return _ast::parse_func_type(self, source, filename, optimize, target_version)
417 .map_err(|e| (e, Some(source), allow_incomplete).to_pyexception(self));
418 }
419 let (parser_mode, interactive) = match start {
420 CompileStart::Single => (ruff_python_parser::Mode::Module, true),
421 CompileStart::File => (ruff_python_parser::Mode::Module, false),
422 CompileStart::Eval => (ruff_python_parser::Mode::Expression, false),
423 CompileStart::FuncType => unreachable!(),
424 };
425 let parsed = _ast::parse(
426 self,
427 source,
428 filename,
429 parser_mode,
430 optimize,
431 target_version,
432 type_comments,
433 optimized_ast,
434 interactive,
435 future_features,
436 dont_imply_dedent,
437 )
438 .map_err(|e| (e, Some(source), allow_incomplete).to_pyexception(self))?;
439 if start == CompileStart::Single {
440 return _ast::wrap_interactive(self, &parsed);
441 }
442 return Ok(parsed);
443 }
444
445 if type_comments {
446 let parser_mode = match start {
447 CompileStart::Single | CompileStart::File => ruff_python_parser::Mode::Module,
448 CompileStart::Eval => ruff_python_parser::Mode::Expression,
449 CompileStart::FuncType => ruff_python_parser::Mode::Module,
450 };
451 _ast::parse(
452 self,
453 source,
454 filename,
455 parser_mode,
456 optimize,
457 None,
458 type_comments,
459 false,
460 start == CompileStart::Single,
461 future_features,
462 dont_imply_dedent,
463 )
464 .map_err(|e| (e, Some(source), allow_incomplete).to_pyexception(self))?;
465 }
466
467 let mode = match start {
468 CompileStart::Single => compiler::Mode::Single,
469 CompileStart::File => compiler::Mode::Exec,
470 CompileStart::Eval => compiler::Mode::Eval,
471 CompileStart::FuncType => compiler::Mode::BlockExpr,
472 };
473 let mut opts = self.compile_opts();
474 opts.optimize = optimize;
475 opts.allow_top_level_await = cf.contains(CompilerFlags::ALLOW_TOP_LEVEL_AWAIT);
476 opts.future_features = future_features;
477 opts.dont_imply_dedent = dont_imply_dedent;
478 let code = self
479 .compile_with_opts(source, mode, filename, opts)
480 .map_err(|err| {
481 err.into_pyexception_maybe_incomplete(self, Some(source), allow_incomplete)
482 })?;
483 Ok(code.into())
484 }
485
486 pub fn compile(
487 &self,
488 source: &str,
489 mode: compiler::Mode,
490 source_path: impl Into<String>,
491 ) -> Result<PyRef<PyCode>, VmCompileError> {
492 self.compile_with_opts(source, mode, source_path, self.compile_opts())
493 }
494
495 pub fn compile_with_opts(
496 &self,
497 source: &str,
498 mode: compiler::Mode,
499 source_path: impl Into<String>,
500 opts: CompileOpts,
501 ) -> Result<PyRef<PyCode>, VmCompileError> {
502 let source_path = source_path.into();
503 #[cfg(feature = "parser")]
504 {
505 self.emit_tokenizer_syntax_warnings(source, &source_path)
506 .map_err(VmCompileError::Warning)?;
507 self.emit_string_escape_warnings(source, &source_path)
508 .map_err(VmCompileError::Warning)?;
509 }
510 #[cfg(feature = "parser")]
511 let code = {
512 let escalated: core::cell::Cell<Option<CompileWarningError>> =
515 core::cell::Cell::new(None);
516 let mut syntax_warning_handler = |location, message| {
517 escape_warnings::warn_syntax_at_location(&source_path, location, message, self)
518 .map_err(|warning| {
519 escalated.set(Some(warning));
520 compiler::codegen::error::CodegenError {
522 location: Some(location),
523 end_location: None,
524 error: compiler::codegen::error::CodegenErrorType::SyntaxError(
525 String::new(),
526 ),
527 source_path: source_path.clone(),
528 }
529 })
530 };
531 let result = compiler::compile_with_syntax_warning_handler(
532 source,
533 mode,
534 &source_path,
535 opts,
536 &mut syntax_warning_handler,
537 );
538 match escalated.take() {
539 Some(warning) => return Err(VmCompileError::Warning(warning)),
540 None => result,
541 }
542 };
543 #[cfg(not(feature = "parser"))]
544 let code = compiler::compile(source, mode, &source_path, opts);
545 let code = code
546 .map(|code| PyCode::new_ref_from_bytecode(self, code))
547 .map_err(VmCompileError::Compile)?;
548 Ok(code)
549 }
550}
551
552#[cfg(feature = "parser")]
559mod escape_warnings {
560 use super::*;
561 use crate::warn;
562 use ruff_python_ast::{self as ast, visitor::Visitor};
563 use ruff_text_size::TextRange;
564
565 fn line_number_at(source: &str, offset: usize) -> usize {
567 source[..offset.min(source.len())]
568 .bytes()
569 .filter(|&b| b == b'\n')
570 .count()
571 + 1
572 }
573
574 fn line_offset_at(source: &str, offset: usize) -> (usize, usize) {
575 let offset = offset.min(source.len());
576 let prefix = &source[..offset];
577 let lineno = prefix.bytes().filter(|&b| b == b'\n').count() + 1;
578 let line_start = prefix.rfind('\n').map_or(0, |index| index + 1);
579 let column = source[line_start..offset].chars().count() + 1;
580 (lineno, column)
581 }
582
583 fn compile_warning_error(
584 exception: PyBaseExceptionRef,
585 source: &str,
586 filename: &str,
587 offset: usize,
588 ) -> CompileWarningError {
589 let (lineno, offset) = line_offset_at(source, offset);
590 CompileWarningError {
591 exception,
592 filename: filename.to_owned(),
593 lineno,
594 offset,
595 replacement: None,
596 }
597 }
598
599 fn content_bounds(source: &str, range: TextRange) -> Option<(usize, usize)> {
602 let s = range.start().to_usize();
603 let e = range.end().to_usize();
604 if s >= e || e > source.len() {
605 return None;
606 }
607 let bytes = &source.as_bytes()[s..e];
608 let qi = bytes.iter().position(|&c| c == b'\'' || c == b'"')?;
610 let qc = bytes[qi];
611 let ql = if bytes.get(qi + 1) == Some(&qc) && bytes.get(qi + 2) == Some(&qc) {
612 3
613 } else {
614 1
615 };
616 let cs = s + qi + ql;
617 let ce = e.checked_sub(ql)?;
618 if cs <= ce { Some((cs, ce)) } else { None }
619 }
620
621 enum InvalidEscape {
622 Char { ch: char, offset: usize },
623 Octal { digits: [u8; 3], offset: usize },
624 }
625
626 impl InvalidEscape {
627 fn offset(&self) -> usize {
628 match self {
629 Self::Char { offset, .. } | Self::Octal { offset, .. } => *offset,
630 }
631 }
632
633 fn octal_text(digits: [u8; 3]) -> [char; 3] {
634 [digits[0] as char, digits[1] as char, digits[2] as char]
635 }
636
637 fn warning_message(&self) -> String {
638 match self {
639 Self::Char { ch, .. } => format!(
640 "\"\\{ch}\" is an invalid escape sequence. \
641 Such sequences will not work in the future. \
642 Did you mean \"\\\\{ch}\"? A raw string is also an option."
643 ),
644 Self::Octal { digits, .. } => {
645 let [a, b, c] = Self::octal_text(*digits);
646 format!(
647 "\"\\{a}{b}{c}\" is an invalid octal escape sequence. \
648 Such sequences will not work in the future. \
649 Did you mean \"\\\\{a}{b}{c}\"? A raw string is also an option."
650 )
651 }
652 }
653 }
654
655 fn syntax_error_message(&self) -> String {
656 match self {
657 Self::Char { ch, .. } => format!(
658 "\"\\{ch}\" is an invalid escape sequence. \
659 Did you mean \"\\\\{ch}\"? A raw string is also an option."
660 ),
661 Self::Octal { digits, .. } => {
662 let [a, b, c] = Self::octal_text(*digits);
663 format!(
664 "\"\\{a}{b}{c}\" is an invalid octal escape sequence. \
665 Did you mean \"\\\\{a}{b}{c}\"? A raw string is also an option."
666 )
667 }
668 }
669 }
670 }
671
672 fn first_invalid_escape(
681 source: &str,
682 start: usize,
683 end: usize,
684 is_bytes: bool,
685 ) -> Option<InvalidEscape> {
686 let raw = &source[start..end];
687 let mut chars = raw.char_indices().peekable();
688 while let Some((i, ch)) = chars.next() {
689 if ch != '\\' {
690 continue;
691 }
692 let Some((_, next)) = chars.next() else {
693 break;
694 };
695 let offset = start + i;
696 let valid = match next {
697 '\\' | '\'' | '"' | 'a' | 'b' | 'f' | 'n' | 'r' | 't' | 'v' => true,
698 '\n' => true,
699 '\r' => {
700 if matches!(chars.peek(), Some(&(_, '\n'))) {
701 chars.next();
702 }
703 true
704 }
705 '0'..='7' => {
706 let mut digits = [next as u8, 0, 0];
707 let mut len = 1;
708 for _ in 0..2 {
709 if matches!(chars.peek(), Some(&(_, '0'..='7'))) {
710 let (_, digit) = chars.next().unwrap();
711 digits[len] = digit as u8;
712 len += 1;
713 } else {
714 break;
715 }
716 }
717 if len == 3 {
718 let value = ((digits[0] - b'0') as u16) * 64
719 + ((digits[1] - b'0') as u16) * 8
720 + (digits[2] - b'0') as u16;
721 if value > 0o377 {
722 return Some(InvalidEscape::Octal { digits, offset });
723 }
724 }
725 true
726 }
727 'x' | 'u' | 'U' => {
728 if is_bytes && next != 'x' {
730 false
731 } else {
732 let count = match next {
733 'x' => 2,
734 'u' => 4,
735 'U' => 8,
736 _ => unreachable!(),
737 };
738 for _ in 0..count {
739 if chars.peek().is_some_and(|&(_, c)| c.is_ascii_hexdigit()) {
740 chars.next();
741 } else {
742 break;
743 }
744 }
745 true
746 }
747 }
748 'N' => {
749 if is_bytes {
751 false
752 } else {
753 if matches!(chars.peek(), Some(&(_, '{'))) {
754 chars.next();
755 for (_, c) in chars.by_ref() {
756 if c == '}' {
757 break;
758 }
759 }
760 }
761 true
762 }
763 }
764 _ => false,
765 };
766 if !valid {
767 return Some(InvalidEscape::Char { ch: next, offset });
768 }
769 }
770 None
771 }
772
773 fn warn_invalid_escape_sequence(
777 source: &str,
778 escape: InvalidEscape,
779 filename: &str,
780 vm: &VirtualMachine,
781 ) -> Result<(), CompileWarningError> {
782 let (lineno, column) = line_offset_at(source, escape.offset());
783 let warning = escape.warning_message();
784 let syntax_error = escape.syntax_error_message();
785 let fname = vm.ctx.new_str(filename);
786 warn::warn_explicit(
787 Some(vm.ctx.exceptions.syntax_warning.to_owned()),
788 vm.ctx.new_str(warning).into(),
789 fname,
790 lineno,
791 None,
792 vm.ctx.none(),
793 None,
794 None,
795 vm,
796 )
797 .map_err(|exception| CompileWarningError {
798 exception,
799 filename: filename.to_owned(),
800 lineno,
801 offset: column,
802 replacement: Some(SyntaxErrorReplacement {
803 message: syntax_error,
804 end_offset: column + 2,
805 }),
806 })
807 }
808
809 fn warn_syntax_at_offset(
810 source: &str,
811 filename: &str,
812 offset: usize,
813 message: String,
814 vm: &VirtualMachine,
815 ) -> Result<(), CompileWarningError> {
816 let lineno = line_number_at(source, offset);
817 let fname = vm.ctx.new_str(filename);
818 let message = vm.ctx.new_str(message);
819 warn::warn_explicit(
820 Some(vm.ctx.exceptions.syntax_warning.to_owned()),
821 message.into(),
822 fname,
823 lineno,
824 None,
825 vm.ctx.none(),
826 None,
827 None,
828 vm,
829 )
830 .map_err(|err| compile_warning_error(err, source, filename, offset))
831 }
832
833 pub(super) fn warn_syntax_at_location(
834 filename: &str,
835 location: compiler::core::SourceLocation,
836 message: String,
837 vm: &VirtualMachine,
838 ) -> Result<(), CompileWarningError> {
839 let fname = vm.ctx.new_str(filename);
840 let message = vm.ctx.new_str(message);
841 warn::warn_explicit(
842 Some(vm.ctx.exceptions.syntax_warning.to_owned()),
843 message.into(),
844 fname,
845 location.line.get(),
846 None,
847 vm.ctx.none(),
848 None,
849 None,
850 vm,
851 )
852 .map_err(|exception| CompileWarningError {
853 exception,
854 filename: filename.to_owned(),
855 lineno: location.line.get(),
856 offset: location.character_offset.get(),
857 replacement: None,
858 })
859 }
860
861 fn is_ascii_identifier_char(byte: u8) -> bool {
862 byte == b'_' || byte.is_ascii_alphanumeric()
863 }
864
865 fn numeric_keyword_suffix(rest: &[u8]) -> bool {
866 rest.starts_with(b"and")
867 || rest.starts_with(b"else")
868 || rest.starts_with(b"for")
869 || rest.starts_with(b"if")
870 || rest.starts_with(b"in")
871 || rest.starts_with(b"is")
872 || rest.starts_with(b"or")
873 || rest.starts_with(b"not")
874 }
875
876 fn consume_decimal_digits(bytes: &[u8], mut index: usize) -> usize {
877 while index < bytes.len() {
878 match bytes[index] {
879 b'0'..=b'9' => index += 1,
880 b'_' if bytes
881 .get(index + 1)
882 .is_some_and(|byte| byte.is_ascii_digit()) =>
883 {
884 index += 2;
885 }
886 _ => break,
887 }
888 }
889 index
890 }
891
892 fn consume_radix_digits(
893 bytes: &[u8],
894 mut index: usize,
895 is_digit: impl Fn(u8) -> bool,
896 ) -> usize {
897 while index < bytes.len() {
898 if is_digit(bytes[index]) {
899 index += 1;
900 } else if bytes.get(index) == Some(&b'_')
901 && bytes.get(index + 1).is_some_and(|&byte| is_digit(byte))
902 {
903 index += 2;
904 } else {
905 break;
906 }
907 }
908 index
909 }
910
911 fn number_literal_end(bytes: &[u8], start: usize) -> Option<(&'static str, usize)> {
912 if bytes.get(start) == Some(&b'.') {
913 if !bytes
914 .get(start + 1)
915 .is_some_and(|byte| byte.is_ascii_digit())
916 {
917 return None;
918 }
919 let mut index = consume_decimal_digits(bytes, start + 1);
920 index = consume_exponent(bytes, index);
921 if matches!(bytes.get(index), Some(b'j' | b'J')) {
922 return Some(("imaginary", index + 1));
923 }
924 return Some(("decimal", index));
925 }
926
927 if !bytes.get(start).is_some_and(|byte| byte.is_ascii_digit()) {
928 return None;
929 }
930
931 if bytes.get(start) == Some(&b'0') {
932 match bytes.get(start + 1) {
933 Some(b'x' | b'X') => {
934 let end =
935 consume_radix_digits(bytes, start + 2, |byte| byte.is_ascii_hexdigit());
936 return Some(("hexadecimal", end));
937 }
938 Some(b'o' | b'O') => {
939 let end =
940 consume_radix_digits(bytes, start + 2, |byte| matches!(byte, b'0'..=b'7'));
941 return Some(("octal", end));
942 }
943 Some(b'b' | b'B') => {
944 let end =
945 consume_radix_digits(bytes, start + 2, |byte| matches!(byte, b'0' | b'1'));
946 return Some(("binary", end));
947 }
948 _ => {}
949 }
950 }
951
952 let mut index = consume_decimal_digits(bytes, start);
953 if bytes.get(index) == Some(&b'.') {
954 index = consume_decimal_digits(bytes, index + 1);
955 }
956 index = consume_exponent(bytes, index);
957 if matches!(bytes.get(index), Some(b'j' | b'J')) {
958 return Some(("imaginary", index + 1));
959 }
960 Some(("decimal", index))
961 }
962
963 fn consume_exponent(bytes: &[u8], index: usize) -> usize {
964 if !matches!(bytes.get(index), Some(b'e' | b'E')) {
965 return index;
966 }
967 let mut cursor = index + 1;
968 if matches!(bytes.get(cursor), Some(b'+' | b'-')) {
969 cursor += 1;
970 }
971 if bytes.get(cursor).is_some_and(|byte| byte.is_ascii_digit()) {
972 consume_decimal_digits(bytes, cursor)
973 } else {
974 index
975 }
976 }
977
978 fn skip_quoted_string(bytes: &[u8], mut index: usize) -> usize {
979 let quote = bytes[index];
980 let triple = bytes.get(index + 1) == Some("e) && bytes.get(index + 2) == Some("e);
981 let quote_len = if triple { 3 } else { 1 };
982 index += quote_len;
983 while index < bytes.len() {
984 if bytes[index] == b'\\' {
985 index = (index + 2).min(bytes.len());
986 } else if triple
987 && bytes.get(index) == Some("e)
988 && bytes.get(index + 1) == Some("e)
989 && bytes.get(index + 2) == Some("e)
990 {
991 return index + 3;
992 } else if !triple && bytes[index] == quote {
993 return index + 1;
994 } else {
995 index += 1;
996 }
997 }
998 index
999 }
1000
1001 #[derive(Clone, Copy)]
1002 enum StringPrefix {
1003 None,
1004 Invalid,
1005 Valid {
1006 is_raw: bool,
1007 is_bytes: bool,
1008 is_interpolated: bool,
1009 },
1010 }
1011
1012 fn is_string_prefix_letter(byte: u8) -> bool {
1013 matches!(
1014 byte,
1015 b'r' | b'R' | b'b' | b'B' | b'u' | b'U' | b'f' | b'F' | b't' | b'T'
1016 )
1017 }
1018
1019 fn prefix_continues_identifier(byte: u8) -> bool {
1020 byte == b'_' || byte.is_ascii_alphabetic() || byte >= 0x80
1021 }
1022
1023 fn two_char_prefix(first: u8, second: u8) -> Option<(bool, bool, bool)> {
1024 match [first.to_ascii_lowercase(), second.to_ascii_lowercase()] {
1025 [b'r', b'f' | b't'] | [b'f' | b't', b'r'] => Some((true, false, true)),
1026 [b'r', b'b'] | [b'b', b'r'] => Some((true, true, false)),
1027 _ => None,
1028 }
1029 }
1030
1031 fn string_prefix(bytes: &[u8], quote_index: usize) -> StringPrefix {
1033 let mut taken = [0u8; 2];
1034 let mut count = 0;
1035 let mut index = quote_index;
1036 while index > 0 && count < 2 {
1037 let byte = bytes[index - 1];
1038 if !is_string_prefix_letter(byte) {
1039 break;
1040 }
1041 taken[count] = byte;
1042 count += 1;
1043 index -= 1;
1044 }
1045 if count == 0 {
1046 return StringPrefix::None;
1047 }
1048 if index > 0 && prefix_continues_identifier(bytes[index - 1]) {
1049 return StringPrefix::None;
1050 }
1051 if count == 2 {
1052 return match two_char_prefix(taken[1], taken[0]) {
1053 Some((is_raw, is_bytes, is_interpolated)) => StringPrefix::Valid {
1054 is_raw,
1055 is_bytes,
1056 is_interpolated,
1057 },
1058 None => StringPrefix::Invalid,
1059 };
1060 }
1061 let (is_raw, is_bytes, is_interpolated) = match taken[0] {
1062 b'r' | b'R' => (true, false, false),
1063 b'b' | b'B' => (false, true, false),
1064 b'f' | b'F' | b't' | b'T' => (false, false, true),
1065 _ => (false, false, false),
1066 };
1067 StringPrefix::Valid {
1068 is_raw,
1069 is_bytes,
1070 is_interpolated,
1071 }
1072 }
1073
1074 fn text_range_from_bounds(start: usize, end: usize) -> TextRange {
1075 TextRange::new(
1076 ruff_text_size::TextSize::try_from(start).unwrap_or_default(),
1077 ruff_text_size::TextSize::try_from(end).unwrap_or_default(),
1078 )
1079 }
1080
1081 fn warn_quoted_literal(
1082 source: &str,
1083 quote_index: usize,
1084 end: usize,
1085 is_bytes: bool,
1086 filename: &str,
1087 vm: &VirtualMachine,
1088 ) -> Result<(), CompileWarningError> {
1089 if let Some((start, content_end)) =
1090 content_bounds(source, text_range_from_bounds(quote_index, end))
1091 && let Some(escape) = first_invalid_escape(source, start, content_end, is_bytes)
1092 {
1093 warn_invalid_escape_sequence(source, escape, filename, vm)?;
1094 }
1095 Ok(())
1096 }
1097
1098 fn warn_fstring_literal_part(
1099 source: &str,
1100 start: usize,
1101 end: usize,
1102 filename: &str,
1103 vm: &VirtualMachine,
1104 ) -> Result<(), CompileWarningError> {
1105 if start >= end || end > source.len() {
1106 return Ok(());
1107 }
1108 if let Some(escape) = first_invalid_escape(source, start, end, false) {
1109 return warn_invalid_escape_sequence(source, escape, filename, vm);
1110 }
1111 let trailing_bs = source.as_bytes()[start..end]
1112 .iter()
1113 .rev()
1114 .take_while(|&&byte| byte == b'\\')
1115 .count();
1116 if trailing_bs % 2 == 1
1117 && let Some(&after) = source.as_bytes().get(end)
1118 && (after == b'{' || after == b'}')
1119 {
1120 warn_invalid_escape_sequence(
1121 source,
1122 InvalidEscape::Char {
1123 ch: after as char,
1124 offset: end - 1,
1125 },
1126 filename,
1127 vm,
1128 )?;
1129 }
1130 Ok(())
1131 }
1132
1133 fn find_interpolation_end(bytes: &[u8], open: usize, limit: usize) -> usize {
1134 let mut index = open + 1;
1135 let mut depth: u32 = 1;
1136 let mut paren: u32 = 0;
1137 let mut bracket: u32 = 0;
1138 while index < limit {
1139 match bytes[index] {
1140 b'#' if paren == 0 && bracket == 0 && depth == 1 => {
1141 while index < limit && bytes[index] != b'\n' {
1142 index += 1;
1143 }
1144 }
1145 b'\'' | b'"' => {
1146 index = skip_quoted_string(bytes, index).max(index + 1);
1147 }
1148 b'(' => {
1149 paren += 1;
1150 index += 1;
1151 }
1152 b')' => {
1153 paren = paren.saturating_sub(1);
1154 index += 1;
1155 }
1156 b'[' => {
1157 bracket += 1;
1158 index += 1;
1159 }
1160 b']' => {
1161 bracket = bracket.saturating_sub(1);
1162 index += 1;
1163 }
1164 b'{' => {
1165 depth += 1;
1166 index += 1;
1167 }
1168 b'}' => {
1169 depth -= 1;
1170 index += 1;
1171 if depth == 0 {
1172 return index;
1173 }
1174 }
1175 b'\\' => index = (index + 2).min(limit),
1176 _ => index += 1,
1177 }
1178 }
1179 limit
1180 }
1181
1182 fn scan_interpolated_content(
1183 source: &str,
1184 start: usize,
1185 end: usize,
1186 is_raw: bool,
1187 filename: &str,
1188 vm: &VirtualMachine,
1189 ) -> Result<(), CompileWarningError> {
1190 let bytes = source.as_bytes();
1191 let mut index = start;
1192 let mut literal_start = start;
1193 while index < end {
1194 match bytes[index] {
1195 b'{' if bytes.get(index + 1) == Some(&b'{') => {
1196 index += 2;
1197 }
1198 b'{' => {
1199 if !is_raw {
1200 warn_fstring_literal_part(source, literal_start, index, filename, vm)?;
1201 }
1202 let close = find_interpolation_end(bytes, index, end);
1203 let body_end = if close > index + 1 { close - 1 } else { close };
1204 emit_string_escape_warnings_in_range(
1205 source,
1206 index + 1,
1207 body_end,
1208 Some(is_raw),
1209 filename,
1210 vm,
1211 )?;
1212 index = close.max(index + 1);
1213 literal_start = index;
1214 }
1215 b'}' if bytes.get(index + 1) == Some(&b'}') => {
1216 index += 2;
1217 }
1218 b'\\' => index = (index + 2).min(end),
1219 _ => index += 1,
1220 }
1221 }
1222 if !is_raw {
1223 warn_fstring_literal_part(source, literal_start, end, filename, vm)?;
1224 }
1225 Ok(())
1226 }
1227
1228 fn emit_string_escape_warnings_unparsed(
1231 source: &str,
1232 filename: &str,
1233 vm: &VirtualMachine,
1234 ) -> Result<(), CompileWarningError> {
1235 emit_string_escape_warnings_in_range(source, 0, source.len(), None, filename, vm)
1236 }
1237
1238 fn emit_string_escape_warnings_in_range(
1239 source: &str,
1240 start: usize,
1241 end: usize,
1242 format_spec_raw: Option<bool>,
1243 filename: &str,
1244 vm: &VirtualMachine,
1245 ) -> Result<(), CompileWarningError> {
1246 let bytes = source.as_bytes();
1247 let end = end.min(bytes.len());
1248 let mut index = start;
1249 let mut paren: u32 = 0;
1250 let mut bracket: u32 = 0;
1251 let mut brace: u32 = 0;
1252 while index < end {
1253 match bytes[index] {
1254 b'#' => {
1255 while index < end && bytes[index] != b'\n' {
1256 index += 1;
1257 }
1258 }
1259 b'\'' | b'"' => {
1260 let prefix = string_prefix(bytes, index);
1261 let quote_end = skip_quoted_string(bytes, index).min(bytes.len());
1262 match prefix {
1263 StringPrefix::Valid {
1264 is_raw,
1265 is_bytes,
1266 is_interpolated,
1267 } => {
1268 if is_interpolated {
1269 if let Some((content_start, content_end)) =
1270 content_bounds(source, text_range_from_bounds(index, quote_end))
1271 {
1272 scan_interpolated_content(
1273 source,
1274 content_start,
1275 content_end.min(end),
1276 is_raw,
1277 filename,
1278 vm,
1279 )?;
1280 }
1281 } else if !is_raw {
1282 warn_quoted_literal(
1283 source, index, quote_end, is_bytes, filename, vm,
1284 )?;
1285 }
1286 }
1287 StringPrefix::None => {
1288 warn_quoted_literal(source, index, quote_end, false, filename, vm)?;
1289 }
1290 StringPrefix::Invalid => {}
1291 }
1292 index = quote_end.max(index + 1);
1293 }
1294 b':' if let Some(is_raw) = format_spec_raw
1295 && paren == 0
1296 && bracket == 0
1297 && brace == 0 =>
1298 {
1299 scan_interpolated_content(source, index + 1, end, is_raw, filename, vm)?;
1300 return Ok(());
1301 }
1302 b'(' if format_spec_raw.is_some() => {
1303 paren += 1;
1304 index += 1;
1305 }
1306 b')' if format_spec_raw.is_some() => {
1307 paren = paren.saturating_sub(1);
1308 index += 1;
1309 }
1310 b'[' if format_spec_raw.is_some() => {
1311 bracket += 1;
1312 index += 1;
1313 }
1314 b']' if format_spec_raw.is_some() => {
1315 bracket = bracket.saturating_sub(1);
1316 index += 1;
1317 }
1318 b'{' if format_spec_raw.is_some() => {
1319 brace += 1;
1320 index += 1;
1321 }
1322 b'}' if format_spec_raw.is_some() => {
1323 brace = brace.saturating_sub(1);
1324 index += 1;
1325 }
1326 _ => index += 1,
1327 }
1328 }
1329 Ok(())
1330 }
1331
1332 fn emit_numeric_literal_warnings(
1333 source: &str,
1334 filename: &str,
1335 vm: &VirtualMachine,
1336 ) -> Result<(), CompileWarningError> {
1337 let bytes = source.as_bytes();
1338 let mut index = 0;
1339 while index < bytes.len() {
1340 match bytes[index] {
1341 b'#' => {
1342 while index < bytes.len() && bytes[index] != b'\n' {
1343 index += 1;
1344 }
1345 }
1346 b'\'' | b'"' => {
1347 index = skip_quoted_string(bytes, index);
1348 }
1349 byte if byte >= 0x80 || byte == b'_' || byte.is_ascii_alphabetic() => {
1350 index += 1;
1351 while index < bytes.len()
1352 && (bytes[index] >= 0x80 || is_ascii_identifier_char(bytes[index]))
1353 {
1354 index += 1;
1355 }
1356 }
1357 b'.' | b'0'..=b'9' => {
1358 let Some((kind, end)) = number_literal_end(bytes, index) else {
1359 index += 1;
1360 continue;
1361 };
1362 if end > index && numeric_keyword_suffix(&bytes[end..]) {
1363 warn_syntax_at_offset(
1364 source,
1365 filename,
1366 index,
1367 format!("invalid {kind} literal"),
1368 vm,
1369 )?;
1370 }
1371 index = end.max(index + 1);
1372 }
1373 _ => index += 1,
1374 }
1375 }
1376 Ok(())
1377 }
1378
1379 struct EscapeWarningVisitor<'a> {
1380 source: &'a str,
1381 filename: &'a str,
1382 vm: &'a VirtualMachine,
1383 error: Option<CompileWarningError>,
1384 depth: usize,
1387 depth_limit: usize,
1388 }
1389
1390 impl<'a> EscapeWarningVisitor<'a> {
1391 fn record_warning(&mut self, result: Result<(), CompileWarningError>) {
1392 if self.error.is_none()
1393 && let Err(err) = result
1394 {
1395 self.error = Some(err);
1396 }
1397 }
1398
1399 fn check_quoted_literal(&mut self, range: TextRange, is_bytes: bool) {
1402 if let Some((start, end)) = content_bounds(self.source, range)
1403 && let Some(escape) = first_invalid_escape(self.source, start, end, is_bytes)
1404 {
1405 let result =
1406 warn_invalid_escape_sequence(self.source, escape, self.filename, self.vm);
1407 self.record_warning(result);
1408 }
1409 }
1410
1411 fn check_fstring_literal(&mut self, range: TextRange) {
1418 let start = range.start().to_usize();
1419 let end = range.end().to_usize();
1420 if start >= end || end > self.source.len() {
1421 return;
1422 }
1423 if let Some(escape) = first_invalid_escape(self.source, start, end, false) {
1424 let result =
1425 warn_invalid_escape_sequence(self.source, escape, self.filename, self.vm);
1426 self.record_warning(result);
1427 return;
1428 }
1429 let trailing_bs = self.source.as_bytes()[start..end]
1436 .iter()
1437 .rev()
1438 .take_while(|&&b| b == b'\\')
1439 .count();
1440 if trailing_bs % 2 == 1
1441 && let Some(&after) = self.source.as_bytes().get(end)
1442 && (after == b'{' || after == b'}')
1443 {
1444 let result = warn_invalid_escape_sequence(
1445 self.source,
1446 InvalidEscape::Char {
1447 ch: after as char,
1448 offset: end - 1,
1449 },
1450 self.filename,
1451 self.vm,
1452 );
1453 self.record_warning(result);
1454 }
1455 }
1456
1457 fn visit_fstring_elements(&mut self, elements: &'a ast::InterpolatedStringElements) {
1460 for element in elements {
1461 if self.error.is_some() {
1462 return;
1463 }
1464 match element {
1465 ast::InterpolatedStringElement::Literal(lit) => {
1466 self.check_fstring_literal(lit.range);
1467 }
1468 ast::InterpolatedStringElement::Interpolation(interp) => {
1469 self.visit_expr(&interp.expression);
1470 if let Some(spec) = &interp.format_spec {
1471 self.visit_fstring_elements(&spec.elements);
1472 }
1473 }
1474 }
1475 }
1476 }
1477 }
1478
1479 impl<'a> Visitor<'a> for EscapeWarningVisitor<'a> {
1480 fn visit_expr(&mut self, expr: &'a ast::Expr) {
1481 if self.error.is_some() {
1482 return;
1483 }
1484 match expr {
1485 ast::Expr::StringLiteral(string) => {
1487 for part in string.value.as_slice() {
1488 if !matches!(
1489 part.flags.prefix(),
1490 ast::str_prefix::StringLiteralPrefix::Raw { .. }
1491 ) {
1492 self.check_quoted_literal(part.range, false);
1493 }
1494 }
1495 }
1496 ast::Expr::BytesLiteral(bytes) => {
1498 for part in bytes.value.as_slice() {
1499 if !matches!(
1500 part.flags.prefix(),
1501 ast::str_prefix::ByteStringPrefix::Raw { .. }
1502 ) {
1503 self.check_quoted_literal(part.range, true);
1504 }
1505 }
1506 }
1507 ast::Expr::FString(fstring_expr) => {
1509 for part in fstring_expr.value.as_slice() {
1510 match part {
1511 ast::FStringPart::Literal(string_lit) => {
1512 if !matches!(
1514 string_lit.flags.prefix(),
1515 ast::str_prefix::StringLiteralPrefix::Raw { .. }
1516 ) {
1517 self.check_quoted_literal(string_lit.range, false);
1518 }
1519 }
1520 ast::FStringPart::FString(fstring) => {
1521 if matches!(
1522 fstring.flags.prefix(),
1523 ast::str_prefix::FStringPrefix::Raw { .. }
1524 ) {
1525 continue;
1526 }
1527 self.visit_fstring_elements(&fstring.elements);
1528 }
1529 }
1530 }
1531 }
1532 _ => {
1533 if self.depth < self.depth_limit {
1534 self.depth += 1;
1535 ast::visitor::walk_expr(self, expr);
1536 self.depth -= 1;
1537 }
1538 }
1539 }
1540 }
1541 }
1542
1543 impl VirtualMachine {
1544 pub(super) fn emit_tokenizer_syntax_warnings(
1547 &self,
1548 source: &str,
1549 filename: &str,
1550 ) -> Result<(), CompileWarningError> {
1551 emit_numeric_literal_warnings(source, filename, self)
1552 }
1553
1554 pub(super) fn emit_string_escape_warnings(
1557 &self,
1558 source: &str,
1559 filename: &str,
1560 ) -> Result<(), CompileWarningError> {
1561 let source_file = compiler::core::SourceFileBuilder::new(filename, source).finish();
1564 if compiler::pre_parse_source_error(&source_file).is_err() {
1565 return Ok(());
1566 }
1567 let Ok(parsed) =
1568 ruff_python_parser::parse(source, ruff_python_parser::Mode::Module.into())
1569 else {
1570 return emit_string_escape_warnings_unparsed(source, filename, self);
1571 };
1572 let ast = parsed.into_syntax();
1573 let mut visitor = EscapeWarningVisitor {
1574 source,
1575 filename,
1576 vm: self,
1577 error: None,
1578 depth: 0,
1579 depth_limit: compiler::CompileOpts::default().recursion_limit,
1580 };
1581 match &ast {
1582 ast::Mod::Module(module) => {
1583 for stmt in &module.body {
1584 visitor.visit_stmt(stmt);
1585 }
1586 }
1587 ast::Mod::Expression(expr) => {
1588 visitor.visit_expr(&expr.body);
1589 }
1590 }
1591 visitor.error.map_or(Ok(()), Err)
1592 }
1593 }
1594
1595 #[cfg(test)]
1596 mod tests {
1597 use super::*;
1598 use crate::{Interpreter, builtins::PyTuple};
1599
1600 fn install_syntax_warning_error_filter(vm: &VirtualMachine) {
1601 let error_filter = PyTuple::new_ref(
1602 vec![
1603 vm.ctx.new_str("error").into(),
1604 vm.ctx.none(),
1605 vm.ctx.exceptions.syntax_warning.as_object().to_owned(),
1606 vm.ctx.none(),
1607 vm.ctx.new_int(0).into(),
1608 ],
1609 &vm.ctx,
1610 );
1611 vm.state
1612 .warnings
1613 .filters
1614 .borrow_vec_mut()
1615 .insert(0, error_filter.into());
1616 vm.state.warnings.filters_mutated();
1617 }
1618
1619 fn first_compiler_warning(source: &str) -> String {
1620 Interpreter::without_stdlib(Default::default()).enter(|vm| {
1621 install_syntax_warning_error_filter(vm);
1622 let err = vm
1623 .compile(source, compiler::Mode::Exec, "<test>")
1624 .expect_err("expected compiler SyntaxWarning");
1625 let exception = err.into_pyexception(vm, Some(source));
1626 exception
1627 .as_object()
1628 .str(vm)
1629 .expect("warning message should stringify")
1630 .as_wtf8()
1631 .to_string()
1632 })
1633 }
1634
1635 fn compile_error_message(source: &str) -> String {
1636 Interpreter::without_stdlib(Default::default()).enter(|vm| {
1637 install_syntax_warning_error_filter(vm);
1638 let err = match vm.compile(source, compiler::Mode::Exec, "<test>") {
1639 Ok(_) => panic!("expected compile error"),
1640 Err(err) => err,
1641 };
1642 err.into_pyexception(vm, Some(source))
1643 .as_object()
1644 .str(vm)
1645 .expect("compile error should stringify")
1646 .as_wtf8()
1647 .to_string()
1648 })
1649 }
1650
1651 #[test]
1652 fn ast_only_compile_honors_barry_as_flufl() {
1653 Interpreter::without_stdlib(Default::default()).enter(|vm| {
1654 let flags = CompilerFlags::ONLY_AST.bits();
1655 vm.compile_string_object_with_flags(
1656 b"from __future__ import barry_as_FLUFL\n2 <> 3\n",
1657 "<test>",
1658 CompileStart::File.as_i32(),
1659 flags,
1660 -1,
1661 -1,
1662 )
1663 .expect("PyCF_ONLY_AST should accept <> in Barry mode");
1664
1665 let err = vm
1666 .compile_string_object_with_flags(
1667 b"from __future__ import barry_as_FLUFL\n2 != 3\n",
1668 "<test>",
1669 CompileStart::File.as_i32(),
1670 flags,
1671 -1,
1672 -1,
1673 )
1674 .expect_err("PyCF_ONLY_AST should reject != in Barry mode");
1675 assert!(
1676 err.as_object()
1677 .str(vm)
1678 .unwrap()
1679 .as_wtf8()
1680 .to_string()
1681 .contains("with Barry as BDFL")
1682 );
1683 });
1684 }
1685
1686 #[test]
1687 fn type_comment_preparse_honors_inherited_barry_as_flufl() {
1688 Interpreter::without_stdlib(Default::default()).enter(|vm| {
1689 let flags = CompilerFlags::TYPE_COMMENTS.bits()
1690 | crate::bytecode::CodeFlags::FUTURE_BARRY_AS_BDFL.bits() as i32;
1691 vm.compile_string_object_with_flags(
1692 b"2 <> 3\n",
1693 "<test>",
1694 CompileStart::File.as_i32(),
1695 flags,
1696 -1,
1697 -1,
1698 )
1699 .expect("type-comment preparse should accept <> in inherited Barry mode");
1700 });
1701 }
1702
1703 #[test]
1704 fn codegen_caller_warning_precedes_later_return_error() {
1705 let message = compile_error_message("(1)()\nreturn\n");
1706 assert!(
1707 message.contains("'int' object is not callable"),
1708 "expected caller SyntaxWarning first, got {message:?}"
1709 );
1710 }
1711
1712 #[test]
1713 fn symboltable_error_still_precedes_codegen_caller_warning() {
1714 let message = compile_error_message("(1)()\ndef f():\n from x import *\n");
1715 assert!(
1716 message.contains("import * only allowed at module level"),
1717 "expected symboltable error first, got {message:?}"
1718 );
1719 }
1720
1721 #[test]
1722 fn codegen_compare_warning_precedes_later_return_error() {
1723 let message = compile_error_message("1 is 1\nreturn\n");
1724 assert!(
1725 message.contains("\"is\" with 'int' literal"),
1726 "expected compare SyntaxWarning first, got {message:?}"
1727 );
1728 }
1729
1730 #[test]
1731 fn codegen_assert_warning_precedes_later_return_error() {
1732 let message = compile_error_message("assert (1,)\nreturn\n");
1733 assert!(
1734 message.contains("assertion is always true"),
1735 "expected assert SyntaxWarning first, got {message:?}"
1736 );
1737 }
1738
1739 #[test]
1740 fn codegen_subscript_warning_precedes_later_return_error() {
1741 let message = compile_error_message("(1)[None]\nreturn\n");
1742 assert!(
1743 message.contains("'int' object is not subscriptable"),
1744 "expected subscript SyntaxWarning first, got {message:?}"
1745 );
1746 }
1747
1748 #[test]
1749 fn codegen_index_warning_precedes_later_return_error() {
1750 let message = compile_error_message("'x'[None]\nreturn\n");
1751 assert!(
1752 message.contains("str indices must be integers or slices, not NoneType"),
1753 "expected index SyntaxWarning first, got {message:?}"
1754 );
1755 }
1756
1757 #[test]
1758 fn string_escape_warning_precedes_later_return_error() {
1759 let message = compile_error_message("\"\\z\"\nreturn\n");
1760 assert!(
1761 message.contains("\"\\z\" is an invalid escape sequence"),
1762 "expected invalid escape SyntaxWarning first, got {message:?}"
1763 );
1764 }
1765
1766 #[test]
1767 fn string_escape_warning_precedes_later_symboltable_error() {
1768 let message = compile_error_message("\"\\z\"\ndef f():\n from x import *\n");
1769 assert!(
1770 message.contains("\"\\z\" is an invalid escape sequence"),
1771 "expected invalid escape SyntaxWarning first, got {message:?}"
1772 );
1773 }
1774
1775 #[test]
1776 fn string_escape_warning_precedes_later_parse_error() {
1777 let message = compile_error_message("'\\e' $\n");
1778 assert!(
1779 message.contains("\"\\e\" is an invalid escape sequence"),
1780 "expected invalid escape before parse error, got {message:?}"
1781 );
1782 }
1783
1784 #[test]
1785 fn invalid_octal_escape_warning_escalates() {
1786 let message = compile_error_message("'''\n\\407'''\n");
1787 assert!(
1788 message.contains("\"\\407\" is an invalid octal escape sequence"),
1789 "expected invalid octal escape, got {message:?}"
1790 );
1791 }
1792
1793 #[test]
1794 fn unparsed_identifier_suffix_is_not_a_raw_prefix() {
1795 let message = compile_error_message("bar'\\z' $\n");
1796 assert!(
1797 message.contains("\"\\z\" is an invalid escape sequence"),
1798 "trailing r in an identifier must not suppress the warning, got {message:?}"
1799 );
1800 }
1801
1802 #[test]
1803 fn unparsed_raw_prefix_after_dot_stays_raw() {
1804 let message = compile_error_message("obj.r'\\z'\n");
1805 assert!(
1806 !message.contains("invalid escape"),
1807 "r after '.' is a raw prefix, got {message:?}"
1808 );
1809 }
1810
1811 #[test]
1812 fn unparsed_incompatible_prefix_skips_escape_scan() {
1813 let message = compile_error_message("ur'\\z'\n");
1814 assert!(
1815 !message.contains("invalid escape"),
1816 "incompatible prefixes should not emit an escape warning, got {message:?}"
1817 );
1818 }
1819
1820 #[test]
1821 fn unparsed_raw_interpolation_does_not_warn() {
1822 let message = compile_error_message("f\"{r'\\z'}\" $\n");
1823 assert!(
1824 !message.contains("invalid escape"),
1825 "raw interpolation must not be scanned as f-string text, got {message:?}"
1826 );
1827 }
1828
1829 #[test]
1830 fn unparsed_fstring_literal_parts_still_warn() {
1831 let message = compile_error_message("f\"pre\\z{r'\\e'}post\\q\" $\n");
1832 assert!(
1833 message.contains("\"\\z\" is an invalid escape sequence"),
1834 "f-string literal parts should still warn, got {message:?}"
1835 );
1836 }
1837
1838 #[test]
1839 fn unparsed_nested_fstring_in_interpolation_warns() {
1840 let message = compile_error_message("f\"{f'\\z'}\" $\n");
1841 assert!(
1842 message.contains("\"\\z\" is an invalid escape sequence"),
1843 "non-raw nested f-string should warn, got {message:?}"
1844 );
1845 }
1846
1847 #[test]
1848 fn unparsed_format_spec_escape_warns() {
1849 let message = compile_error_message("f\"{x:\\z}\" $\n");
1850 assert!(
1851 message.contains("\"\\z\" is an invalid escape sequence"),
1852 "format spec is f-string text, got {message:?}"
1853 );
1854 }
1855
1856 #[test]
1857 fn ast_preprocess_finally_warning_precedes_later_return_error() {
1858 let message = compile_error_message("try:\n pass\nfinally:\n return\nreturn\n");
1859 assert!(
1860 message.contains("'return' in a 'finally' block"),
1861 "expected finally SyntaxWarning first, got {message:?}"
1862 );
1863 }
1864
1865 #[test]
1866 fn ast_preprocess_finally_warning_precedes_symboltable_error() {
1867 let message = compile_error_message(
1868 "def f():\n from x import *\ntry:\n pass\nfinally:\n return\n",
1869 );
1870 assert!(
1871 message.contains("'return' in a 'finally' block"),
1872 "expected finally SyntaxWarning first, got {message:?}"
1873 );
1874 }
1875
1876 #[test]
1877 fn compiler_warning_visits_function_decorators_before_defaults_and_body() {
1878 let message = first_compiler_warning(
1879 r#"
1880@(b"decorator")()
1881def f(x=(1)()):
1882 assert (1,)
1883"#,
1884 );
1885 assert!(
1886 message.contains("'bytes' object is not callable"),
1887 "expected decorator warning first, got {message:?}"
1888 );
1889 }
1890
1891 #[test]
1892 fn compiler_warning_visits_function_defaults_before_annotations() {
1893 let message = first_compiler_warning(
1894 r#"
1895def f(x: (1)() = ("default")()):
1896 pass
1897"#,
1898 );
1899 assert!(
1900 message.contains("'str' object is not callable"),
1901 "expected default warning before annotation warning, got {message:?}"
1902 );
1903 }
1904
1905 #[test]
1906 fn compiler_warning_visits_class_decorators_before_body_and_bases() {
1907 let message = first_compiler_warning(
1908 r#"
1909@(b"decorator")()
1910class C((1)()):
1911 assert (1,)
1912"#,
1913 );
1914 assert!(
1915 message.contains("'bytes' object is not callable"),
1916 "expected class decorator warning first, got {message:?}"
1917 );
1918 }
1919
1920 #[test]
1921 fn compiler_warning_visits_class_body_before_bases() {
1922 let message = first_compiler_warning(
1923 r#"
1924class C((1)()):
1925 assert (1,)
1926"#,
1927 );
1928 assert!(
1929 message.contains("assertion is always true"),
1930 "expected class body warning before base warning, got {message:?}"
1931 );
1932 }
1933
1934 #[test]
1935 fn compiler_warning_visits_type_alias_type_params_before_value() {
1936 let message = first_compiler_warning(
1937 r#"
1938type Alias[T: (1)()] = ("value")()
1939"#,
1940 );
1941 assert!(
1942 message.contains("'int' object is not callable"),
1943 "expected type parameter warning before alias value warning, got {message:?}"
1944 );
1945 }
1946 }
1947}