1use codehelion_core::conditional::{ArmPath, ArmTracker, StaticCondition};
19use codehelion_core::frontend::{
20 Diagnostic, DiagnosticKind, LexemeInterner, LiteralKind, SourceSpan, Token, TokenKind,
21};
22
23use crate::dialect::Dialect;
24
25const RAW_STRING_PREFIXES: &[&str] = &["u8R", "LR", "uR", "UR", "R"];
27
28const TEXT_PREFIXES: &[&str] = &["u8", "L", "u", "U"];
30
31const DIGRAPHS: &[(&str, &str)] = &[
33 ("%:%:", "##"),
34 ("<:", "["),
35 (":>", "]"),
36 ("<%", "{"),
37 ("%>", "}"),
38 ("%:", "#"),
39];
40
41const TRIGRAPHS: &[(&str, &str)] = &[
43 ("??=", "#"),
44 ("??(", "["),
45 ("??)", "]"),
46 ("??<", "{"),
47 ("??>", "}"),
48 ("??!", "|"),
49 ("??'", "^"),
50 ("??-", "~"),
51];
52
53fn is_ident_start(c: char) -> bool {
54 c.is_alphabetic() || c == '_'
55}
56
57fn is_ident_continue(c: char) -> bool {
58 c.is_alphanumeric() || c == '_'
59}
60
61#[cfg(test)]
62thread_local! {
63 static LEXES_RUN: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
66}
67
68#[cfg(test)]
70fn lexes_run() -> usize {
71 LEXES_RUN.with(std::cell::Cell::get)
72}
73
74struct Lexer<'d, 's> {
75 dialect: &'d Dialect,
76 source: &'s str,
77 chars: Vec<char>,
78 byte_at: Vec<usize>,
79 i: usize,
80 line: u32,
81 column: u32,
82 line_has_token: bool,
86 interner: LexemeInterner,
87 tokens: Vec<Token>,
88 diagnostics: Vec<Diagnostic>,
89 conditional_directives: Vec<(usize, ConditionalDirective)>,
90}
91
92#[derive(Clone, Copy)]
94struct Mark {
95 index: usize,
96 line: u32,
97 column: u32,
98}
99
100impl<'d, 's> Lexer<'d, 's> {
101 fn new(source: &'s str, dialect: &'d Dialect) -> Self {
102 let chars: Vec<char> = source.chars().collect();
103 let mut byte_at = Vec::with_capacity(chars.len() + 1);
104 let mut byte = 0;
105 for c in &chars {
106 byte_at.push(byte);
107 byte += c.len_utf8();
108 }
109 byte_at.push(source.len());
110 Self {
111 dialect,
112 source,
113 chars,
114 byte_at,
115 i: 0,
116 line: 1,
117 column: 1,
118 line_has_token: false,
119 interner: LexemeInterner::new(),
120 tokens: Vec::new(),
121 diagnostics: Vec::new(),
122 conditional_directives: Vec::new(),
123 }
124 }
125
126 fn text_from(&self, start: Mark) -> &'s str {
128 &self.source[self.byte_at[start.index]..self.byte_at[self.i]]
129 }
130
131 fn peek(&self, ahead: usize) -> Option<char> {
132 self.chars.get(self.i + ahead).copied()
133 }
134
135 const fn mark(&self) -> Mark {
136 Mark {
137 index: self.i,
138 line: self.line,
139 column: self.column,
140 }
141 }
142
143 fn bump(&mut self) {
145 if let Some(c) = self.chars.get(self.i) {
146 if *c == '\n' {
147 self.line += 1;
148 self.column = 1;
149 } else {
150 self.column += 1;
151 }
152 self.i += 1;
153 }
154 }
155
156 fn span_from(&self, start: Mark) -> SourceSpan {
157 SourceSpan {
158 start_byte: self.byte_at[start.index],
159 end_byte: self.byte_at[self.i],
160 start_line: start.line,
161 start_column: start.column,
162 }
163 }
164
165 fn push(&mut self, kind: TokenKind, start: Mark) {
166 let text = self.interner.intern(self.text_from(start));
167 self.tokens.push(Token {
168 kind,
169 text,
170 span: self.span_from(start),
171 });
172 }
173
174 fn push_normalized(&mut self, kind: TokenKind, start: Mark, text: &str) {
176 let text = self.interner.intern(text);
177 self.tokens.push(Token {
178 kind,
179 text,
180 span: self.span_from(start),
181 });
182 }
183
184 fn diagnose(&mut self, kind: DiagnosticKind, start: Mark) {
185 let span = self.span_from(start);
186 self.diagnostics.push(Diagnostic { kind, span });
187 }
188
189 fn try_line_splice(&mut self) -> bool {
192 let width = self.splice_width_at(0);
193 if width == 0 {
194 return false;
195 }
196 for _ in 0..width {
197 self.bump();
198 }
199 true
200 }
201
202 fn splice_width_at(&self, ahead: usize) -> usize {
210 let mut width = 0;
211 loop {
212 let at = ahead + width;
213 let marker = if self.peek(at) == Some('\\') {
214 1
215 } else if self.matches_ahead_at(at, "??/") {
216 3
217 } else {
218 return width;
219 };
220 match (self.peek(at + marker), self.peek(at + marker + 1)) {
221 (Some('\n'), _) => width += marker + 1,
222 (Some('\r'), Some('\n')) => width += marker + 2,
223 _ => return width,
224 }
225 }
226 }
227
228 fn spliced_char_at(&self, ahead: usize) -> Option<char> {
231 self.peek(ahead + self.splice_width_at(ahead))
232 }
233
234 fn spliced_match(&self, text: &str) -> Option<usize> {
237 let mut ahead = 0usize;
238 for expected in text.chars() {
239 ahead += self.splice_width_at(ahead);
240 if self.peek(ahead) != Some(expected) {
241 return None;
242 }
243 ahead += 1;
244 }
245 Some(ahead)
246 }
247
248 fn run(self) -> (Vec<Token>, Vec<Diagnostic>) {
249 let (tokens, diagnostics, _) = self.run_with_directives();
250 (tokens, diagnostics)
251 }
252
253 fn run_with_directives(
254 mut self,
255 ) -> (
256 Vec<Token>,
257 Vec<Diagnostic>,
258 Vec<(usize, ConditionalDirective)>,
259 ) {
260 #[cfg(test)]
261 LEXES_RUN.with(|count| count.set(count.get() + 1));
262 while let Some(c) = self.peek(0) {
263 if self.i == 0 && c == '\u{feff}' {
267 self.i += 1;
268 continue;
269 }
270 if c.is_whitespace() {
271 if c == '\n' {
272 self.line_has_token = false;
273 }
274 self.bump();
275 continue;
276 }
277 if let Some(width) = self.spliced_match("//") {
281 self.consume_line_comment(width);
282 continue;
283 }
284 if let Some(width) = self.spliced_match("/*") {
285 self.consume_block_comment(width);
286 continue;
287 }
288 if self.is_directive_marker() && !self.line_has_token {
289 self.consume_directive();
290 continue;
291 }
292 if self.try_line_splice() {
293 continue;
294 }
295 self.line_has_token = true;
296 if self.try_prefixed_literal() {
297 continue;
298 }
299 if c == '"' {
300 let start = self.mark();
301 self.consume_string_from(start);
302 continue;
303 }
304 if c == '\'' {
305 let start = self.mark();
306 self.consume_char_from(start);
307 continue;
308 }
309 if c.is_ascii_digit()
310 || (c == '.' && self.spliced_char_at(1).is_some_and(|d| d.is_ascii_digit()))
311 {
312 self.consume_number();
313 continue;
314 }
315 if is_ident_start(c) {
316 self.consume_ident();
317 continue;
318 }
319 self.consume_punct();
320 }
321 self.tokens.shrink_to_fit();
324 self.conditional_directives.shrink_to_fit();
325 (self.tokens, self.diagnostics, self.conditional_directives)
326 }
327
328 fn consume_line_comment(&mut self, open_width: usize) {
331 for _ in 0..open_width {
332 self.bump();
333 }
334 while let Some(c) = self.peek(0) {
335 if self.try_line_splice() {
336 continue;
338 }
339 if c == '\n' {
340 break;
341 }
342 self.bump();
343 }
344 }
345
346 fn consume_block_comment(&mut self, open_width: usize) {
349 let start = self.mark();
350 for _ in 0..open_width {
351 self.bump();
352 }
353 loop {
354 if let Some(width) = self.spliced_match("*/") {
355 for _ in 0..width {
356 self.bump();
357 }
358 break;
359 }
360 if self.peek(0).is_none() {
361 self.diagnose(DiagnosticKind::UnterminatedBlockComment, start);
362 break;
363 }
364 self.bump();
365 }
366 if self.line > start.line {
370 self.line_has_token = false;
371 }
372 }
373
374 fn consume_directive(&mut self) {
377 let start = self.mark();
378 while let Some(c) = self.peek(0) {
379 if self.try_line_splice() {
380 continue;
381 }
382 if c == '\\' {
383 self.bump();
384 continue;
385 }
386 if c == '\n' {
387 break;
390 }
391 if let Some(width) = self.spliced_match("/*") {
392 self.consume_block_comment(width);
393 continue;
394 }
395 if let Some(width) = self.spliced_match("//") {
396 self.consume_line_comment(width);
397 break;
398 }
399 self.bump();
400 }
401 if let Some(directive) = directive(self.text_from(start)) {
402 self.conditional_directives
403 .push((self.byte_at[start.index], directive));
404 }
405 }
406
407 fn matches_ahead(&self, text: &str) -> bool {
408 self.matches_ahead_at(0, text)
409 }
410
411 fn matches_ahead_at(&self, ahead: usize, text: &str) -> bool {
412 text.chars()
413 .enumerate()
414 .all(|(k, ch)| self.peek(ahead + k) == Some(ch))
415 }
416
417 fn try_prefixed_literal(&mut self) -> bool {
420 if self.dialect.raw_strings {
421 for prefix in RAW_STRING_PREFIXES {
422 if self.matches_ahead(prefix) && self.peek(prefix.len()) == Some('"') {
423 let start = self.mark();
424 for _ in 0..prefix.len() {
425 self.bump();
426 }
427 self.consume_raw_string_body(start);
428 return true;
429 }
430 }
431 }
432 for prefix in TEXT_PREFIXES {
433 if !self.matches_ahead(prefix) {
434 continue;
435 }
436 match self.peek(prefix.len()) {
437 Some('"') => {
438 let start = self.mark();
439 for _ in 0..prefix.len() {
440 self.bump();
441 }
442 self.consume_string_from(start);
443 return true;
444 }
445 Some('\'') => {
446 let start = self.mark();
447 for _ in 0..prefix.len() {
448 self.bump();
449 }
450 self.consume_char_from(start);
451 return true;
452 }
453 _ => {}
454 }
455 }
456 false
457 }
458
459 fn consume_string_from(&mut self, start: Mark) {
464 self.consume_quoted(
465 start,
466 '"',
467 LiteralKind::String,
468 DiagnosticKind::UnterminatedString,
469 );
470 }
471
472 fn consume_char_from(&mut self, start: Mark) {
475 self.consume_quoted(
476 start,
477 '\'',
478 LiteralKind::Char,
479 DiagnosticKind::UnterminatedChar,
480 );
481 }
482
483 fn consume_quoted(
488 &mut self,
489 start: Mark,
490 quote: char,
491 kind: LiteralKind,
492 unterminated: DiagnosticKind,
493 ) {
494 let mut normalized = String::from(self.text_from(start));
495 self.bump();
496 normalized.push(quote);
497 loop {
498 if self.try_line_splice() {
499 continue;
500 }
501 match self.peek(0) {
502 None | Some('\n') => {
503 self.push_normalized(TokenKind::Literal(kind), start, &normalized);
504 self.diagnose(unterminated, start);
505 return;
506 }
507 Some('\\') => {
508 normalized.push('\\');
509 self.bump();
510 while self.try_line_splice() {}
511 match self.peek(0) {
512 None | Some('\n') => {}
513 Some(escaped) => {
514 normalized.push(escaped);
515 self.bump();
516 }
517 }
518 }
519 Some(c) if c == quote => {
520 normalized.push(c);
521 self.bump();
522 self.push_normalized(TokenKind::Literal(kind), start, &normalized);
523 return;
524 }
525 Some(c) => {
526 normalized.push(c);
527 self.bump();
528 }
529 }
530 }
531 }
532
533 fn consume_raw_string_body(&mut self, start: Mark) {
536 self.bump();
538 let mut delim: Vec<char> = Vec::new();
539 loop {
540 match self.peek(0) {
541 Some('(') => {
542 self.bump();
543 break;
544 }
545 Some(c)
546 if c != '"'
547 && c != ')'
548 && c != '\\'
549 && !c.is_whitespace()
550 && delim.len() < 16 =>
551 {
552 delim.push(c);
553 self.bump();
554 }
555 _ => {
556 self.push(TokenKind::Literal(LiteralKind::String), start);
558 self.diagnose(DiagnosticKind::UnterminatedString, start);
559 return;
560 }
561 }
562 }
563 loop {
564 match self.peek(0) {
565 None => {
566 self.push(TokenKind::Literal(LiteralKind::String), start);
567 self.diagnose(DiagnosticKind::UnterminatedString, start);
568 return;
569 }
570 Some(')') => {
571 let closes = delim
572 .iter()
573 .enumerate()
574 .all(|(k, &dc)| self.peek(1 + k) == Some(dc))
575 && self.peek(1 + delim.len()) == Some('"');
576 if closes {
577 for _ in 0..(delim.len() + 2) {
578 self.bump();
579 }
580 self.push(TokenKind::Literal(LiteralKind::String), start);
581 return;
582 }
583 self.bump();
584 }
585 Some(_) => self.bump(),
586 }
587 }
588 }
589
590 fn consume_number(&mut self) {
594 let start = self.mark();
595 let hex = self.spliced_match("0x").is_some() || self.spliced_match("0X").is_some();
596 let mut is_float = false;
597 let mut in_suffix = false;
600 let mut normalized = String::new();
601 loop {
602 if self.try_line_splice() {
603 continue;
604 }
605 let Some(ch) = self.peek(0) else {
606 break;
607 };
608 if ch == '_' {
609 in_suffix = true;
610 }
611 if in_suffix {
612 if !is_ident_continue(ch) {
613 break;
614 }
615 normalized.push(ch);
616 self.bump();
617 } else if ch == '.' {
618 if self.spliced_char_at(1) == Some('.') {
620 break;
621 }
622 is_float = true;
623 normalized.push(ch);
624 self.bump();
625 } else if (hex && matches!(ch, 'p' | 'P')) || (!hex && matches!(ch, 'e' | 'E')) {
626 is_float = true;
628 normalized.push(ch);
629 self.bump();
630 while self.try_line_splice() {}
631 if let Some(sign @ ('+' | '-')) = self.peek(0) {
632 normalized.push(sign);
633 self.bump();
634 }
635 } else if ch == '\''
636 && self.dialect.digit_separators
637 && self
638 .spliced_char_at(1)
639 .is_some_and(|c| c.is_ascii_alphanumeric())
640 {
641 normalized.push(ch);
643 self.bump();
644 } else if is_ident_continue(ch) {
645 normalized.push(ch);
646 self.bump();
647 } else {
648 break;
649 }
650 }
651 let kind = if is_float {
652 LiteralKind::Float
653 } else {
654 LiteralKind::Integer
655 };
656 self.push_normalized(TokenKind::Literal(kind), start, &normalized);
657 }
658
659 fn consume_ident(&mut self) {
660 let start = self.mark();
661 let mut normalized = String::new();
662 loop {
663 if self.try_line_splice() {
664 continue;
665 }
666 let Some(ch) = self.peek(0) else {
667 break;
668 };
669 if !is_ident_continue(ch) {
670 break;
671 }
672 normalized.push(ch);
673 self.bump();
674 }
675 let kind = if self.dialect.keywords.contains(&normalized.as_str()) {
676 TokenKind::Keyword
677 } else {
678 TokenKind::Identifier
679 };
680 self.push_normalized(kind, start, &normalized);
681 }
682
683 fn consume_punct(&mut self) {
688 let start = self.mark();
689 for &(spelling, normalized) in DIGRAPHS.iter().chain(TRIGRAPHS) {
690 if spelling == "<:"
696 && self.dialect.multi_punct.contains(&"::")
697 && let Some(width) = self.spliced_match("<::")
698 && !matches!(self.spliced_char_at(width), Some(':' | '>'))
699 {
700 continue;
701 }
702 if let Some(width) = self.spliced_match(spelling) {
703 for _ in 0..width {
704 self.bump();
705 }
706 self.push_normalized(TokenKind::Punctuation, start, normalized);
707 return;
708 }
709 }
710 for op in self.dialect.multi_punct {
711 if let Some(width) = self.spliced_match(op) {
712 for _ in 0..width {
713 self.bump();
714 }
715 self.push_normalized(TokenKind::Punctuation, start, op);
716 return;
717 }
718 }
719 let c = self.peek(0).unwrap_or('\0');
722 self.bump();
723 if c.is_ascii() && !c.is_alphanumeric() {
724 self.push(TokenKind::Punctuation, start);
725 } else {
726 self.push(TokenKind::Unknown, start);
727 self.diagnose(DiagnosticKind::UnexpectedCharacter, start);
728 }
729 }
730
731 fn is_directive_marker(&self) -> bool {
732 self.peek(0) == Some('#') || self.matches_ahead("%:") || self.matches_ahead("??=")
733 }
734}
735
736#[must_use]
738pub fn lex(source: &str, dialect: &Dialect) -> (Vec<Token>, Vec<Diagnostic>) {
739 Lexer::new(source, dialect).run()
740}
741
742#[must_use]
748pub fn lex_with_directives(
749 source: &str,
750 dialect: &Dialect,
751) -> (Vec<Token>, Vec<Diagnostic>, ConditionalDirectives) {
752 let (tokens, diagnostics, directives) = Lexer::new(source, dialect).run_with_directives();
753 (tokens, diagnostics, ConditionalDirectives(directives))
754}
755
756#[derive(Debug, Clone, Default)]
762pub struct ConditionalDirectives(Vec<(usize, ConditionalDirective)>);
763
764#[must_use]
774pub fn conditional_paths(tokens: &[Token], directives: &ConditionalDirectives) -> Vec<ArmPath> {
775 let directives = &directives.0;
776 if !directives_are_balanced(directives) {
777 return vec![ArmPath::default(); tokens.len()];
778 }
779 let mut next_directive = 0usize;
780 let mut tracker = ArmTracker::default();
781 let mut paths = Vec::with_capacity(tokens.len());
782 for token in tokens {
783 while directives
784 .get(next_directive)
785 .is_some_and(|(offset, _)| *offset < token.span.start_byte)
786 {
787 apply_directive(&mut tracker, directives[next_directive].1);
788 next_directive += 1;
789 }
790 paths.push(tracker.current());
791 }
792 paths
793}
794
795fn directives_are_balanced(directives: &[(usize, ConditionalDirective)]) -> bool {
798 let mut depth = 0usize;
799 for (_, directive) in directives {
800 match directive {
801 ConditionalDirective::Begin(_) => depth = depth.saturating_add(1),
802 ConditionalDirective::Next(_) | ConditionalDirective::End if depth == 0 => {
803 return false;
804 }
805 ConditionalDirective::Next(_) => {}
806 ConditionalDirective::End => depth -= 1,
807 }
808 }
809 depth == 0
810}
811
812#[derive(Debug, Clone, Copy)]
814enum ConditionalDirective {
815 Begin(StaticCondition),
817 Next(StaticCondition),
819 End,
821}
822
823fn directive(line: &str) -> Option<ConditionalDirective> {
825 let line = line.trim_start_matches([' ', '\t', '\r']);
826 let line = line
827 .strip_prefix('#')
828 .or_else(|| line.strip_prefix("%:"))
829 .or_else(|| line.strip_prefix("??="))?
830 .trim_start();
831 let word_end = line
832 .find(|ch: char| !ch.is_ascii_alphabetic())
833 .unwrap_or(line.len());
834 let (word, tail) = line.split_at(word_end);
835 let condition = static_condition(tail);
836 match word {
837 "if" => Some(ConditionalDirective::Begin(condition)),
838 "ifdef" | "ifndef" => Some(ConditionalDirective::Begin(StaticCondition::Unknown)),
839 "elif" => Some(ConditionalDirective::Next(condition)),
840 "elifdef" | "elifndef" | "else" => {
841 Some(ConditionalDirective::Next(StaticCondition::Unknown))
842 }
843 "endif" => Some(ConditionalDirective::End),
844 _ => None,
845 }
846}
847
848fn static_condition(tail: &str) -> StaticCondition {
854 let tail = tail.split_once("//").map_or(tail, |(before, _)| before);
855 let mut condition = String::new();
856 let mut rest = tail;
857 while let Some((before, after)) = rest.split_once("/*") {
858 condition.push_str(before);
859 condition.push(' ');
861 let Some((_, remainder)) = after.split_once("*/") else {
862 rest = "";
864 break;
865 };
866 rest = remainder;
867 }
868 condition.push_str(rest);
869 match condition.trim() {
870 "0" => StaticCondition::False,
871 "1" => StaticCondition::True,
872 _ => StaticCondition::Unknown,
873 }
874}
875
876fn apply_directive(tracker: &mut ArmTracker, directive: ConditionalDirective) {
878 match directive {
879 ConditionalDirective::Begin(condition) => tracker.begin(condition),
880 ConditionalDirective::Next(condition) => tracker.next_arm(condition),
881 ConditionalDirective::End => tracker.end(),
882 }
883}
884
885#[cfg(test)]
886#[allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)]
887mod tests {
888 use super::*;
889 use crate::dialect;
890
891 fn lex_c(source: &str) -> (Vec<Token>, Vec<Diagnostic>) {
892 lex(source, &dialect::C)
893 }
894
895 fn texts(source: &str) -> Vec<String> {
896 lex_c(source).0.iter().map(|t| t.text.to_string()).collect()
897 }
898
899 #[test]
900 fn splits_keywords_identifiers_and_operators() {
901 let (tokens, diags) = lex_c("static int add(int a, struct pair *p) { return a + p->x; }");
902 assert!(diags.is_empty(), "diags: {diags:?}");
903 let pairs: Vec<_> = tokens.iter().map(|t| (t.kind, t.text.as_str())).collect();
904 assert_eq!(pairs[0], (TokenKind::Keyword, "static"));
905 assert_eq!(pairs[1], (TokenKind::Keyword, "int"));
906 assert_eq!(pairs[2], (TokenKind::Identifier, "add"));
907 assert!(pairs.contains(&(TokenKind::Punctuation, "->")));
908 assert!(pairs.contains(&(TokenKind::Keyword, "struct")));
909 }
910
911 #[test]
912 fn drops_comments_and_whitespace() {
913 let src = "int x; // trailing\n/* block\nspanning lines */ int y;";
914 let texts = texts(src);
915 assert!(!texts.iter().any(|t| t.contains("trailing")));
916 assert!(!texts.iter().any(|t| t.contains("spanning")));
917 assert!(texts.contains(&"x".to_string()));
918 assert!(texts.contains(&"y".to_string()));
919 }
920
921 #[test]
922 fn preprocessor_directives_are_dropped_whole() {
923 let src = "#include <stdio.h>\n#define TWICE(x) \\\n ((x) + (x))\nint y;\n";
924 let (tokens, diags) = lex_c(src);
925 assert!(diags.is_empty(), "diags: {diags:?}");
926 let texts: Vec<_> = tokens.iter().map(|t| t.text.as_str()).collect();
927 assert_eq!(texts, vec!["int", "y", ";"]);
928 }
929
930 #[test]
931 fn directive_after_a_multiline_comment_is_still_a_directive() {
932 let src = "int x; /* comment\nspanning */ #define GONE 1\nint y;";
933 let texts = texts(src);
934 assert_eq!(texts, vec!["int", "x", ";", "int", "y", ";"]);
935 }
936
937 #[test]
938 fn a_hash_after_code_on_the_same_line_is_ordinary_punctuation() {
939 let (tokens, _) = lex_c("int a; # 1\nint b;");
941 assert!(
942 tokens
943 .iter()
944 .any(|t| t.kind == TokenKind::Punctuation && t.text == "#")
945 );
946 }
947
948 #[test]
949 fn conditional_compilation_keeps_both_branches() {
950 let src = "#if FLAG\nint a;\n#else\nint b;\n#endif\n";
951 let texts = texts(src);
952 assert_eq!(texts, vec!["int", "a", ";", "int", "b", ";"]);
953 }
954
955 #[test]
956 fn conditional_paths_separate_alternative_arms_and_literal_dead_code() {
957 let src = "#ifdef _WIN32\nint windows_value;\n#else\nint unix_value;\n#endif\n#if 0\nint dead_value;\n#else\nint live_value;\n#endif\n";
958 let (tokens, diagnostics, directives) = lex_with_directives(src, &dialect::C);
959 assert!(diagnostics.is_empty(), "diagnostics: {diagnostics:?}");
960 let paths = conditional_paths(&tokens, &directives);
961 let path_for = |name: &str| {
962 let index = tokens
963 .iter()
964 .position(|token| token.text == name)
965 .unwrap_or_else(|| panic!("missing {name}"));
966 &paths[index]
967 };
968
969 assert!(path_for("windows_value").excludes(path_for("unix_value")));
970 assert!(path_for("dead_value").is_unreachable());
971 assert!(!path_for("live_value").is_unreachable());
972 }
973
974 #[test]
975 fn unclosed_conditionals_do_not_invent_an_exclusion() {
976 let src = "#ifdef MAYBE\nint first_value;\n#else\nint second_value;\n";
977 let (tokens, diagnostics, directives) = lex_with_directives(src, &dialect::C);
978 assert!(diagnostics.is_empty(), "diagnostics: {diagnostics:?}");
979 let paths = conditional_paths(&tokens, &directives);
980 let path_for = |name: &str| {
981 let index = tokens
982 .iter()
983 .position(|token| token.text == name)
984 .unwrap_or_else(|| panic!("missing {name}"));
985 &paths[index]
986 };
987
988 assert!(
989 !path_for("first_value").excludes(path_for("second_value")),
990 "malformed directives must not hide a Fast finding"
991 );
992 }
993
994 #[test]
995 fn comment_pseudo_directives_do_not_make_code_unreachable() {
996 let src = "// #if 0\nint still_live;\n// #endif\n";
997 let (tokens, diagnostics, directives) = lex_with_directives(src, &dialect::C);
998 assert!(diagnostics.is_empty(), "diagnostics: {diagnostics:?}");
999 let paths = conditional_paths(&tokens, &directives);
1000 let index = tokens
1001 .iter()
1002 .position(|token| token.text == "still_live")
1003 .unwrap_or_else(|| panic!("missing still_live"));
1004 assert!(!paths[index].is_unreachable());
1005 }
1006
1007 #[test]
1008 fn a_fast_pass_over_one_source_lexes_it_once() {
1009 let src = "#if 0\nint dead_value;\n#else\nint live_value;\n#endif\n";
1010 let before = lexes_run();
1011 let (tokens, diagnostics, directives) = lex_with_directives(src, &dialect::C);
1012 let paths = conditional_paths(&tokens, &directives);
1013 assert_eq!(
1014 lexes_run() - before,
1015 1,
1016 "arm paths come from the lex that already ran, not from a second one"
1017 );
1018
1019 assert!(diagnostics.is_empty(), "diagnostics: {diagnostics:?}");
1020 assert_eq!(paths.len(), tokens.len());
1021 let path_for = |name: &str| {
1022 let index = tokens
1023 .iter()
1024 .position(|token| token.text == name)
1025 .unwrap_or_else(|| panic!("missing {name}"));
1026 &paths[index]
1027 };
1028 assert!(path_for("dead_value").is_unreachable());
1029 assert!(!path_for("live_value").is_unreachable());
1030 }
1031
1032 #[test]
1033 fn strings_and_chars_lex_with_escapes_and_prefixes() {
1034 let (tokens, diags) = lex_c("char *s = \"a \\\"q\\\" b\"; char c = 'x'; int m = 'ab';");
1035 assert!(diags.is_empty(), "diags: {diags:?}");
1036 let strings: Vec<_> = tokens
1037 .iter()
1038 .filter(|t| t.kind == TokenKind::Literal(LiteralKind::String))
1039 .map(|t| t.text.as_str())
1040 .collect();
1041 assert_eq!(strings, vec!["\"a \\\"q\\\" b\""]);
1042 let chars: Vec<_> = tokens
1043 .iter()
1044 .filter(|t| t.kind == TokenKind::Literal(LiteralKind::Char))
1045 .map(|t| t.text.as_str())
1046 .collect();
1047 assert_eq!(chars, vec!["'x'", "'ab'"]);
1048
1049 let (tokens, diags) = lex_c("const wchar_t *w = L\"wide\"; int u = u8\"n\"[0];");
1050 assert!(diags.is_empty(), "diags: {diags:?}");
1051 let strings: Vec<_> = tokens
1052 .iter()
1053 .filter(|t| t.kind == TokenKind::Literal(LiteralKind::String))
1054 .map(|t| t.text.as_str())
1055 .collect();
1056 assert_eq!(strings, vec!["L\"wide\"", "u8\"n\""]);
1057 }
1058
1059 #[test]
1060 fn unterminated_string_recovers_at_the_line_break() {
1061 let (tokens, diags) = lex_c("char *s = \"open;\nint next;");
1062 assert_eq!(diags.len(), 1);
1063 assert_eq!(diags[0].kind, DiagnosticKind::UnterminatedString);
1064 assert!(
1065 tokens
1066 .iter()
1067 .any(|t| t.kind == TokenKind::Keyword && t.text == "int"),
1068 "lexing must continue on the next line"
1069 );
1070 }
1071
1072 #[test]
1073 fn unterminated_char_and_block_comment_are_diagnosed() {
1074 let (_, diags) = lex_c("char c = 'x\nint y;");
1075 assert!(
1076 diags
1077 .iter()
1078 .any(|d| d.kind == DiagnosticKind::UnterminatedChar)
1079 );
1080 let (_, diags) = lex_c("int x; /* open");
1081 assert!(
1082 diags
1083 .iter()
1084 .any(|d| d.kind == DiagnosticKind::UnterminatedBlockComment)
1085 );
1086 }
1087
1088 #[test]
1089 fn numbers_cover_hex_float_and_suffix_forms() {
1090 let (tokens, diags) = lex_c(
1091 "int a = 0xFF; double b = 1.5e3; double c = 0x1.8p3; long d = 100UL; float e = .5f; float f = 1.f;",
1092 );
1093 assert!(diags.is_empty(), "diags: {diags:?}");
1094 let by_text = |needle: &str| {
1095 tokens
1096 .iter()
1097 .find(|t| t.text == needle)
1098 .unwrap_or_else(|| panic!("token {needle} missing"))
1099 .kind
1100 };
1101 assert_eq!(by_text("0xFF"), TokenKind::Literal(LiteralKind::Integer));
1102 assert_eq!(by_text("1.5e3"), TokenKind::Literal(LiteralKind::Float));
1103 assert_eq!(by_text("0x1.8p3"), TokenKind::Literal(LiteralKind::Float));
1104 assert_eq!(by_text("100UL"), TokenKind::Literal(LiteralKind::Integer));
1105 assert_eq!(by_text(".5f"), TokenKind::Literal(LiteralKind::Float));
1106 assert_eq!(by_text("1.f"), TokenKind::Literal(LiteralKind::Float));
1107 }
1108
1109 #[test]
1110 fn recovers_after_an_unexpected_character() {
1111 let (tokens, diags) = lex_c("int x = \u{20ac}; int next;");
1112 assert_eq!(diags.len(), 1);
1113 assert_eq!(diags[0].kind, DiagnosticKind::UnexpectedCharacter);
1114 assert!(
1115 tokens
1116 .iter()
1117 .any(|t| t.kind == TokenKind::Identifier && t.text == "next")
1118 );
1119 }
1120
1121 #[test]
1122 fn line_splices_join_code_lines() {
1123 let texts = texts("int a \\\n= 1;");
1125 assert_eq!(texts, vec!["int", "a", "=", "1", ";"]);
1126 }
1127
1128 #[test]
1129 fn line_splices_inside_identifiers_preserve_one_normalized_token() {
1130 let texts = texts("int spl\\\nit = 1; int mo??/\nre = split;");
1131 assert_eq!(
1132 texts,
1133 vec![
1134 "int", "split", "=", "1", ";", "int", "more", "=", "split", ";"
1135 ]
1136 );
1137 }
1138
1139 #[test]
1140 fn a_line_splice_is_transparent_inside_every_kind_of_token() {
1141 for (spliced, joined) in [
1145 ("int a = 1; a +\\\n= 1;", "int a = 1; a += 1;"),
1146 ("int a = 12\\\n34;", "int a = 1234;"),
1147 ("double d = 1.5e\\\n3;", "double d = 1.5e3;"),
1148 ("double d = 0x1.8p\\\n3;", "double d = 0x1.8p3;"),
1149 ("int a = 0\\\nx1F;", "int a = 0x1F;"),
1150 (
1151 "int a = 1; /\\\n/ comment\nint b;",
1152 "int a = 1; // comment\nint b;",
1153 ),
1154 (
1155 "int a = 1; /\\\n* comment *\\\n/ int b;",
1156 "int a = 1; /* comment */ int b;",
1157 ),
1158 ("int a = 1 <\\\n< 2;", "int a = 1 << 2;"),
1159 ("int values ?\\\n?(2:> = 0;", "int values <:2:> = 0;"),
1160 ] {
1161 let observed: Vec<_> = lex_c(spliced)
1162 .0
1163 .iter()
1164 .map(|token| (token.kind, token.text.to_string()))
1165 .collect();
1166 let expected: Vec<_> = lex_c(joined)
1167 .0
1168 .iter()
1169 .map(|token| (token.kind, token.text.to_string()))
1170 .collect();
1171 assert_eq!(observed, expected, "{spliced:?}");
1172 }
1173 }
1174
1175 #[test]
1176 fn a_line_continuation_inside_a_quoted_literal_is_transparent() {
1177 for (spliced, joined) in [
1178 ("char *s = \"abc\\\ndef\";", "char *s = \"abcdef\";"),
1179 ("char *s = \"abc\\\r\ndef\";", "char *s = \"abcdef\";"),
1180 ("char *s = \"abc??/\ndef\";", "char *s = \"abcdef\";"),
1181 ("char *s = \"abc\\\\\ndef\";", "char *s = \"abc\\def\";"),
1182 ("char *s = \"a\\\\\\\ndef\";", "char *s = \"a\\\\def\";"),
1183 ("char *s = \"a\\\n\\\nb\";", "char *s = \"ab\";"),
1184 ("char *s = \"a\\\\\\\"\";", "char *s = \"a\\\\\\\"\";"),
1185 ("int c = '\\\n\\n';", "int c = '\\n';"),
1186 ("char *s = L\"w\\\nx\";", "char *s = L\"wx\";"),
1187 ] {
1188 let (observed_tokens, observed_diags) = lex_c(spliced);
1189 let (expected_tokens, expected_diags) = lex_c(joined);
1190 let observed: Vec<_> = observed_tokens
1191 .iter()
1192 .map(|token| (token.kind, token.text.to_string()))
1193 .collect();
1194 let expected: Vec<_> = expected_tokens
1195 .iter()
1196 .map(|token| (token.kind, token.text.to_string()))
1197 .collect();
1198 assert_eq!(observed, expected, "{spliced:?}");
1199 assert_eq!(observed_diags.len(), expected_diags.len(), "{spliced:?}");
1200 assert!(observed_diags.is_empty(), "{spliced:?}");
1201 }
1202 }
1203
1204 #[test]
1205 fn a_user_defined_suffix_does_not_decide_the_numeric_kind() {
1206 let kind_of = |source: &str| lex_c(source).0[0].kind;
1207 for source in ["5_sec", "7_min", "3_e", "0x1F_x", "10_f", "2_pm"] {
1208 assert_eq!(
1209 kind_of(source),
1210 TokenKind::Literal(LiteralKind::Integer),
1211 "{source}"
1212 );
1213 }
1214 for source in ["1.5_deg", "1e3_x", "2.5e-3_e", "0x1p3_x"] {
1215 assert_eq!(
1216 kind_of(source),
1217 TokenKind::Literal(LiteralKind::Float),
1218 "{source}"
1219 );
1220 }
1221 }
1222
1223 #[test]
1224 fn a_condition_classifies_the_same_whichever_comment_syntax_trails_it() {
1225 for source in [
1226 "#if 0\nint dead_value;\n#endif\nint live_value;\n",
1227 "#if 0 // off\nint dead_value;\n#endif\nint live_value;\n",
1228 "#if 0 /* off */\nint dead_value;\n#endif\nint live_value;\n",
1229 "#if /* why */ 0\nint dead_value;\n#endif\nint live_value;\n",
1230 ] {
1231 let (tokens, _, directives) = lex_with_directives(source, &dialect::C);
1232 let paths = conditional_paths(&tokens, &directives);
1233 let path_for = |name: &str| {
1234 let index = tokens
1235 .iter()
1236 .position(|token| token.text == name)
1237 .unwrap_or_else(|| panic!("missing {name} in {source:?}"));
1238 &paths[index]
1239 };
1240 assert!(path_for("dead_value").is_unreachable(), "{source:?}");
1241 assert!(!path_for("live_value").is_unreachable(), "{source:?}");
1242 }
1243
1244 for source in [
1246 "#if FLAG\nint maybe_value;\n#endif\n",
1247 "#if FLAG /* x */\nint maybe_value;\n#endif\n",
1248 ] {
1249 let (tokens, _, directives) = lex_with_directives(source, &dialect::C);
1250 let paths = conditional_paths(&tokens, &directives);
1251 let index = tokens
1252 .iter()
1253 .position(|token| token.text == "maybe_value")
1254 .unwrap_or_else(|| panic!("missing maybe_value"));
1255 assert!(!paths[index].is_unreachable(), "{source:?}");
1256 }
1257 }
1258
1259 #[test]
1260 fn digraphs_and_trigraphs_use_their_canonical_punctuation() {
1261 let texts = texts("%:define COUNT 2\nint values<:COUNT:> = <% 1, 2 %>; int flag ??= 1;");
1262 assert_eq!(
1263 texts,
1264 vec![
1265 "int", "values", "[", "COUNT", "]", "=", "{", "1", ",", "2", "}", ";", "int",
1266 "flag", "#", "1", ";"
1267 ]
1268 );
1269 }
1270
1271 #[test]
1272 fn raw_strings_do_not_exist_in_c() {
1273 let (tokens, diags) = lex_c("R\"(x)\"");
1275 assert!(diags.is_empty(), "diags: {diags:?}");
1276 assert_eq!(tokens[0].kind, TokenKind::Identifier);
1277 assert_eq!(tokens[0].text, "R");
1278 assert_eq!(tokens[1].kind, TokenKind::Literal(LiteralKind::String));
1279 }
1280
1281 #[test]
1282 fn spans_are_byte_accurate() {
1283 let (tokens, _) = lex_c("int x;");
1284 let x = tokens.iter().find(|t| t.text == "x").expect("x token");
1285 assert_eq!(x.span.start_byte, 4);
1286 assert_eq!(x.span.end_byte, 5);
1287 assert_eq!(x.span.start_line, 1);
1288 }
1289
1290 #[test]
1291 fn skips_a_leading_utf8_bom_without_shifting_source_columns() {
1292 let (tokens, diagnostics) = lex_c("\u{feff}int value;");
1293 assert!(diagnostics.is_empty(), "diagnostics: {diagnostics:?}");
1294
1295 let keyword = &tokens[0];
1296 assert_eq!(keyword.kind, TokenKind::Keyword);
1297 assert_eq!(keyword.text, "int");
1298 assert_eq!(keyword.span.start_byte, 3);
1299 assert_eq!(keyword.span.start_line, 1);
1300 assert_eq!(keyword.span.start_column, 1);
1301 }
1302}