1use super::MetricValues;
4
5#[derive(Debug)]
14pub struct CodeAccumulator {
15 syntax: Syntax,
16 state: State,
17 line: Vec<u8>,
18 previous_cr: bool,
19 javascript: JavaScriptContext,
20 shell: ShellContext,
21 metrics: MetricValues,
22}
23
24#[derive(Debug)]
25struct JavaScriptContext {
26 regex_allowed: bool,
27 pending_control_paren: bool,
28 paren_control: Vec<bool>,
29 after_dot: bool,
30}
31
32#[derive(Debug, Default)]
33struct ShellContext {
34 arithmetic_parens: usize,
37}
38
39impl Default for JavaScriptContext {
40 fn default() -> Self {
41 Self {
42 regex_allowed: true,
43 pending_control_paren: false,
44 paren_control: Vec::new(),
45 after_dot: false,
46 }
47 }
48}
49
50impl CodeAccumulator {
51 pub fn for_type(file_type: &str) -> Option<Self> {
54 let syntax = Syntax::for_type(file_type)?;
55 Some(Self {
56 syntax,
57 state: State::Normal,
58 line: Vec::new(),
59 previous_cr: false,
60 javascript: JavaScriptContext::default(),
61 shell: ShellContext::default(),
62 metrics: MetricValues::default(),
63 })
64 }
65
66 pub fn push(&mut self, chunk: &[u8]) {
68 for &byte in chunk {
69 if self.previous_cr {
70 self.previous_cr = false;
71 if byte == b'\n' {
72 continue;
73 }
74 }
75 match byte {
76 b'\r' => {
77 self.finish_line();
78 self.previous_cr = true;
79 }
80 b'\n' => self.finish_line(),
81 _ => self.line.push(byte),
82 }
83 }
84 }
85
86 pub fn finish(mut self) -> MetricValues {
88 if !self.line.is_empty() && self.line.as_slice() != [0xef, 0xbb, 0xbf] {
89 self.finish_line();
90 }
91 self.metrics
92 }
93
94 fn finish_line(&mut self) {
95 let class = classify_line(
96 self.syntax,
97 &mut self.state,
98 &mut self.javascript,
99 &mut self.shell,
100 &self.line,
101 );
102 self.metrics.physical_lines = self.metrics.physical_lines.saturating_add(1);
103 match class {
104 LineClass::Code => {
105 self.metrics.code_lines = self.metrics.code_lines.saturating_add(1);
106 }
107 LineClass::Comment => {
108 self.metrics.comment_lines = self.metrics.comment_lines.saturating_add(1);
109 }
110 LineClass::Blank => {
111 self.metrics.code_blank_lines = self.metrics.code_blank_lines.saturating_add(1);
112 }
113 }
114 self.line.clear();
115 }
116}
117
118#[derive(Clone, Copy, Debug)]
119#[allow(clippy::struct_excessive_bools)]
120struct Syntax {
121 language: Language,
122 line_comments: &'static [&'static [u8]],
123 block: Option<BlockSyntax>,
124 nested_blocks: bool,
125 triple_quotes: bool,
126 backtick_strings: bool,
127 rust_raw_strings: bool,
128 shell_hash_boundary: bool,
129 ruby_blocks: bool,
130}
131
132#[derive(Clone, Copy, Debug, PartialEq, Eq)]
133enum Language {
134 Rust,
135 JavaScript,
136 C,
137 Cpp,
138 CSharp,
139 Java,
140 Kotlin,
141 Swift,
142 Go,
143 Php,
144 Python,
145 Ruby,
146 Shell,
147 Sql,
148}
149
150impl Syntax {
151 fn for_type(file_type: &str) -> Option<Self> {
152 let c_like = Self {
153 language: Language::C,
154 line_comments: &[b"//"],
155 block: Some(BlockSyntax { open: b"/*", close: b"*/" }),
156 nested_blocks: false,
157 triple_quotes: false,
158 backtick_strings: false,
159 rust_raw_strings: false,
160 shell_hash_boundary: false,
161 ruby_blocks: false,
162 };
163 match file_type {
164 "rust" => Some(Self {
165 language: Language::Rust,
166 nested_blocks: true,
167 rust_raw_strings: true,
168 ..c_like
169 }),
170 "javascript" | "typescript" => {
171 Some(Self { language: Language::JavaScript, backtick_strings: true, ..c_like })
172 }
173 "go" => Some(Self { language: Language::Go, backtick_strings: true, ..c_like }),
174 "c" => Some(c_like),
175 "cpp" => Some(Self { language: Language::Cpp, ..c_like }),
176 "csharp" => Some(Self { language: Language::CSharp, ..c_like }),
177 "java" => Some(Self { language: Language::Java, ..c_like }),
178 "kotlin" => Some(Self {
179 language: Language::Kotlin,
180 nested_blocks: true,
181 triple_quotes: true,
182 ..c_like
183 }),
184 "swift" => Some(Self {
185 language: Language::Swift,
186 nested_blocks: true,
187 triple_quotes: true,
188 ..c_like
189 }),
190 "php" => {
191 Some(Self { language: Language::Php, line_comments: &[b"//", b"#"], ..c_like })
192 }
193 "python" | "ruby" => Some(Self {
194 language: if file_type == "ruby" { Language::Ruby } else { Language::Python },
195 line_comments: &[b"#"],
196 block: None,
197 nested_blocks: false,
198 triple_quotes: true,
199 backtick_strings: false,
200 rust_raw_strings: false,
201 shell_hash_boundary: false,
202 ruby_blocks: file_type == "ruby",
203 }),
204 "shell" => Some(Self {
205 language: Language::Shell,
206 line_comments: &[b"#"],
207 block: None,
208 nested_blocks: false,
209 triple_quotes: false,
210 backtick_strings: true,
211 rust_raw_strings: false,
212 shell_hash_boundary: true,
213 ruby_blocks: false,
214 }),
215 "sql" => Some(Self {
216 language: Language::Sql,
217 line_comments: &[b"--"],
218 block: Some(BlockSyntax { open: b"/*", close: b"*/" }),
219 nested_blocks: false,
220 triple_quotes: false,
221 backtick_strings: true,
222 rust_raw_strings: false,
223 shell_hash_boundary: false,
224 ruby_blocks: false,
225 }),
226 _ => None,
227 }
228 }
229}
230
231#[derive(Clone, Copy, Debug)]
232struct BlockSyntax {
233 open: &'static [u8],
234 close: &'static [u8],
235}
236
237#[derive(Clone, Debug)]
238enum State {
239 Normal,
240 BlockComment { depth: u16 },
241 Quoted { quote: u8, escaped: bool, multiline: bool, doubled: bool },
242 TripleQuoted { quote: u8, width: usize },
243 RustRaw { hashes: u8 },
244 Delimited { close: Vec<u8> },
245 Heredoc { terminator: Vec<u8>, indent: bool, php: bool },
246 RubyPercent { open: u8, close: u8, depth: usize },
247 Regex { escaped: bool, in_class: bool },
248 RubyBlock,
249}
250
251#[derive(Clone, Copy, Debug)]
252enum LineClass {
253 Code,
254 Comment,
255 Blank,
256}
257
258fn classify_line(
259 syntax: Syntax,
260 state: &mut State,
261 javascript: &mut JavaScriptContext,
262 shell: &mut ShellContext,
263 line: &[u8],
264) -> LineClass {
265 let mut index = usize::from(line.starts_with(&[0xef, 0xbb, 0xbf]));
266 index = index.saturating_mul(3);
267 let mut whitespace_boundary = matches!(state, State::Normal);
268 let mut code = matches!(
269 state,
270 State::Quoted { .. }
271 | State::TripleQuoted { .. }
272 | State::RustRaw { .. }
273 | State::Delimited { .. }
274 | State::Heredoc { .. }
275 | State::RubyPercent { .. }
276 | State::Regex { .. }
277 );
278 let mut comment = matches!(state, State::BlockComment { .. });
279
280 if let State::Heredoc { terminator, indent, php } = state {
281 let candidate = if *indent {
282 let indentation = line
283 .iter()
284 .take_while(|byte| {
285 if syntax.language == Language::Shell {
286 **byte == b'\t'
287 } else {
288 byte.is_ascii_whitespace()
289 }
290 })
291 .count();
292 &line[indentation..]
293 } else {
294 line
295 };
296 if candidate == terminator
297 || (*php && candidate.strip_suffix(b";") == Some(terminator.as_slice()))
298 {
299 *state = State::Normal;
300 }
301 return LineClass::Code;
302 }
303
304 if matches!(state, State::RubyBlock) {
305 if line.starts_with(b"=end") {
306 *state = State::Normal;
307 }
308 return LineClass::Comment;
309 }
310 if syntax.ruby_blocks && matches!(state, State::Normal) && line.starts_with(b"=begin") {
311 *state = State::RubyBlock;
312 return LineClass::Comment;
313 }
314
315 while index < line.len() {
316 if let State::Delimited { close } = state {
317 code = true;
318 if line[index..].starts_with(close) {
319 index += close.len();
320 *state = State::Normal;
321 } else {
322 index += 1;
323 }
324 continue;
325 }
326 match state.clone() {
327 State::BlockComment { mut depth } => {
328 comment = true;
329 let block = syntax.block.expect("block state requires block syntax");
330 if syntax.nested_blocks && line[index..].starts_with(block.open) {
331 depth = depth.saturating_add(1);
332 *state = State::BlockComment { depth };
333 index += block.open.len();
334 } else if line[index..].starts_with(block.close) {
335 depth = depth.saturating_sub(1);
336 *state = if depth == 0 { State::Normal } else { State::BlockComment { depth } };
337 index += block.close.len();
338 } else {
339 index += 1;
340 }
341 }
342 State::Quoted { quote, mut escaped, multiline, doubled } => {
343 code = true;
344 let byte = line[index];
345 if escaped {
346 escaped = false;
347 } else if byte == b'\\' {
348 escaped = true;
349 } else if byte == quote {
350 if doubled && line.get(index + 1) == Some("e) {
351 index += 2;
352 continue;
353 }
354 *state = State::Normal;
355 if syntax.language == Language::JavaScript {
356 javascript.regex_allowed = false;
357 }
358 index += 1;
359 continue;
360 }
361 *state = State::Quoted { quote, escaped, multiline, doubled };
362 index += 1;
363 }
364 State::TripleQuoted { quote, width } => {
365 code = true;
366 if line[index..].iter().take(width).all(|b| *b == quote)
367 && line.len() - index >= width
368 {
369 *state = State::Normal;
370 if syntax.language == Language::JavaScript {
371 javascript.regex_allowed = false;
372 }
373 index += width;
374 } else {
375 index += 1;
376 }
377 }
378 State::RustRaw { hashes } => {
379 code = true;
380 if rust_raw_close(&line[index..], hashes) {
381 *state = State::Normal;
382 index += usize::from(hashes) + 1;
383 } else {
384 index += 1;
385 }
386 }
387 State::Delimited { .. } => {
388 unreachable!("delimited strings are handled before matching")
389 }
390 State::RubyPercent { open, close, mut depth } => {
391 code = true;
392 match line[index] {
393 b'\\' => index += usize::min(2, line.len() - index),
394 byte if byte == open && open != close => {
395 depth += 1;
396 *state = State::RubyPercent { open, close, depth };
397 index += 1;
398 }
399 byte if byte == close => {
400 depth -= 1;
401 *state = if depth == 0 {
402 State::Normal
403 } else {
404 State::RubyPercent { open, close, depth }
405 };
406 index += 1;
407 }
408 _ => index += 1,
409 }
410 }
411 State::Regex { mut escaped, mut in_class } => {
412 code = true;
413 let byte = line[index];
414 if escaped {
415 escaped = false;
416 } else if byte == b'\\' {
417 escaped = true;
418 } else if byte == b'[' {
419 in_class = true;
420 } else if byte == b']' {
421 in_class = false;
422 } else if byte == b'/' && !in_class {
423 *state = State::Normal;
424 index += 1;
425 javascript.regex_allowed = false;
426 continue;
427 }
428 *state = State::Regex { escaped, in_class };
429 index += 1;
430 }
431 State::Heredoc { .. } => unreachable!("heredocs return before byte scanning"),
432 State::RubyBlock => unreachable!("Ruby blocks return before byte scanning"),
433 State::Normal => {
434 let byte = line[index];
435 if byte.is_ascii_whitespace() {
436 whitespace_boundary = true;
437 index += 1;
438 continue;
439 }
440 if let Some((character, width)) = leading_utf8_character(&line[index..]) {
441 if super::content_basic_metrics::is_content_whitespace(character) {
442 whitespace_boundary = true;
443 index += width;
444 continue;
445 }
446 }
447 if syntax.language == Language::Shell {
448 if shell.arithmetic_parens == 0 && line[index..].starts_with(b"$((") {
449 shell.arithmetic_parens = 2;
450 code = true;
451 whitespace_boundary = false;
452 index += 3;
453 continue;
454 }
455 if shell.arithmetic_parens == 0
456 && whitespace_boundary
457 && line[index..].starts_with(b"((")
458 {
459 shell.arithmetic_parens = 2;
460 code = true;
461 whitespace_boundary = false;
462 index += 2;
463 continue;
464 }
465 if shell.arithmetic_parens > 0 {
466 match byte {
467 b'(' => {
468 shell.arithmetic_parens = shell.arithmetic_parens.saturating_add(1);
469 }
470 b')' => shell.arithmetic_parens -= 1,
471 _ => {}
472 }
473 }
474 }
475 if syntax.language == Language::JavaScript
476 && byte == b'/'
477 && javascript.regex_allowed
478 && !line[index..].starts_with(b"//")
479 && !line[index..].starts_with(b"/*")
480 {
481 code = true;
482 whitespace_boundary = false;
483 *state = State::Regex { escaped: false, in_class: false };
484 index += 1;
485 continue;
486 }
487 if syntax.line_comments.iter().any(|marker| {
488 line[index..].starts_with(marker)
489 && (!syntax.shell_hash_boundary || whitespace_boundary)
490 }) {
491 comment = true;
492 break;
493 }
494 if let Some(block) = syntax.block {
495 if line[index..].starts_with(block.open) {
496 comment = true;
497 whitespace_boundary = false;
498 *state = State::BlockComment { depth: 1 };
499 index += block.open.len();
500 continue;
501 }
502 }
503 if syntax.rust_raw_strings {
504 if let Some((hashes, consumed)) = rust_raw_open(&line[index..]) {
505 code = true;
506 whitespace_boundary = false;
507 *state = State::RustRaw { hashes };
508 index += consumed;
509 continue;
510 }
511 }
512 if syntax.language == Language::Cpp {
513 if let Some((close, consumed)) = cpp_raw_open(&line[index..]) {
514 code = true;
515 *state = State::Delimited { close };
516 index += consumed;
517 continue;
518 }
519 }
520 if syntax.language == Language::Sql {
521 if let Some((close, consumed)) = sql_dollar_open(&line[index..]) {
522 code = true;
523 *state = State::Delimited { close };
524 index += consumed;
525 continue;
526 }
527 }
528 if syntax.language == Language::Ruby {
529 if let Some((open, close, consumed)) = ruby_percent_open(&line[index..]) {
530 code = true;
531 *state = State::RubyPercent { open, close, depth: 1 };
532 index += consumed;
533 continue;
534 }
535 }
536 if let Some((terminator, indent, php)) = (syntax.language != Language::Shell
537 || shell.arithmetic_parens == 0)
538 .then(|| heredoc_open(syntax.language, &line[index..]))
539 .flatten()
540 {
541 code = true;
542 *state = State::Heredoc { terminator, indent, php };
543 break;
544 }
545 if syntax.language == Language::CSharp && line[index..].starts_with(b"@\"") {
546 code = true;
547 *state = State::Quoted {
548 quote: b'"',
549 escaped: false,
550 multiline: true,
551 doubled: true,
552 };
553 index += 2;
554 continue;
555 }
556 if matches!(syntax.language, Language::CSharp | Language::Java)
557 && line[index..].starts_with(b"\"\"\"")
558 {
559 code = true;
560 let width = if syntax.language == Language::CSharp {
561 line[index..].iter().take_while(|b| **b == b'"').count()
562 } else {
563 3
564 };
565 *state = State::TripleQuoted { quote: b'"', width };
566 index += width;
567 continue;
568 }
569 if syntax.triple_quotes
570 && matches!(byte, b'\'' | b'"')
571 && line[index..].starts_with(&[byte, byte, byte])
572 {
573 code = true;
574 whitespace_boundary = false;
575 *state = State::TripleQuoted { quote: byte, width: 3 };
576 index += 3;
577 continue;
578 }
579 if matches!(byte, b'\'' | b'"') || (syntax.backtick_strings && byte == b'`') {
580 if syntax.language == Language::Rust
581 && byte == b'\''
582 && rust_lifetime(&line[index..])
583 {
584 code = true;
585 index += 1;
586 continue;
587 }
588 code = true;
589 whitespace_boundary = false;
590 let multiline = byte == b'`'
591 || (byte == b'"' && syntax.language == Language::Rust)
592 || matches!(
593 syntax.language,
594 Language::Ruby | Language::Shell | Language::Sql
595 )
596 || (syntax.language == Language::C
597 && byte == b'"'
598 && line.ends_with(b"\\"));
599 *state = State::Quoted {
600 quote: byte,
601 escaped: false,
602 multiline,
603 doubled: syntax.language == Language::Sql,
604 };
605 index += 1;
606 continue;
607 }
608 if syntax.language == Language::JavaScript {
609 if byte.is_ascii_alphanumeric() || matches!(byte, b'_' | b'$') {
610 let end = line[index..]
611 .iter()
612 .take_while(|b| b.is_ascii_alphanumeric() || matches!(b, b'_' | b'$'))
613 .count()
614 + index;
615 let word = &line[index..end];
616 javascript.pending_control_paren = !javascript.after_dot
617 && matches!(word, b"if" | b"while" | b"for" | b"with");
618 javascript.after_dot = false;
619 javascript.regex_allowed = matches!(
620 word,
621 b"return"
622 | b"else"
623 | b"do"
624 | b"throw"
625 | b"case"
626 | b"delete"
627 | b"typeof"
628 | b"void"
629 | b"yield"
630 | b"await"
631 | b"instanceof"
632 | b"in"
633 | b"of"
634 );
635 code = true;
636 index = end;
637 continue;
638 }
639 if byte == b'(' {
640 javascript.paren_control.push(javascript.pending_control_paren);
641 javascript.pending_control_paren = false;
642 javascript.after_dot = false;
643 javascript.regex_allowed = true;
644 code = true;
645 index += 1;
646 continue;
647 }
648 if byte == b')' {
649 javascript.regex_allowed = javascript.paren_control.pop().unwrap_or(false);
650 javascript.pending_control_paren = false;
651 javascript.after_dot = false;
652 code = true;
653 index += 1;
654 continue;
655 }
656 javascript.pending_control_paren = false;
657 javascript.after_dot = byte == b'.';
658 javascript.regex_allowed = matches!(
659 byte,
660 b'=' | b'('
661 | b'['
662 | b'{'
663 | b':'
664 | b','
665 | b';'
666 | b'!'
667 | b'?'
668 | b'&'
669 | b'|'
670 | b'+'
671 | b'-'
672 | b'*'
673 | b'%'
674 | b'^'
675 | b'~'
676 | b'<'
677 | b'>'
678 | b'/'
679 );
680 }
681 code = true;
682 whitespace_boundary = syntax.language == Language::Shell
683 && matches!(byte, b';' | b'&' | b'|' | b'(' | b')' | b'<' | b'>');
684 index += 1;
685 }
686 }
687 }
688
689 match state {
690 State::Quoted { multiline: false, escaped: true, .. }
691 if matches!(syntax.language, Language::C | Language::Cpp) =>
692 {
693 *state =
694 State::Quoted { quote: b'"', escaped: false, multiline: false, doubled: false };
695 }
696 State::Quoted { multiline: false, .. } | State::Regex { .. } => *state = State::Normal,
697 State::Quoted { multiline: true, escaped, .. } => *escaped = false,
698 _ => {}
699 }
700 if code {
701 LineClass::Code
702 } else if comment {
703 LineClass::Comment
704 } else {
705 LineClass::Blank
706 }
707}
708
709fn leading_utf8_character(input: &[u8]) -> Option<(char, usize)> {
710 let width = match *input.first()? {
711 0x00..=0x7f => 1,
712 0xc2..=0xdf => 2,
713 0xe0..=0xef => 3,
714 0xf0..=0xf4 => 4,
715 _ => return None,
716 };
717 let character = std::str::from_utf8(input.get(..width)?).ok()?.chars().next()?;
718 Some((character, width))
719}
720
721fn rust_raw_open(input: &[u8]) -> Option<(u8, usize)> {
722 if input.first() != Some(&b'r') {
723 return None;
724 }
725 let hashes = input[1..].iter().take_while(|byte| **byte == b'#').count();
726 if hashes > usize::from(u8::MAX) || input.get(hashes + 1) != Some(&b'"') {
727 return None;
728 }
729 Some((u8::try_from(hashes).expect("bounded above"), hashes + 2))
730}
731
732fn rust_raw_close(input: &[u8], hashes: u8) -> bool {
733 input.first() == Some(&b'"')
734 && (hashes == 0
735 || input
736 .get(1..=usize::from(hashes))
737 .is_some_and(|tail| tail.iter().all(|byte| *byte == b'#')))
738}
739
740fn rust_lifetime(input: &[u8]) -> bool {
741 let Some(first) = input.get(1) else { return false };
742 if !first.is_ascii_alphabetic() && *first != b'_' {
743 return false;
744 }
745 let name_len =
746 input[1..].iter().take_while(|byte| byte.is_ascii_alphanumeric() || **byte == b'_').count();
747 !(name_len == 1 && input.get(2) == Some(&b'\''))
748}
749
750fn cpp_raw_open(input: &[u8]) -> Option<(Vec<u8>, usize)> {
751 if !input.starts_with(b"R\"") {
752 return None;
753 }
754 let end = input[2..].iter().position(|byte| *byte == b'(')? + 2;
755 let delimiter = &input[2..end];
756 if delimiter.len() > 16
757 || delimiter.iter().any(|b| b.is_ascii_whitespace() || matches!(b, b'\\' | b'(' | b')'))
758 {
759 return None;
760 }
761 let mut close = Vec::with_capacity(delimiter.len() + 2);
762 close.push(b')');
763 close.extend_from_slice(delimiter);
764 close.push(b'"');
765 Some((close, end + 1))
766}
767
768fn sql_dollar_open(input: &[u8]) -> Option<(Vec<u8>, usize)> {
769 if input.first() != Some(&b'$') {
770 return None;
771 }
772 let end = input[1..].iter().position(|byte| *byte == b'$')? + 1;
773 let tag = &input[1..end];
774 if !tag.is_empty()
775 && (!tag[0].is_ascii_alphabetic() && tag[0] != b'_'
776 || tag.iter().any(|b| !b.is_ascii_alphanumeric() && *b != b'_'))
777 {
778 return None;
779 }
780 Some((input[..=end].to_vec(), end + 1))
781}
782
783fn ruby_percent_open(input: &[u8]) -> Option<(u8, u8, usize)> {
784 if input.first() != Some(&b'%') {
785 return None;
786 }
787 let delimiter_index = match input.get(1) {
788 Some(b'q' | b'Q' | b'w' | b'W' | b'i' | b'I' | b'r' | b'x' | b's') => 2,
789 Some(b'{' | b'[' | b'(' | b'<') => 1,
790 _ => return None,
791 };
792 let open = *input.get(delimiter_index)?;
793 let close = match open {
794 b'{' => b'}',
795 b'[' => b']',
796 b'(' => b')',
797 b'<' => b'>',
798 b'/' | b'!' | b'|' => open,
799 _ => return None,
800 };
801 Some((open, close, delimiter_index + 1))
802}
803
804fn heredoc_open(language: Language, input: &[u8]) -> Option<(Vec<u8>, bool, bool)> {
805 let php = language == Language::Php;
806 if !matches!(language, Language::Ruby | Language::Shell | Language::Php) {
807 return None;
808 }
809 let mut tail = if php { input.strip_prefix(b"<<<")? } else { input.strip_prefix(b"<<")? };
810 let indent = if !php && matches!(tail.first(), Some(b'-' | b'~')) {
811 tail = &tail[1..];
812 true
813 } else {
814 false
815 };
816 let quote = if matches!(tail.first(), Some(b'\'' | b'"')) {
817 let quote = tail[0];
818 tail = &tail[1..];
819 Some(quote)
820 } else {
821 None
822 };
823 let width = tail.iter().take_while(|b| b.is_ascii_alphanumeric() || **b == b'_').count();
824 if width == 0 || !tail[0].is_ascii_alphabetic() && tail[0] != b'_' {
825 return None;
826 }
827 if let Some(quote) = quote {
828 if tail.get(width) != Some("e) {
829 return None;
830 }
831 }
832 Some((tail[..width].to_vec(), indent, php))
833}
834
835#[cfg(test)]
836mod tests {
837 use super::*;
838
839 type PartitionCase<'a> = (&'a str, &'a [u8], (u64, u64, u64));
840
841 fn count(language: &str, chunks: &[&[u8]]) -> MetricValues {
842 let mut counter = CodeAccumulator::for_type(language).expect("supported language");
843 for chunk in chunks {
844 counter.push(chunk);
845 }
846 counter.finish()
847 }
848
849 #[test]
850 fn partitions_c_like_source_and_counts_mixed_lines_as_code() {
851 let metrics = count(
852 "javascript",
853 &[b"// first\r\nlet url = \"https://example.test\"; // tail\r/* block\n\nend */\n`// text\nmore`;"],
854 );
855 assert_eq!(metrics.physical_lines, 7);
856 assert_eq!(metrics.code_lines, 3);
857 assert_eq!(metrics.comment_lines, 4);
858 assert_eq!(metrics.code_blank_lines, 0);
859 }
860
861 #[test]
862 fn rust_nested_comments_and_raw_strings_ignore_comment_markers() {
863 let source =
864 b"/* outer\n/* inner */\n*/\nlet raw = r##\"/* text */\n// still text\"##;\n\n";
865 let expected = count("rust", &[source]);
866 assert_eq!(expected.physical_lines, 6);
867 assert_eq!(expected.code_lines, 2);
868 assert_eq!(expected.comment_lines, 3);
869 assert_eq!(expected.code_blank_lines, 1);
870 for split in 0..=source.len() {
871 assert_eq!(count("rust", &[&source[..split], &source[split..]]), expected);
872 }
873 }
874
875 #[test]
876 fn triple_quoted_docstrings_are_code_in_v1() {
877 let metrics = count("python", &[b"\"\"\"docs\n# text\n\"\"\"\n# comment\npass\n"]);
878 assert_eq!(metrics.physical_lines, 5);
879 assert_eq!(metrics.code_lines, 4);
880 assert_eq!(metrics.comment_lines, 1);
881 assert_eq!(metrics.code_blank_lines, 0);
882 }
883
884 #[test]
885 fn every_line_ending_convention_has_the_same_partition() {
886 for source in [
887 "// comment\nlet value = 1;\n\n",
888 "// comment\r\nlet value = 1;\r\n\r\n",
889 "// comment\rlet value = 1;\r\r",
890 "// comment\r\nlet value = 1;\r\n",
891 ] {
892 let metrics = count("rust", &[source.as_bytes()]);
893 let expected_lines = if source.ends_with("value = 1;\r\n") { 2 } else { 3 };
894 assert_eq!(metrics.physical_lines, expected_lines, "{source:?}");
895 assert_eq!(metrics.code_lines, 1, "{source:?}");
896 assert_eq!(metrics.comment_lines, 1, "{source:?}");
897 assert_eq!(metrics.code_blank_lines, expected_lines - 2, "{source:?}");
898 }
899 }
900
901 #[test]
902 fn a_leading_utf8_bom_is_not_an_invented_line() {
903 let empty = count("rust", &[b"\xef\xbb\xbf"]);
904 assert_eq!(empty.physical_lines, 0);
905
906 let blank = count("rust", &[b"\xef\xbb\xbf\n"]);
907 assert_eq!(blank.physical_lines, 1);
908 assert_eq!(blank.code_blank_lines, 1);
909 }
910
911 #[test]
912 fn unicode_whitespace_uses_the_basic_analyzers_pinned_table() {
913 let metrics = count("rust", &["\u{3000}\n\u{2003}// comment\n".as_bytes()]);
914 assert_eq!(metrics.physical_lines, 2);
915 assert_eq!(metrics.code_blank_lines, 1);
916 assert_eq!(metrics.comment_lines, 1);
917 assert_eq!(metrics.code_lines, 0);
918
919 let shell = count("shell", &["\u{3000}# comment\n".as_bytes()]);
920 assert_eq!(shell.comment_lines, 1);
921 assert_eq!(shell.code_lines, 0);
922 }
923
924 #[test]
925 fn unsupported_languages_are_explicit() {
926 assert!(CodeAccumulator::for_type("haskell").is_none());
927 }
928
929 #[test]
930 fn rust_multiline_string_and_lifetime_preserve_following_comment_state() {
931 let string = b"const S: &str = \"first\n// text\nlast\";\n";
932 let lifetime = b"fn x<'a>() { /*\ncomment\n*/ }\n";
933 for (source, code, comment) in [(string.as_slice(), 3, 0), (lifetime.as_slice(), 2, 1)] {
934 for split in 0..=source.len() {
935 let metrics = count("rust", &[&source[..split], &source[split..]]);
936 assert_eq!(
937 (metrics.code_lines, metrics.comment_lines),
938 (code, comment),
939 "split {split}"
940 );
941 }
942 }
943 }
944
945 #[test]
946 fn multiline_literal_families_preserve_comment_markers() {
947 let cases: &[PartitionCase<'_>] = &[
948 ("cpp", b"const char *s = R\"tag(first\n// text\nlast)tag\";\n", (3, 0, 0)),
949 ("c", b"const char *s = \"first\\\n// text\";\n", (2, 0, 0)),
950 ("java", b"class C { String s = \"\"\"\n// text\nlast\n\"\"\"; }\n", (4, 0, 0)),
951 ("csharp", b"class C { string s = @\"first\n// text\nlast\"; }\n", (3, 0, 0)),
952 ("csharp", b"class C { string s = \"\"\"\n// text\nlast\n\"\"\"; }\n", (4, 0, 0)),
953 ("ruby", b"s = %q{first\n# text\nlast}\n", (3, 0, 0)),
954 ("ruby", b"s = <<~TEXT\n# text\nlast\nTEXT\n", (4, 0, 0)),
955 ("ruby", b"s = \"first\n# text\nlast\"\n", (3, 0, 0)),
956 ("shell", b"cat <<'TEXT'\n# text\nlast\nTEXT\n", (4, 0, 0)),
957 ("shell", b"value='first\n# text\nlast'\n", (3, 0, 0)),
958 ("sql", b"SELECT $tag$first\n-- text\nlast$tag$;\n", (3, 0, 0)),
959 ("sql", b"SELECT 'first\n-- text\nlast';\n", (3, 0, 0)),
960 ("php", b"<?php\n$s = <<<TEXT\n// text\nlast\nTEXT;\n", (5, 0, 0)),
961 ];
962 for (language, source, expected) in cases {
963 for split in 0..=source.len() {
964 let metrics = count(language, &[&source[..split], &source[split..]]);
965 assert_eq!(
966 (metrics.code_lines, metrics.comment_lines, metrics.code_blank_lines),
967 *expected,
968 "{language} split {split}"
969 );
970 }
971 }
972 }
973
974 #[test]
975 fn multiline_delimiters_restore_comment_recognition_after_closing() {
976 let cases: &[(&str, &[u8], (u64, u64))] = &[
977 ("cpp", b"auto s = R\"x(/*\n// body\n)x\";\n// comment\n", (3, 1)),
978 ("csharp", b"var s = @\"first \"\" quote\n// body\nlast\";\n// comment\n", (3, 1)),
979 ("ruby", b"s = %q{outer {inner\n# body\n}}\n# comment\n", (3, 1)),
980 ("ruby", b"s = <<~TEXT\n# body\n TEXT\n# comment\n", (3, 1)),
981 ("shell", b"cat <<'TEXT'\n# body\nTEXT\n# comment\n", (3, 1)),
982 ("shell", b"cat <<-TEXT\n TEXT\n# body\nTEXT\n# comment\n", (4, 1)),
983 ("shell", b"cat <<-TEXT\n\tTEXT\n# comment\n", (2, 1)),
984 ("sql", b"SELECT $tag$first\n-- body\nlast$tag$;\n-- comment\n", (3, 1)),
985 ("php", b"$s = <<<TEXT\n// body\nTEXT;\n// comment\n", (3, 1)),
986 ];
987 for (language, source, expected) in cases {
988 let metrics = count(language, &[source]);
989 assert_eq!((metrics.code_lines, metrics.comment_lines), *expected, "{language}");
990 }
991 }
992
993 #[test]
994 fn shell_comments_and_arithmetic_shifts_do_not_hold_lexer_state() {
995 let cases: &[PartitionCase<'_>] = &[
996 ("shell", b"true;# \"unterminated\n# following\nprintf ok\n", (2, 1, 0)),
999 ("shell", b"N=2\nx=$((1<<N))\n# following\n", (2, 1, 0)),
1002 ("shell", b"N=2\n((1<<N))\n# following\n", (2, 1, 0)),
1003 ("shell", b"printf '%s' foo#bar\ncat <<TEXT\n# literal\nTEXT\n# comment\n", (4, 1, 0)),
1005 ];
1006 for (language, source, expected) in cases {
1007 for split in 0..=source.len() {
1008 let metrics = count(language, &[&source[..split], &source[split..]]);
1009 assert_eq!(
1010 (metrics.code_lines, metrics.comment_lines, metrics.code_blank_lines),
1011 *expected,
1012 "source {source:?}, split {split}"
1013 );
1014 }
1015 }
1016 }
1017
1018 #[test]
1019 fn javascript_regex_and_division_leave_comment_state_correct() {
1020 let cases: &[PartitionCase<'_>] = &[
1021 ("javascript", b"const re = /[/*]/;\nconst answer = 42;\n", (2, 0, 0)),
1022 ("typescript", b"const re = /\\/\\* inside [/] /;\nconst n = 9;\n", (2, 0, 0)),
1023 ("javascript", b"const ratio = total / count;\n/* comment */\nconst re = /[//]/;\n", (2, 1, 0)),
1024 ("typescript", b"return /[/*]/.test(value);\n// comment\nnext();\n", (2, 1, 0)),
1025 ("javascript", b"const quotient = total\n / count; /* real comment */\nconst re = /[/*]/;\nnext();\n", (4, 0, 0)),
1026 ("javascript", b"const text = \"plain\" / count;\nconst re = /[/*]/;\nnext();\n", (3, 0, 0)),
1027 ("javascript", b"if (ok) /[/*]/.test(value);\nconst next = 1;\n", (2, 0, 0)),
1028 ("javascript", b"if (first) {}\nif (ok) /[/*]/.test(value);\nnext();\n", (3, 0, 0)),
1029 ("typescript", b"while (ready && check(x)) /[/*]/.test(value);\nnext();\n", (2, 0, 0)),
1030 ("javascript", b"if (ok &&\n check(x)) /[/*]/.test(value);\nnext();\n", (3, 0, 0)),
1031 ("javascript", b"if (ok) fn(value) / count;\n/* comment */\n", (1, 1, 0)),
1032 ("javascript", b"const ratio = object.if(value) / count;\n/* comment */\n", (1, 1, 0)),
1033 ];
1034 for (language, source, expected) in cases {
1035 for split in 0..=source.len() {
1036 let metrics = count(language, &[&source[..split], &source[split..]]);
1037 assert_eq!(
1038 (metrics.code_lines, metrics.comment_lines, metrics.code_blank_lines),
1039 *expected,
1040 "{language} split {split}"
1041 );
1042 }
1043 }
1044 }
1045}