Skip to main content

fdu_core/content/
content_code_metrics.rs

1//! Streaming common-language source-line classification.
2
3use super::MetricValues;
4
5/// Streaming `code-sloc-v1` counter for a supported language (analyzer version 3).
6///
7/// The counter retains the current logical line, its allocated capacity, and parser
8/// state. A one-line minified or generated source can therefore require file-sized
9/// memory per active worker. Mixed
10/// code/comment lines are code, blank lines inside block comments are comments, and
11/// multiline string lines are code. Line endings follow the same LF, CRLF, lone-CR,
12/// and unterminated-final-line contract as the basic analyzer.
13#[derive(Debug)]
14pub struct CodeAccumulator {
15    syntax: Syntax,
16    state: State,
17    line: Vec<u8>,
18    previous_cr: bool,
19    javascript: JavaScriptContext,
20    shell: ShellContext,
21    metrics: MetricValues,
22}
23
24#[derive(Debug)]
25struct JavaScriptContext {
26    regex_allowed: bool,
27    pending_control_paren: bool,
28    paren_control: Vec<bool>,
29    after_dot: bool,
30}
31
32#[derive(Debug, Default)]
33struct ShellContext {
34    // The number of unmatched parentheses in a shell arithmetic expression.
35    // `<<` there is a shift operator, not a heredoc opener.
36    arithmetic_parens: usize,
37}
38
39impl Default for JavaScriptContext {
40    fn default() -> Self {
41        Self {
42            regex_allowed: true,
43            pending_control_paren: false,
44            paren_control: Vec::new(),
45            after_dot: false,
46        }
47    }
48}
49
50impl CodeAccumulator {
51    /// Create a counter for one stable file-type ID, or return `None` when
52    /// `code-sloc-v1` does not claim that language.
53    pub fn for_type(file_type: &str) -> Option<Self> {
54        let syntax = Syntax::for_type(file_type)?;
55        Some(Self {
56            syntax,
57            state: State::Normal,
58            line: Vec::new(),
59            previous_cr: false,
60            javascript: JavaScriptContext::default(),
61            shell: ShellContext::default(),
62            metrics: MetricValues::default(),
63        })
64    }
65
66    /// Consume an arbitrary byte chunk.
67    pub fn push(&mut self, chunk: &[u8]) {
68        for &byte in chunk {
69            if self.previous_cr {
70                self.previous_cr = false;
71                if byte == b'\n' {
72                    continue;
73                }
74            }
75            match byte {
76                b'\r' => {
77                    self.finish_line();
78                    self.previous_cr = true;
79                }
80                b'\n' => self.finish_line(),
81                _ => self.line.push(byte),
82            }
83        }
84    }
85
86    /// Finish an unterminated final line and return its additive metrics.
87    pub fn finish(mut self) -> MetricValues {
88        if !self.line.is_empty() && self.line.as_slice() != [0xef, 0xbb, 0xbf] {
89            self.finish_line();
90        }
91        self.metrics
92    }
93
94    fn finish_line(&mut self) {
95        let class = classify_line(
96            self.syntax,
97            &mut self.state,
98            &mut self.javascript,
99            &mut self.shell,
100            &self.line,
101        );
102        self.metrics.physical_lines = self.metrics.physical_lines.saturating_add(1);
103        match class {
104            LineClass::Code => {
105                self.metrics.code_lines = self.metrics.code_lines.saturating_add(1);
106            }
107            LineClass::Comment => {
108                self.metrics.comment_lines = self.metrics.comment_lines.saturating_add(1);
109            }
110            LineClass::Blank => {
111                self.metrics.code_blank_lines = self.metrics.code_blank_lines.saturating_add(1);
112            }
113        }
114        self.line.clear();
115    }
116}
117
118#[derive(Clone, Copy, Debug)]
119#[allow(clippy::struct_excessive_bools)]
120struct Syntax {
121    language: Language,
122    line_comments: &'static [&'static [u8]],
123    block: Option<BlockSyntax>,
124    nested_blocks: bool,
125    triple_quotes: bool,
126    backtick_strings: bool,
127    rust_raw_strings: bool,
128    shell_hash_boundary: bool,
129    ruby_blocks: bool,
130}
131
132#[derive(Clone, Copy, Debug, PartialEq, Eq)]
133enum Language {
134    Rust,
135    JavaScript,
136    C,
137    Cpp,
138    CSharp,
139    Java,
140    Kotlin,
141    Swift,
142    Go,
143    Php,
144    Python,
145    Ruby,
146    Shell,
147    Sql,
148}
149
150impl Syntax {
151    fn for_type(file_type: &str) -> Option<Self> {
152        let c_like = Self {
153            language: Language::C,
154            line_comments: &[b"//"],
155            block: Some(BlockSyntax { open: b"/*", close: b"*/" }),
156            nested_blocks: false,
157            triple_quotes: false,
158            backtick_strings: false,
159            rust_raw_strings: false,
160            shell_hash_boundary: false,
161            ruby_blocks: false,
162        };
163        match file_type {
164            "rust" => Some(Self {
165                language: Language::Rust,
166                nested_blocks: true,
167                rust_raw_strings: true,
168                ..c_like
169            }),
170            "javascript" | "typescript" => {
171                Some(Self { language: Language::JavaScript, backtick_strings: true, ..c_like })
172            }
173            "go" => Some(Self { language: Language::Go, backtick_strings: true, ..c_like }),
174            "c" => Some(c_like),
175            "cpp" => Some(Self { language: Language::Cpp, ..c_like }),
176            "csharp" => Some(Self { language: Language::CSharp, ..c_like }),
177            "java" => Some(Self { language: Language::Java, ..c_like }),
178            "kotlin" => Some(Self {
179                language: Language::Kotlin,
180                nested_blocks: true,
181                triple_quotes: true,
182                ..c_like
183            }),
184            "swift" => Some(Self {
185                language: Language::Swift,
186                nested_blocks: true,
187                triple_quotes: true,
188                ..c_like
189            }),
190            "php" => {
191                Some(Self { language: Language::Php, line_comments: &[b"//", b"#"], ..c_like })
192            }
193            "python" | "ruby" => Some(Self {
194                language: if file_type == "ruby" { Language::Ruby } else { Language::Python },
195                line_comments: &[b"#"],
196                block: None,
197                nested_blocks: false,
198                triple_quotes: true,
199                backtick_strings: false,
200                rust_raw_strings: false,
201                shell_hash_boundary: false,
202                ruby_blocks: file_type == "ruby",
203            }),
204            "shell" => Some(Self {
205                language: Language::Shell,
206                line_comments: &[b"#"],
207                block: None,
208                nested_blocks: false,
209                triple_quotes: false,
210                backtick_strings: true,
211                rust_raw_strings: false,
212                shell_hash_boundary: true,
213                ruby_blocks: false,
214            }),
215            "sql" => Some(Self {
216                language: Language::Sql,
217                line_comments: &[b"--"],
218                block: Some(BlockSyntax { open: b"/*", close: b"*/" }),
219                nested_blocks: false,
220                triple_quotes: false,
221                backtick_strings: true,
222                rust_raw_strings: false,
223                shell_hash_boundary: false,
224                ruby_blocks: false,
225            }),
226            _ => None,
227        }
228    }
229}
230
231#[derive(Clone, Copy, Debug)]
232struct BlockSyntax {
233    open: &'static [u8],
234    close: &'static [u8],
235}
236
237#[derive(Clone, Debug)]
238enum State {
239    Normal,
240    BlockComment { depth: u16 },
241    Quoted { quote: u8, escaped: bool, multiline: bool, doubled: bool },
242    TripleQuoted { quote: u8, width: usize },
243    RustRaw { hashes: u8 },
244    Delimited { close: Vec<u8> },
245    Heredoc { terminator: Vec<u8>, indent: bool, php: bool },
246    RubyPercent { open: u8, close: u8, depth: usize },
247    Regex { escaped: bool, in_class: bool },
248    RubyBlock,
249}
250
251#[derive(Clone, Copy, Debug)]
252enum LineClass {
253    Code,
254    Comment,
255    Blank,
256}
257
258fn classify_line(
259    syntax: Syntax,
260    state: &mut State,
261    javascript: &mut JavaScriptContext,
262    shell: &mut ShellContext,
263    line: &[u8],
264) -> LineClass {
265    let mut index = usize::from(line.starts_with(&[0xef, 0xbb, 0xbf]));
266    index = index.saturating_mul(3);
267    let mut whitespace_boundary = matches!(state, State::Normal);
268    let mut code = matches!(
269        state,
270        State::Quoted { .. }
271            | State::TripleQuoted { .. }
272            | State::RustRaw { .. }
273            | State::Delimited { .. }
274            | State::Heredoc { .. }
275            | State::RubyPercent { .. }
276            | State::Regex { .. }
277    );
278    let mut comment = matches!(state, State::BlockComment { .. });
279
280    if let State::Heredoc { terminator, indent, php } = state {
281        let candidate = if *indent {
282            let indentation = line
283                .iter()
284                .take_while(|byte| {
285                    if syntax.language == Language::Shell {
286                        **byte == b'\t'
287                    } else {
288                        byte.is_ascii_whitespace()
289                    }
290                })
291                .count();
292            &line[indentation..]
293        } else {
294            line
295        };
296        if candidate == terminator
297            || (*php && candidate.strip_suffix(b";") == Some(terminator.as_slice()))
298        {
299            *state = State::Normal;
300        }
301        return LineClass::Code;
302    }
303
304    if matches!(state, State::RubyBlock) {
305        if line.starts_with(b"=end") {
306            *state = State::Normal;
307        }
308        return LineClass::Comment;
309    }
310    if syntax.ruby_blocks && matches!(state, State::Normal) && line.starts_with(b"=begin") {
311        *state = State::RubyBlock;
312        return LineClass::Comment;
313    }
314
315    while index < line.len() {
316        if let State::Delimited { close } = state {
317            code = true;
318            if line[index..].starts_with(close) {
319                index += close.len();
320                *state = State::Normal;
321            } else {
322                index += 1;
323            }
324            continue;
325        }
326        match state.clone() {
327            State::BlockComment { mut depth } => {
328                comment = true;
329                let block = syntax.block.expect("block state requires block syntax");
330                if syntax.nested_blocks && line[index..].starts_with(block.open) {
331                    depth = depth.saturating_add(1);
332                    *state = State::BlockComment { depth };
333                    index += block.open.len();
334                } else if line[index..].starts_with(block.close) {
335                    depth = depth.saturating_sub(1);
336                    *state = if depth == 0 { State::Normal } else { State::BlockComment { depth } };
337                    index += block.close.len();
338                } else {
339                    index += 1;
340                }
341            }
342            State::Quoted { quote, mut escaped, multiline, doubled } => {
343                code = true;
344                let byte = line[index];
345                if escaped {
346                    escaped = false;
347                } else if byte == b'\\' {
348                    escaped = true;
349                } else if byte == quote {
350                    if doubled && line.get(index + 1) == Some(&quote) {
351                        index += 2;
352                        continue;
353                    }
354                    *state = State::Normal;
355                    if syntax.language == Language::JavaScript {
356                        javascript.regex_allowed = false;
357                    }
358                    index += 1;
359                    continue;
360                }
361                *state = State::Quoted { quote, escaped, multiline, doubled };
362                index += 1;
363            }
364            State::TripleQuoted { quote, width } => {
365                code = true;
366                if line[index..].iter().take(width).all(|b| *b == quote)
367                    && line.len() - index >= width
368                {
369                    *state = State::Normal;
370                    if syntax.language == Language::JavaScript {
371                        javascript.regex_allowed = false;
372                    }
373                    index += width;
374                } else {
375                    index += 1;
376                }
377            }
378            State::RustRaw { hashes } => {
379                code = true;
380                if rust_raw_close(&line[index..], hashes) {
381                    *state = State::Normal;
382                    index += usize::from(hashes) + 1;
383                } else {
384                    index += 1;
385                }
386            }
387            State::Delimited { .. } => {
388                unreachable!("delimited strings are handled before matching")
389            }
390            State::RubyPercent { open, close, mut depth } => {
391                code = true;
392                match line[index] {
393                    b'\\' => index += usize::min(2, line.len() - index),
394                    byte if byte == open && open != close => {
395                        depth += 1;
396                        *state = State::RubyPercent { open, close, depth };
397                        index += 1;
398                    }
399                    byte if byte == close => {
400                        depth -= 1;
401                        *state = if depth == 0 {
402                            State::Normal
403                        } else {
404                            State::RubyPercent { open, close, depth }
405                        };
406                        index += 1;
407                    }
408                    _ => index += 1,
409                }
410            }
411            State::Regex { mut escaped, mut in_class } => {
412                code = true;
413                let byte = line[index];
414                if escaped {
415                    escaped = false;
416                } else if byte == b'\\' {
417                    escaped = true;
418                } else if byte == b'[' {
419                    in_class = true;
420                } else if byte == b']' {
421                    in_class = false;
422                } else if byte == b'/' && !in_class {
423                    *state = State::Normal;
424                    index += 1;
425                    javascript.regex_allowed = false;
426                    continue;
427                }
428                *state = State::Regex { escaped, in_class };
429                index += 1;
430            }
431            State::Heredoc { .. } => unreachable!("heredocs return before byte scanning"),
432            State::RubyBlock => unreachable!("Ruby blocks return before byte scanning"),
433            State::Normal => {
434                let byte = line[index];
435                if byte.is_ascii_whitespace() {
436                    whitespace_boundary = true;
437                    index += 1;
438                    continue;
439                }
440                if let Some((character, width)) = leading_utf8_character(&line[index..]) {
441                    if super::content_basic_metrics::is_content_whitespace(character) {
442                        whitespace_boundary = true;
443                        index += width;
444                        continue;
445                    }
446                }
447                if syntax.language == Language::Shell {
448                    if shell.arithmetic_parens == 0 && line[index..].starts_with(b"$((") {
449                        shell.arithmetic_parens = 2;
450                        code = true;
451                        whitespace_boundary = false;
452                        index += 3;
453                        continue;
454                    }
455                    if shell.arithmetic_parens == 0
456                        && whitespace_boundary
457                        && line[index..].starts_with(b"((")
458                    {
459                        shell.arithmetic_parens = 2;
460                        code = true;
461                        whitespace_boundary = false;
462                        index += 2;
463                        continue;
464                    }
465                    if shell.arithmetic_parens > 0 {
466                        match byte {
467                            b'(' => {
468                                shell.arithmetic_parens = shell.arithmetic_parens.saturating_add(1);
469                            }
470                            b')' => shell.arithmetic_parens -= 1,
471                            _ => {}
472                        }
473                    }
474                }
475                if syntax.language == Language::JavaScript
476                    && byte == b'/'
477                    && javascript.regex_allowed
478                    && !line[index..].starts_with(b"//")
479                    && !line[index..].starts_with(b"/*")
480                {
481                    code = true;
482                    whitespace_boundary = false;
483                    *state = State::Regex { escaped: false, in_class: false };
484                    index += 1;
485                    continue;
486                }
487                if syntax.line_comments.iter().any(|marker| {
488                    line[index..].starts_with(marker)
489                        && (!syntax.shell_hash_boundary || whitespace_boundary)
490                }) {
491                    comment = true;
492                    break;
493                }
494                if let Some(block) = syntax.block {
495                    if line[index..].starts_with(block.open) {
496                        comment = true;
497                        whitespace_boundary = false;
498                        *state = State::BlockComment { depth: 1 };
499                        index += block.open.len();
500                        continue;
501                    }
502                }
503                if syntax.rust_raw_strings {
504                    if let Some((hashes, consumed)) = rust_raw_open(&line[index..]) {
505                        code = true;
506                        whitespace_boundary = false;
507                        *state = State::RustRaw { hashes };
508                        index += consumed;
509                        continue;
510                    }
511                }
512                if syntax.language == Language::Cpp {
513                    if let Some((close, consumed)) = cpp_raw_open(&line[index..]) {
514                        code = true;
515                        *state = State::Delimited { close };
516                        index += consumed;
517                        continue;
518                    }
519                }
520                if syntax.language == Language::Sql {
521                    if let Some((close, consumed)) = sql_dollar_open(&line[index..]) {
522                        code = true;
523                        *state = State::Delimited { close };
524                        index += consumed;
525                        continue;
526                    }
527                }
528                if syntax.language == Language::Ruby {
529                    if let Some((open, close, consumed)) = ruby_percent_open(&line[index..]) {
530                        code = true;
531                        *state = State::RubyPercent { open, close, depth: 1 };
532                        index += consumed;
533                        continue;
534                    }
535                }
536                if let Some((terminator, indent, php)) = (syntax.language != Language::Shell
537                    || shell.arithmetic_parens == 0)
538                    .then(|| heredoc_open(syntax.language, &line[index..]))
539                    .flatten()
540                {
541                    code = true;
542                    *state = State::Heredoc { terminator, indent, php };
543                    break;
544                }
545                if syntax.language == Language::CSharp && line[index..].starts_with(b"@\"") {
546                    code = true;
547                    *state = State::Quoted {
548                        quote: b'"',
549                        escaped: false,
550                        multiline: true,
551                        doubled: true,
552                    };
553                    index += 2;
554                    continue;
555                }
556                if matches!(syntax.language, Language::CSharp | Language::Java)
557                    && line[index..].starts_with(b"\"\"\"")
558                {
559                    code = true;
560                    let width = if syntax.language == Language::CSharp {
561                        line[index..].iter().take_while(|b| **b == b'"').count()
562                    } else {
563                        3
564                    };
565                    *state = State::TripleQuoted { quote: b'"', width };
566                    index += width;
567                    continue;
568                }
569                if syntax.triple_quotes
570                    && matches!(byte, b'\'' | b'"')
571                    && line[index..].starts_with(&[byte, byte, byte])
572                {
573                    code = true;
574                    whitespace_boundary = false;
575                    *state = State::TripleQuoted { quote: byte, width: 3 };
576                    index += 3;
577                    continue;
578                }
579                if matches!(byte, b'\'' | b'"') || (syntax.backtick_strings && byte == b'`') {
580                    if syntax.language == Language::Rust
581                        && byte == b'\''
582                        && rust_lifetime(&line[index..])
583                    {
584                        code = true;
585                        index += 1;
586                        continue;
587                    }
588                    code = true;
589                    whitespace_boundary = false;
590                    let multiline = byte == b'`'
591                        || (byte == b'"' && syntax.language == Language::Rust)
592                        || matches!(
593                            syntax.language,
594                            Language::Ruby | Language::Shell | Language::Sql
595                        )
596                        || (syntax.language == Language::C
597                            && byte == b'"'
598                            && line.ends_with(b"\\"));
599                    *state = State::Quoted {
600                        quote: byte,
601                        escaped: false,
602                        multiline,
603                        doubled: syntax.language == Language::Sql,
604                    };
605                    index += 1;
606                    continue;
607                }
608                if syntax.language == Language::JavaScript {
609                    if byte.is_ascii_alphanumeric() || matches!(byte, b'_' | b'$') {
610                        let end = line[index..]
611                            .iter()
612                            .take_while(|b| b.is_ascii_alphanumeric() || matches!(b, b'_' | b'$'))
613                            .count()
614                            + index;
615                        let word = &line[index..end];
616                        javascript.pending_control_paren = !javascript.after_dot
617                            && matches!(word, b"if" | b"while" | b"for" | b"with");
618                        javascript.after_dot = false;
619                        javascript.regex_allowed = matches!(
620                            word,
621                            b"return"
622                                | b"else"
623                                | b"do"
624                                | b"throw"
625                                | b"case"
626                                | b"delete"
627                                | b"typeof"
628                                | b"void"
629                                | b"yield"
630                                | b"await"
631                                | b"instanceof"
632                                | b"in"
633                                | b"of"
634                        );
635                        code = true;
636                        index = end;
637                        continue;
638                    }
639                    if byte == b'(' {
640                        javascript.paren_control.push(javascript.pending_control_paren);
641                        javascript.pending_control_paren = false;
642                        javascript.after_dot = false;
643                        javascript.regex_allowed = true;
644                        code = true;
645                        index += 1;
646                        continue;
647                    }
648                    if byte == b')' {
649                        javascript.regex_allowed = javascript.paren_control.pop().unwrap_or(false);
650                        javascript.pending_control_paren = false;
651                        javascript.after_dot = false;
652                        code = true;
653                        index += 1;
654                        continue;
655                    }
656                    javascript.pending_control_paren = false;
657                    javascript.after_dot = byte == b'.';
658                    javascript.regex_allowed = matches!(
659                        byte,
660                        b'=' | b'('
661                            | b'['
662                            | b'{'
663                            | b':'
664                            | b','
665                            | b';'
666                            | b'!'
667                            | b'?'
668                            | b'&'
669                            | b'|'
670                            | b'+'
671                            | b'-'
672                            | b'*'
673                            | b'%'
674                            | b'^'
675                            | b'~'
676                            | b'<'
677                            | b'>'
678                            | b'/'
679                    );
680                }
681                code = true;
682                whitespace_boundary = syntax.language == Language::Shell
683                    && matches!(byte, b';' | b'&' | b'|' | b'(' | b')' | b'<' | b'>');
684                index += 1;
685            }
686        }
687    }
688
689    match state {
690        State::Quoted { multiline: false, escaped: true, .. }
691            if matches!(syntax.language, Language::C | Language::Cpp) =>
692        {
693            *state =
694                State::Quoted { quote: b'"', escaped: false, multiline: false, doubled: false };
695        }
696        State::Quoted { multiline: false, .. } | State::Regex { .. } => *state = State::Normal,
697        State::Quoted { multiline: true, escaped, .. } => *escaped = false,
698        _ => {}
699    }
700    if code {
701        LineClass::Code
702    } else if comment {
703        LineClass::Comment
704    } else {
705        LineClass::Blank
706    }
707}
708
709fn leading_utf8_character(input: &[u8]) -> Option<(char, usize)> {
710    let width = match *input.first()? {
711        0x00..=0x7f => 1,
712        0xc2..=0xdf => 2,
713        0xe0..=0xef => 3,
714        0xf0..=0xf4 => 4,
715        _ => return None,
716    };
717    let character = std::str::from_utf8(input.get(..width)?).ok()?.chars().next()?;
718    Some((character, width))
719}
720
721fn rust_raw_open(input: &[u8]) -> Option<(u8, usize)> {
722    if input.first() != Some(&b'r') {
723        return None;
724    }
725    let hashes = input[1..].iter().take_while(|byte| **byte == b'#').count();
726    if hashes > usize::from(u8::MAX) || input.get(hashes + 1) != Some(&b'"') {
727        return None;
728    }
729    Some((u8::try_from(hashes).expect("bounded above"), hashes + 2))
730}
731
732fn rust_raw_close(input: &[u8], hashes: u8) -> bool {
733    input.first() == Some(&b'"')
734        && (hashes == 0
735            || input
736                .get(1..=usize::from(hashes))
737                .is_some_and(|tail| tail.iter().all(|byte| *byte == b'#')))
738}
739
740fn rust_lifetime(input: &[u8]) -> bool {
741    let Some(first) = input.get(1) else { return false };
742    if !first.is_ascii_alphabetic() && *first != b'_' {
743        return false;
744    }
745    let name_len =
746        input[1..].iter().take_while(|byte| byte.is_ascii_alphanumeric() || **byte == b'_').count();
747    !(name_len == 1 && input.get(2) == Some(&b'\''))
748}
749
750fn cpp_raw_open(input: &[u8]) -> Option<(Vec<u8>, usize)> {
751    if !input.starts_with(b"R\"") {
752        return None;
753    }
754    let end = input[2..].iter().position(|byte| *byte == b'(')? + 2;
755    let delimiter = &input[2..end];
756    if delimiter.len() > 16
757        || delimiter.iter().any(|b| b.is_ascii_whitespace() || matches!(b, b'\\' | b'(' | b')'))
758    {
759        return None;
760    }
761    let mut close = Vec::with_capacity(delimiter.len() + 2);
762    close.push(b')');
763    close.extend_from_slice(delimiter);
764    close.push(b'"');
765    Some((close, end + 1))
766}
767
768fn sql_dollar_open(input: &[u8]) -> Option<(Vec<u8>, usize)> {
769    if input.first() != Some(&b'$') {
770        return None;
771    }
772    let end = input[1..].iter().position(|byte| *byte == b'$')? + 1;
773    let tag = &input[1..end];
774    if !tag.is_empty()
775        && (!tag[0].is_ascii_alphabetic() && tag[0] != b'_'
776            || tag.iter().any(|b| !b.is_ascii_alphanumeric() && *b != b'_'))
777    {
778        return None;
779    }
780    Some((input[..=end].to_vec(), end + 1))
781}
782
783fn ruby_percent_open(input: &[u8]) -> Option<(u8, u8, usize)> {
784    if input.first() != Some(&b'%') {
785        return None;
786    }
787    let delimiter_index = match input.get(1) {
788        Some(b'q' | b'Q' | b'w' | b'W' | b'i' | b'I' | b'r' | b'x' | b's') => 2,
789        Some(b'{' | b'[' | b'(' | b'<') => 1,
790        _ => return None,
791    };
792    let open = *input.get(delimiter_index)?;
793    let close = match open {
794        b'{' => b'}',
795        b'[' => b']',
796        b'(' => b')',
797        b'<' => b'>',
798        b'/' | b'!' | b'|' => open,
799        _ => return None,
800    };
801    Some((open, close, delimiter_index + 1))
802}
803
804fn heredoc_open(language: Language, input: &[u8]) -> Option<(Vec<u8>, bool, bool)> {
805    let php = language == Language::Php;
806    if !matches!(language, Language::Ruby | Language::Shell | Language::Php) {
807        return None;
808    }
809    let mut tail = if php { input.strip_prefix(b"<<<")? } else { input.strip_prefix(b"<<")? };
810    let indent = if !php && matches!(tail.first(), Some(b'-' | b'~')) {
811        tail = &tail[1..];
812        true
813    } else {
814        false
815    };
816    let quote = if matches!(tail.first(), Some(b'\'' | b'"')) {
817        let quote = tail[0];
818        tail = &tail[1..];
819        Some(quote)
820    } else {
821        None
822    };
823    let width = tail.iter().take_while(|b| b.is_ascii_alphanumeric() || **b == b'_').count();
824    if width == 0 || !tail[0].is_ascii_alphabetic() && tail[0] != b'_' {
825        return None;
826    }
827    if let Some(quote) = quote {
828        if tail.get(width) != Some(&quote) {
829            return None;
830        }
831    }
832    Some((tail[..width].to_vec(), indent, php))
833}
834
835#[cfg(test)]
836mod tests {
837    use super::*;
838
839    type PartitionCase<'a> = (&'a str, &'a [u8], (u64, u64, u64));
840
841    fn count(language: &str, chunks: &[&[u8]]) -> MetricValues {
842        let mut counter = CodeAccumulator::for_type(language).expect("supported language");
843        for chunk in chunks {
844            counter.push(chunk);
845        }
846        counter.finish()
847    }
848
849    #[test]
850    fn partitions_c_like_source_and_counts_mixed_lines_as_code() {
851        let metrics = count(
852            "javascript",
853            &[b"// first\r\nlet url = \"https://example.test\"; // tail\r/* block\n\nend */\n`// text\nmore`;"],
854        );
855        assert_eq!(metrics.physical_lines, 7);
856        assert_eq!(metrics.code_lines, 3);
857        assert_eq!(metrics.comment_lines, 4);
858        assert_eq!(metrics.code_blank_lines, 0);
859    }
860
861    #[test]
862    fn rust_nested_comments_and_raw_strings_ignore_comment_markers() {
863        let source =
864            b"/* outer\n/* inner */\n*/\nlet raw = r##\"/* text */\n// still text\"##;\n\n";
865        let expected = count("rust", &[source]);
866        assert_eq!(expected.physical_lines, 6);
867        assert_eq!(expected.code_lines, 2);
868        assert_eq!(expected.comment_lines, 3);
869        assert_eq!(expected.code_blank_lines, 1);
870        for split in 0..=source.len() {
871            assert_eq!(count("rust", &[&source[..split], &source[split..]]), expected);
872        }
873    }
874
875    #[test]
876    fn triple_quoted_docstrings_are_code_in_v1() {
877        let metrics = count("python", &[b"\"\"\"docs\n# text\n\"\"\"\n# comment\npass\n"]);
878        assert_eq!(metrics.physical_lines, 5);
879        assert_eq!(metrics.code_lines, 4);
880        assert_eq!(metrics.comment_lines, 1);
881        assert_eq!(metrics.code_blank_lines, 0);
882    }
883
884    #[test]
885    fn every_line_ending_convention_has_the_same_partition() {
886        for source in [
887            "// comment\nlet value = 1;\n\n",
888            "// comment\r\nlet value = 1;\r\n\r\n",
889            "// comment\rlet value = 1;\r\r",
890            "// comment\r\nlet value = 1;\r\n",
891        ] {
892            let metrics = count("rust", &[source.as_bytes()]);
893            let expected_lines = if source.ends_with("value = 1;\r\n") { 2 } else { 3 };
894            assert_eq!(metrics.physical_lines, expected_lines, "{source:?}");
895            assert_eq!(metrics.code_lines, 1, "{source:?}");
896            assert_eq!(metrics.comment_lines, 1, "{source:?}");
897            assert_eq!(metrics.code_blank_lines, expected_lines - 2, "{source:?}");
898        }
899    }
900
901    #[test]
902    fn a_leading_utf8_bom_is_not_an_invented_line() {
903        let empty = count("rust", &[b"\xef\xbb\xbf"]);
904        assert_eq!(empty.physical_lines, 0);
905
906        let blank = count("rust", &[b"\xef\xbb\xbf\n"]);
907        assert_eq!(blank.physical_lines, 1);
908        assert_eq!(blank.code_blank_lines, 1);
909    }
910
911    #[test]
912    fn unicode_whitespace_uses_the_basic_analyzers_pinned_table() {
913        let metrics = count("rust", &["\u{3000}\n\u{2003}// comment\n".as_bytes()]);
914        assert_eq!(metrics.physical_lines, 2);
915        assert_eq!(metrics.code_blank_lines, 1);
916        assert_eq!(metrics.comment_lines, 1);
917        assert_eq!(metrics.code_lines, 0);
918
919        let shell = count("shell", &["\u{3000}# comment\n".as_bytes()]);
920        assert_eq!(shell.comment_lines, 1);
921        assert_eq!(shell.code_lines, 0);
922    }
923
924    #[test]
925    fn unsupported_languages_are_explicit() {
926        assert!(CodeAccumulator::for_type("haskell").is_none());
927    }
928
929    #[test]
930    fn rust_multiline_string_and_lifetime_preserve_following_comment_state() {
931        let string = b"const S: &str = \"first\n// text\nlast\";\n";
932        let lifetime = b"fn x<'a>() { /*\ncomment\n*/ }\n";
933        for (source, code, comment) in [(string.as_slice(), 3, 0), (lifetime.as_slice(), 2, 1)] {
934            for split in 0..=source.len() {
935                let metrics = count("rust", &[&source[..split], &source[split..]]);
936                assert_eq!(
937                    (metrics.code_lines, metrics.comment_lines),
938                    (code, comment),
939                    "split {split}"
940                );
941            }
942        }
943    }
944
945    #[test]
946    fn multiline_literal_families_preserve_comment_markers() {
947        let cases: &[PartitionCase<'_>] = &[
948            ("cpp", b"const char *s = R\"tag(first\n// text\nlast)tag\";\n", (3, 0, 0)),
949            ("c", b"const char *s = \"first\\\n// text\";\n", (2, 0, 0)),
950            ("java", b"class C { String s = \"\"\"\n// text\nlast\n\"\"\"; }\n", (4, 0, 0)),
951            ("csharp", b"class C { string s = @\"first\n// text\nlast\"; }\n", (3, 0, 0)),
952            ("csharp", b"class C { string s = \"\"\"\n// text\nlast\n\"\"\"; }\n", (4, 0, 0)),
953            ("ruby", b"s = %q{first\n# text\nlast}\n", (3, 0, 0)),
954            ("ruby", b"s = <<~TEXT\n# text\nlast\nTEXT\n", (4, 0, 0)),
955            ("ruby", b"s = \"first\n# text\nlast\"\n", (3, 0, 0)),
956            ("shell", b"cat <<'TEXT'\n# text\nlast\nTEXT\n", (4, 0, 0)),
957            ("shell", b"value='first\n# text\nlast'\n", (3, 0, 0)),
958            ("sql", b"SELECT $tag$first\n-- text\nlast$tag$;\n", (3, 0, 0)),
959            ("sql", b"SELECT 'first\n-- text\nlast';\n", (3, 0, 0)),
960            ("php", b"<?php\n$s = <<<TEXT\n// text\nlast\nTEXT;\n", (5, 0, 0)),
961        ];
962        for (language, source, expected) in cases {
963            for split in 0..=source.len() {
964                let metrics = count(language, &[&source[..split], &source[split..]]);
965                assert_eq!(
966                    (metrics.code_lines, metrics.comment_lines, metrics.code_blank_lines),
967                    *expected,
968                    "{language} split {split}"
969                );
970            }
971        }
972    }
973
974    #[test]
975    fn multiline_delimiters_restore_comment_recognition_after_closing() {
976        let cases: &[(&str, &[u8], (u64, u64))] = &[
977            ("cpp", b"auto s = R\"x(/*\n// body\n)x\";\n// comment\n", (3, 1)),
978            ("csharp", b"var s = @\"first \"\" quote\n// body\nlast\";\n// comment\n", (3, 1)),
979            ("ruby", b"s = %q{outer {inner\n# body\n}}\n# comment\n", (3, 1)),
980            ("ruby", b"s = <<~TEXT\n# body\n  TEXT\n# comment\n", (3, 1)),
981            ("shell", b"cat <<'TEXT'\n# body\nTEXT\n# comment\n", (3, 1)),
982            ("shell", b"cat <<-TEXT\n TEXT\n# body\nTEXT\n# comment\n", (4, 1)),
983            ("shell", b"cat <<-TEXT\n\tTEXT\n# comment\n", (2, 1)),
984            ("sql", b"SELECT $tag$first\n-- body\nlast$tag$;\n-- comment\n", (3, 1)),
985            ("php", b"$s = <<<TEXT\n// body\nTEXT;\n// comment\n", (3, 1)),
986        ];
987        for (language, source, expected) in cases {
988            let metrics = count(language, &[source]);
989            assert_eq!((metrics.code_lines, metrics.comment_lines), *expected, "{language}");
990        }
991    }
992
993    #[test]
994    fn shell_comments_and_arithmetic_shifts_do_not_hold_lexer_state() {
995        let cases: &[PartitionCase<'_>] = &[
996            // A command separator starts a new shell word, so the unmatched quote is
997            // comment text and cannot turn the next line into a multiline string.
998            ("shell", b"true;# \"unterminated\n# following\nprintf ok\n", (2, 1, 0)),
999            // In arithmetic expansion and arithmetic commands, `<<` shifts a value.
1000            // Neither spelling opens a heredoc that consumes the following comment.
1001            ("shell", b"N=2\nx=$((1<<N))\n# following\n", (2, 1, 0)),
1002            ("shell", b"N=2\n((1<<N))\n# following\n", (2, 1, 0)),
1003            // A hash within a shell word is literal, while a real heredoc body is code.
1004            ("shell", b"printf '%s' foo#bar\ncat <<TEXT\n# literal\nTEXT\n# comment\n", (4, 1, 0)),
1005        ];
1006        for (language, source, expected) in cases {
1007            for split in 0..=source.len() {
1008                let metrics = count(language, &[&source[..split], &source[split..]]);
1009                assert_eq!(
1010                    (metrics.code_lines, metrics.comment_lines, metrics.code_blank_lines),
1011                    *expected,
1012                    "source {source:?}, split {split}"
1013                );
1014            }
1015        }
1016    }
1017
1018    #[test]
1019    fn javascript_regex_and_division_leave_comment_state_correct() {
1020        let cases: &[PartitionCase<'_>] = &[
1021            ("javascript", b"const re = /[/*]/;\nconst answer = 42;\n", (2, 0, 0)),
1022            ("typescript", b"const re = /\\/\\* inside [/] /;\nconst n = 9;\n", (2, 0, 0)),
1023            ("javascript", b"const ratio = total / count;\n/* comment */\nconst re = /[//]/;\n", (2, 1, 0)),
1024            ("typescript", b"return /[/*]/.test(value);\n// comment\nnext();\n", (2, 1, 0)),
1025            ("javascript", b"const quotient = total\n / count; /* real comment */\nconst re = /[/*]/;\nnext();\n", (4, 0, 0)),
1026            ("javascript", b"const text = \"plain\" / count;\nconst re = /[/*]/;\nnext();\n", (3, 0, 0)),
1027            ("javascript", b"if (ok) /[/*]/.test(value);\nconst next = 1;\n", (2, 0, 0)),
1028            ("javascript", b"if (first) {}\nif (ok) /[/*]/.test(value);\nnext();\n", (3, 0, 0)),
1029            ("typescript", b"while (ready && check(x)) /[/*]/.test(value);\nnext();\n", (2, 0, 0)),
1030            ("javascript", b"if (ok &&\n check(x)) /[/*]/.test(value);\nnext();\n", (3, 0, 0)),
1031            ("javascript", b"if (ok) fn(value) / count;\n/* comment */\n", (1, 1, 0)),
1032            ("javascript", b"const ratio = object.if(value) / count;\n/* comment */\n", (1, 1, 0)),
1033        ];
1034        for (language, source, expected) in cases {
1035            for split in 0..=source.len() {
1036                let metrics = count(language, &[&source[..split], &source[split..]]);
1037                assert_eq!(
1038                    (metrics.code_lines, metrics.comment_lines, metrics.code_blank_lines),
1039                    *expected,
1040                    "{language} split {split}"
1041                );
1042            }
1043        }
1044    }
1045}