Skip to main content

fdu_core/content/
content_code_metrics.rs

1//! Streaming common-language source-line classification.
2
3use super::MetricValues;
4
5/// Bytes of the current line the accumulator holds before it classifies them piecewise.
6///
7/// A line shorter than this is classified whole, as every line always was; a longer one
8/// is classified in pieces as it arrives, so what a line costs is this window plus the
9/// longest token a piece cannot yet decide, never the line (fdu-1zb6).
10const LINE_WINDOW_BYTES: usize = 64 * 1024;
11
12/// Streaming `code-sloc-v1` counter for a supported language (analyzer version 3).
13///
14/// The counter retains parser state, the per-line facts the classifier reads, and a
15/// bounded window of the current line: a line is classified in pieces as it arrives,
16/// and a one-line minified or generated source costs a worker the window, not the file.
17/// Mixed code/comment lines are code, blank lines inside block comments are comments,
18/// and multiline string lines are code. Line endings follow the same LF, CRLF, lone-CR,
19/// and unterminated-final-line contract as the basic analyzer.
20#[derive(Debug)]
21pub struct CodeAccumulator {
22    syntax: Syntax,
23    state: State,
24    /// Bytes of the current line not yet classified.
25    window: Vec<u8>,
26    /// Bytes of the current line already classified and dropped from the window.
27    consumed: usize,
28    /// The last byte dropped from the window, which the line's trailing backslash reads.
29    last_dropped: Option<u8>,
30    /// The window length at which the next piece is classified.
31    next_scan_at: usize,
32    /// How far the window grows between piece scans.
33    window_bytes: usize,
34    /// The largest window held, for the bound's own test.
35    #[cfg(test)]
36    peak_window: usize,
37    /// The metrics after each line finished, for the oracle's line-by-line comparison.
38    #[cfg(test)]
39    by_line: Vec<MetricValues>,
40    previous_cr: bool,
41    line: LineScan,
42    javascript: JavaScriptContext,
43    shell: ShellContext,
44    metrics: MetricValues,
45}
46
47#[derive(Debug)]
48struct JavaScriptContext {
49    regex_allowed: bool,
50    pending_control_paren: bool,
51    paren_control: Vec<bool>,
52    after_dot: bool,
53}
54
55#[derive(Debug, Default)]
56struct ShellContext {
57    // The number of unmatched parentheses in a shell arithmetic expression.
58    // `<<` there is a shift operator, not a heredoc opener.
59    arithmetic_parens: usize,
60}
61
62impl Default for JavaScriptContext {
63    fn default() -> Self {
64        Self {
65            regex_allowed: true,
66            pending_control_paren: false,
67            paren_control: Vec::new(),
68            after_dot: false,
69        }
70    }
71}
72
73/// What the classifier knows about the line in progress, carried across its pieces.
74///
75/// These were the locals of a classifier that saw each line whole. A piece scan reads
76/// and leaves them exactly as the whole-line scan would have at that byte.
77#[derive(Debug, Default)]
78#[allow(clippy::struct_excessive_bools)]
79struct LineScan {
80    /// Whether the decisions made at the start of a line have been made.
81    started: bool,
82    /// The line began with a byte-order mark, which is skipped and not scanned.
83    bom: bool,
84    code: bool,
85    comment: bool,
86    whitespace_boundary: bool,
87    /// The rest of the line is dropped unread: after a line comment or a heredoc
88    /// opener, and for a line whose class its start decided.
89    rest_ignored: bool,
90    /// A C string opened in this line: its `multiline` is whether the line ends with a
91    /// backslash, which the whole-line classifier read when the string opened and a
92    /// piece scan can only read at the line's end.
93    c_string_opened: bool,
94    /// The terminator probe while the line is a heredoc body.
95    heredoc: Option<HeredocProbe>,
96}
97
98/// Whether a heredoc body line is its terminator, decided as the line arrives.
99///
100/// The whole-line classifier compared the line, less its indentation, with the
101/// terminator; a piece scan keeps the indentation count and at most the terminator's
102/// length of what follows, and a line already longer than that can never match.
103#[derive(Debug, Default)]
104struct HeredocProbe {
105    indent_done: bool,
106    mismatch: bool,
107    candidate: Vec<u8>,
108}
109
110#[derive(Clone, Copy, Debug)]
111#[allow(clippy::struct_excessive_bools)]
112struct Syntax {
113    language: Language,
114    line_comments: &'static [&'static [u8]],
115    block: Option<BlockSyntax>,
116    nested_blocks: bool,
117    triple_quotes: bool,
118    backtick_strings: bool,
119    rust_raw_strings: bool,
120    shell_hash_boundary: bool,
121    ruby_blocks: bool,
122}
123
124#[derive(Clone, Copy, Debug, PartialEq, Eq)]
125enum Language {
126    Rust,
127    JavaScript,
128    C,
129    Cpp,
130    CSharp,
131    Java,
132    Kotlin,
133    Swift,
134    Go,
135    Php,
136    Python,
137    Ruby,
138    Shell,
139    Sql,
140}
141
142impl Syntax {
143    fn for_type(file_type: &str) -> Option<Self> {
144        let c_like = Self {
145            language: Language::C,
146            line_comments: &[b"//"],
147            block: Some(BlockSyntax { open: b"/*", close: b"*/" }),
148            nested_blocks: false,
149            triple_quotes: false,
150            backtick_strings: false,
151            rust_raw_strings: false,
152            shell_hash_boundary: false,
153            ruby_blocks: false,
154        };
155        match file_type {
156            "rust" => Some(Self {
157                language: Language::Rust,
158                nested_blocks: true,
159                rust_raw_strings: true,
160                ..c_like
161            }),
162            "javascript" | "typescript" => {
163                Some(Self { language: Language::JavaScript, backtick_strings: true, ..c_like })
164            }
165            "go" => Some(Self { language: Language::Go, backtick_strings: true, ..c_like }),
166            "c" => Some(c_like),
167            "cpp" => Some(Self { language: Language::Cpp, ..c_like }),
168            "csharp" => Some(Self { language: Language::CSharp, ..c_like }),
169            "java" => Some(Self { language: Language::Java, ..c_like }),
170            "kotlin" => Some(Self {
171                language: Language::Kotlin,
172                nested_blocks: true,
173                triple_quotes: true,
174                ..c_like
175            }),
176            "swift" => Some(Self {
177                language: Language::Swift,
178                nested_blocks: true,
179                triple_quotes: true,
180                ..c_like
181            }),
182            "php" => {
183                Some(Self { language: Language::Php, line_comments: &[b"//", b"#"], ..c_like })
184            }
185            "python" | "ruby" => Some(Self {
186                language: if file_type == "ruby" { Language::Ruby } else { Language::Python },
187                line_comments: &[b"#"],
188                block: None,
189                nested_blocks: false,
190                triple_quotes: true,
191                backtick_strings: false,
192                rust_raw_strings: false,
193                shell_hash_boundary: false,
194                ruby_blocks: file_type == "ruby",
195            }),
196            "shell" => Some(Self {
197                language: Language::Shell,
198                line_comments: &[b"#"],
199                block: None,
200                nested_blocks: false,
201                triple_quotes: false,
202                backtick_strings: true,
203                rust_raw_strings: false,
204                shell_hash_boundary: true,
205                ruby_blocks: false,
206            }),
207            "sql" => Some(Self {
208                language: Language::Sql,
209                line_comments: &[b"--"],
210                block: Some(BlockSyntax { open: b"/*", close: b"*/" }),
211                nested_blocks: false,
212                triple_quotes: false,
213                backtick_strings: true,
214                rust_raw_strings: false,
215                shell_hash_boundary: false,
216                ruby_blocks: false,
217            }),
218            _ => None,
219        }
220    }
221}
222
223#[derive(Clone, Copy, Debug)]
224struct BlockSyntax {
225    open: &'static [u8],
226    close: &'static [u8],
227}
228
229#[derive(Clone, Debug)]
230enum State {
231    Normal,
232    BlockComment { depth: u16 },
233    Quoted { quote: u8, escaped: bool, multiline: bool, doubled: bool },
234    TripleQuoted { quote: u8, width: usize },
235    RustRaw { hashes: u8 },
236    Delimited { close: Vec<u8> },
237    Heredoc { terminator: Vec<u8>, indent: bool, php: bool },
238    RubyPercent { open: u8, close: u8, depth: usize },
239    Regex { escaped: bool, in_class: bool },
240    RubyBlock,
241}
242
243const UTF8_BOM: &[u8] = &[0xef, 0xbb, 0xbf];
244
245impl CodeAccumulator {
246    /// Create a counter for one stable file-type ID, or return `None` when
247    /// `code-sloc-v1` does not claim that language.
248    pub fn for_type(file_type: &str) -> Option<Self> {
249        Self::with_window_bytes(file_type, LINE_WINDOW_BYTES)
250    }
251
252    /// [`Self::for_type`] with the window a line grows to before a piece is classified.
253    fn with_window_bytes(file_type: &str, window_bytes: usize) -> Option<Self> {
254        let syntax = Syntax::for_type(file_type)?;
255        let window_bytes = window_bytes.max(1);
256        Some(Self {
257            syntax,
258            state: State::Normal,
259            window: Vec::new(),
260            consumed: 0,
261            last_dropped: None,
262            next_scan_at: window_bytes,
263            window_bytes,
264            #[cfg(test)]
265            peak_window: 0,
266            #[cfg(test)]
267            by_line: Vec::new(),
268            previous_cr: false,
269            line: LineScan::default(),
270            javascript: JavaScriptContext::default(),
271            shell: ShellContext::default(),
272            metrics: MetricValues::default(),
273        })
274    }
275
276    /// Consume an arbitrary byte chunk.
277    pub fn push(&mut self, chunk: &[u8]) {
278        for &byte in chunk {
279            if self.previous_cr {
280                self.previous_cr = false;
281                if byte == b'\n' {
282                    continue;
283                }
284            }
285            match byte {
286                b'\r' => {
287                    self.finish_line();
288                    self.previous_cr = true;
289                }
290                b'\n' => self.finish_line(),
291                _ => {
292                    self.window.push(byte);
293                    if self.window.len() >= self.next_scan_at {
294                        self.scan(false);
295                    }
296                }
297            }
298        }
299    }
300
301    /// Finish an unterminated final line and return its additive metrics.
302    pub fn finish(mut self) -> MetricValues {
303        self.finish_unterminated();
304        self.metrics
305    }
306
307    fn finish_unterminated(&mut self) {
308        let bom_only = self.consumed == 0 && self.window == UTF8_BOM;
309        if self.consumed + self.window.len() > 0 && !bom_only {
310            self.finish_line();
311        }
312    }
313
314    /// The largest window the accumulator has held, in bytes.
315    #[cfg(test)]
316    fn peak_window(&self) -> usize {
317        self.peak_window
318    }
319
320    /// [`Self::finish`], with the metrics as they stood after each line finished.
321    #[cfg(test)]
322    fn finish_by_line(mut self) -> (MetricValues, Vec<MetricValues>) {
323        self.finish_unterminated();
324        (self.metrics, self.by_line)
325    }
326
327    fn finish_line(&mut self) {
328        self.scan(true);
329        let ends_with_backslash = self.window.last().copied().or(self.last_dropped) == Some(b'\\');
330        if self.syntax.language == Language::C && self.line.c_string_opened {
331            if let State::Quoted { quote: b'"', multiline, .. } = &mut self.state {
332                *multiline = ends_with_backslash;
333            }
334        }
335        match &mut self.state {
336            State::Quoted { multiline: false, escaped: true, .. }
337                if matches!(self.syntax.language, Language::C | Language::Cpp) =>
338            {
339                self.state =
340                    State::Quoted { quote: b'"', escaped: false, multiline: false, doubled: false };
341            }
342            State::Quoted { multiline: false, .. } | State::Regex { .. } => {
343                self.state = State::Normal;
344            }
345            State::Quoted { multiline: true, escaped, .. } => *escaped = false,
346            _ => {}
347        }
348        self.metrics.physical_lines = self.metrics.physical_lines.saturating_add(1);
349        if self.line.code {
350            self.metrics.code_lines = self.metrics.code_lines.saturating_add(1);
351        } else if self.line.comment {
352            self.metrics.comment_lines = self.metrics.comment_lines.saturating_add(1);
353        } else {
354            self.metrics.code_blank_lines = self.metrics.code_blank_lines.saturating_add(1);
355        }
356        #[cfg(test)]
357        self.by_line.push(self.metrics);
358        self.window.clear();
359        self.consumed = 0;
360        self.last_dropped = None;
361        self.next_scan_at = self.window_bytes;
362        self.line = LineScan::default();
363    }
364
365    /// Classify what the window holds, as far as it can be decided.
366    ///
367    /// `final_piece` says the line ends here, so every site decides on what there is,
368    /// as the whole-line classifier did; otherwise a site that would read past the
369    /// window waits for more bytes.
370    fn scan(&mut self, final_piece: bool) {
371        #[cfg(test)]
372        {
373            self.peak_window = self.peak_window.max(self.window.len());
374        }
375        if !self.line.started && !self.start_line(final_piece) {
376            self.next_scan_at = self.window.len().saturating_mul(2).max(self.window_bytes);
377            return;
378        }
379        if self.line.rest_ignored {
380            self.drop_window();
381        } else if self.line.heredoc.is_some() {
382            self.probe_heredoc(final_piece);
383        } else {
384            let piece = scan_piece(
385                self.syntax,
386                &mut self.state,
387                &mut self.javascript,
388                &mut self.shell,
389                &mut self.line,
390                &self.window,
391                final_piece,
392            );
393            let stalled = piece.consumed == 0 && !final_piece;
394            self.drop_consumed(piece.consumed);
395            if piece.rest_ignored {
396                self.line.rest_ignored = true;
397                self.drop_window();
398            }
399            if stalled {
400                // A token the window's edge cuts, such as a run of raw-string hashes:
401                // rescan after the window has grown, not after every byte.
402                self.next_scan_at = self.window.len().saturating_mul(2).max(self.window_bytes);
403                return;
404            }
405        }
406        self.next_scan_at = self.window.len().saturating_add(self.window_bytes);
407    }
408
409    /// The decisions the whole-line classifier made before scanning bytes: the line's
410    /// flags, a heredoc body, a Ruby block line, and a leading byte-order mark.
411    ///
412    /// Returns `false` when they need more of the line than the window holds.
413    fn start_line(&mut self, final_piece: bool) -> bool {
414        // `=begin` is the longest of the line-start patterns.
415        if !final_piece && self.window.len() < 6 {
416            return false;
417        }
418        self.line.started = true;
419        self.line.whitespace_boundary = matches!(self.state, State::Normal);
420        self.line.code = matches!(
421            self.state,
422            State::Quoted { .. }
423                | State::TripleQuoted { .. }
424                | State::RustRaw { .. }
425                | State::Delimited { .. }
426                | State::Heredoc { .. }
427                | State::RubyPercent { .. }
428                | State::Regex { .. }
429        );
430        self.line.comment = matches!(self.state, State::BlockComment { .. });
431        if matches!(self.state, State::Heredoc { .. }) {
432            self.line.heredoc = Some(HeredocProbe::default());
433            return true;
434        }
435        if matches!(self.state, State::RubyBlock) {
436            if self.window.starts_with(b"=end") {
437                self.state = State::Normal;
438            }
439            self.line.comment = true;
440            self.line.rest_ignored = true;
441            return true;
442        }
443        if self.syntax.ruby_blocks
444            && matches!(self.state, State::Normal)
445            && self.window.starts_with(b"=begin")
446        {
447            self.state = State::RubyBlock;
448            self.line.comment = true;
449            self.line.rest_ignored = true;
450            return true;
451        }
452        if self.window.starts_with(UTF8_BOM) {
453            self.line.bom = true;
454            self.drop_consumed(UTF8_BOM.len());
455        }
456        true
457    }
458
459    /// Feed the window to the heredoc terminator probe and drop it.
460    fn probe_heredoc(&mut self, final_piece: bool) {
461        let State::Heredoc { terminator, indent, php } = &self.state else {
462            unreachable!("the probe runs in heredoc state only");
463        };
464        let (terminator, indent, php) = (terminator.clone(), *indent, *php);
465        let shell = self.syntax.language == Language::Shell;
466        let probe = self.line.heredoc.as_mut().expect("a heredoc line has its probe");
467        let mut rest: &[u8] = &self.window;
468        if indent && !probe.indent_done {
469            let indentation = rest
470                .iter()
471                .take_while(|byte| if shell { **byte == b'\t' } else { byte.is_ascii_whitespace() })
472                .count();
473            rest = &rest[indentation..];
474            if !rest.is_empty() || final_piece {
475                probe.indent_done = true;
476            }
477        }
478        let longest = terminator.len() + usize::from(php);
479        if !probe.mismatch {
480            if probe.candidate.len() + rest.len() > longest {
481                probe.mismatch = true;
482                probe.candidate = Vec::new();
483            } else {
484                probe.candidate.extend_from_slice(rest);
485            }
486        }
487        if final_piece
488            && !probe.mismatch
489            && (probe.candidate == terminator
490                || (php && probe.candidate.strip_suffix(b";") == Some(terminator.as_slice())))
491        {
492            self.state = State::Normal;
493        }
494        self.drop_window();
495    }
496
497    fn drop_window(&mut self) {
498        self.drop_consumed(self.window.len());
499    }
500
501    fn drop_consumed(&mut self, count: usize) {
502        if count == 0 {
503            return;
504        }
505        self.last_dropped = Some(self.window[count - 1]);
506        self.consumed += count;
507        self.window.drain(..count);
508    }
509}
510
511/// What a piece scan came to.
512struct PieceScan {
513    /// Bytes of the window the scan decided; the rest wait for more bytes or the end.
514    consumed: usize,
515    /// The rest of the line is dropped unread: a line comment or a heredoc opener.
516    rest_ignored: bool,
517}
518
519/// Whether `rest` is a proper prefix of `token`: more bytes could complete it.
520fn prefix_of(rest: &[u8], token: &[u8]) -> bool {
521    rest.len() < token.len() && token.starts_with(rest)
522}
523
524fn is_tag_byte(byte: u8) -> bool {
525    byte.is_ascii_alphanumeric() || byte == b'_'
526}
527
528/// Whether `rest`, cut by the window's edge, could still become a C++ raw-string opener.
529fn cpp_raw_could_continue(rest: &[u8]) -> bool {
530    if prefix_of(rest, b"R\"") {
531        return true;
532    }
533    let Some(delimiter) = rest.strip_prefix(b"R\"") else { return false };
534    delimiter.len() <= 16
535        && !delimiter.iter().any(|b| b.is_ascii_whitespace() || matches!(b, b'\\' | b'(' | b')'))
536}
537
538/// Whether `rest`, cut by the window's edge, could still become a SQL dollar-quote opener.
539fn sql_dollar_could_continue(rest: &[u8]) -> bool {
540    let Some(tag) = rest.strip_prefix(b"$") else { return false };
541    tag.first().is_none_or(|first| first.is_ascii_alphabetic() || *first == b'_')
542        && tag.iter().all(|byte| is_tag_byte(*byte))
543}
544
545/// Whether `rest`, cut by the window's edge, could still become a heredoc opener.
546fn heredoc_could_continue(language: Language, rest: &[u8]) -> bool {
547    let php = language == Language::Php;
548    if !matches!(language, Language::Ruby | Language::Shell | Language::Php) {
549        return false;
550    }
551    let opener: &[u8] = if php { b"<<<" } else { b"<<" };
552    if prefix_of(rest, opener) {
553        return true;
554    }
555    let Some(mut tail) = rest.strip_prefix(opener) else { return false };
556    if !php && matches!(tail.first(), Some(b'-' | b'~')) {
557        tail = &tail[1..];
558    }
559    if matches!(tail.first(), Some(b'\'' | b'"')) {
560        tail = &tail[1..];
561    }
562    tail.iter().all(|byte| is_tag_byte(*byte))
563}
564
565/// Classify the bytes `window` holds of the current line, from its start, as far as
566/// they can be decided.
567///
568/// This is the whole-line classifier's loop, with one addition at every site that
569/// looks ahead: when the bytes it would read end at the window's edge and the line goes
570/// on (`final_piece` false), it stops and the scan resumes there with more bytes. At the
571/// line's end every site decides on what there is, so the decisions are the ones the
572/// whole line gives.
573fn scan_piece(
574    syntax: Syntax,
575    state: &mut State,
576    javascript: &mut JavaScriptContext,
577    shell: &mut ShellContext,
578    line: &mut LineScan,
579    window: &[u8],
580    final_piece: bool,
581) -> PieceScan {
582    let mut index = 0;
583    let mut rest_ignored = false;
584    // Whether a site needing `need` bytes must wait for them.
585    let short = |index: usize, need: usize| !final_piece && window.len() - index < need;
586
587    'scan: while index < window.len() {
588        let rest = &window[index..];
589        if let State::Delimited { close } = state {
590            line.code = true;
591            if rest.starts_with(close) {
592                index += close.len();
593                *state = State::Normal;
594            } else if !final_piece && prefix_of(rest, close) {
595                break;
596            } else {
597                index += 1;
598            }
599            continue;
600        }
601        match state.clone() {
602            State::BlockComment { mut depth } => {
603                line.comment = true;
604                let block = syntax.block.expect("block state requires block syntax");
605                if syntax.nested_blocks && rest.starts_with(block.open) {
606                    depth = depth.saturating_add(1);
607                    *state = State::BlockComment { depth };
608                    index += block.open.len();
609                } else if rest.starts_with(block.close) {
610                    depth = depth.saturating_sub(1);
611                    *state = if depth == 0 { State::Normal } else { State::BlockComment { depth } };
612                    index += block.close.len();
613                } else if !final_piece
614                    && (prefix_of(rest, block.close)
615                        || (syntax.nested_blocks && prefix_of(rest, block.open)))
616                {
617                    break;
618                } else {
619                    index += 1;
620                }
621            }
622            State::Quoted { quote, mut escaped, multiline, doubled } => {
623                line.code = true;
624                let byte = rest[0];
625                if escaped {
626                    escaped = false;
627                } else if byte == b'\\' {
628                    escaped = true;
629                } else if byte == quote {
630                    if doubled {
631                        if short(index, 2) {
632                            break;
633                        }
634                        if rest.get(1) == Some(&quote) {
635                            index += 2;
636                            continue;
637                        }
638                    }
639                    *state = State::Normal;
640                    if syntax.language == Language::JavaScript {
641                        javascript.regex_allowed = false;
642                    }
643                    index += 1;
644                    continue;
645                }
646                *state = State::Quoted { quote, escaped, multiline, doubled };
647                index += 1;
648            }
649            State::TripleQuoted { quote, width } => {
650                line.code = true;
651                if rest.len() >= width && rest[..width].iter().all(|b| *b == quote) {
652                    *state = State::Normal;
653                    if syntax.language == Language::JavaScript {
654                        javascript.regex_allowed = false;
655                    }
656                    index += width;
657                } else if !final_piece && rest.len() < width && rest.iter().all(|b| *b == quote) {
658                    break;
659                } else {
660                    index += 1;
661                }
662            }
663            State::RustRaw { hashes } => {
664                line.code = true;
665                if rust_raw_close(rest, hashes) {
666                    *state = State::Normal;
667                    index += usize::from(hashes) + 1;
668                } else if !final_piece
669                    && rest.len() < usize::from(hashes) + 1
670                    && rest.first() == Some(&b'"')
671                    && rest[1..].iter().all(|b| *b == b'#')
672                {
673                    break;
674                } else {
675                    index += 1;
676                }
677            }
678            State::Delimited { .. } => {
679                unreachable!("delimited strings are handled before matching")
680            }
681            State::RubyPercent { open, close, mut depth } => {
682                line.code = true;
683                match rest[0] {
684                    b'\\' => {
685                        if short(index, 2) {
686                            break;
687                        }
688                        index += usize::min(2, rest.len());
689                    }
690                    byte if byte == open && open != close => {
691                        depth += 1;
692                        *state = State::RubyPercent { open, close, depth };
693                        index += 1;
694                    }
695                    byte if byte == close => {
696                        depth -= 1;
697                        *state = if depth == 0 {
698                            State::Normal
699                        } else {
700                            State::RubyPercent { open, close, depth }
701                        };
702                        index += 1;
703                    }
704                    _ => index += 1,
705                }
706            }
707            State::Regex { mut escaped, mut in_class } => {
708                line.code = true;
709                let byte = rest[0];
710                if escaped {
711                    escaped = false;
712                } else if byte == b'\\' {
713                    escaped = true;
714                } else if byte == b'[' {
715                    in_class = true;
716                } else if byte == b']' {
717                    in_class = false;
718                } else if byte == b'/' && !in_class {
719                    *state = State::Normal;
720                    index += 1;
721                    javascript.regex_allowed = false;
722                    continue;
723                }
724                *state = State::Regex { escaped, in_class };
725                index += 1;
726            }
727            State::Heredoc { .. } => unreachable!("heredoc lines are probed, not scanned"),
728            State::RubyBlock => unreachable!("Ruby block lines are decided at their start"),
729            State::Normal => {
730                let byte = rest[0];
731                if byte.is_ascii_whitespace() {
732                    line.whitespace_boundary = true;
733                    index += 1;
734                    continue;
735                }
736                let utf8_width = match byte {
737                    0xc2..=0xdf => Some(2),
738                    0xe0..=0xef => Some(3),
739                    0xf0..=0xf4 => Some(4),
740                    _ => None,
741                };
742                if let Some(width) = utf8_width {
743                    if short(index, width) {
744                        break;
745                    }
746                    if let Some((character, width)) = leading_utf8_character(rest) {
747                        if super::content_basic_metrics::is_content_whitespace(character) {
748                            line.whitespace_boundary = true;
749                            index += width;
750                            continue;
751                        }
752                    }
753                }
754                if syntax.language == Language::Shell {
755                    if shell.arithmetic_parens == 0 {
756                        if rest.starts_with(b"$((") {
757                            shell.arithmetic_parens = 2;
758                            line.code = true;
759                            line.whitespace_boundary = false;
760                            index += 3;
761                            continue;
762                        }
763                        if !final_piece && prefix_of(rest, b"$((") {
764                            break;
765                        }
766                    }
767                    if shell.arithmetic_parens == 0 && line.whitespace_boundary {
768                        if rest.starts_with(b"((") {
769                            shell.arithmetic_parens = 2;
770                            line.code = true;
771                            line.whitespace_boundary = false;
772                            index += 2;
773                            continue;
774                        }
775                        if !final_piece && prefix_of(rest, b"((") {
776                            break;
777                        }
778                    }
779                    if shell.arithmetic_parens > 0 {
780                        match byte {
781                            b'(' => {
782                                shell.arithmetic_parens = shell.arithmetic_parens.saturating_add(1);
783                            }
784                            b')' => shell.arithmetic_parens -= 1,
785                            _ => {}
786                        }
787                    }
788                }
789                if syntax.language == Language::JavaScript
790                    && byte == b'/'
791                    && javascript.regex_allowed
792                {
793                    if short(index, 2) {
794                        break;
795                    }
796                    if !rest.starts_with(b"//") && !rest.starts_with(b"/*") {
797                        line.code = true;
798                        line.whitespace_boundary = false;
799                        *state = State::Regex { escaped: false, in_class: false };
800                        index += 1;
801                        continue;
802                    }
803                }
804                let marker_admitted = !syntax.shell_hash_boundary || line.whitespace_boundary;
805                if syntax
806                    .line_comments
807                    .iter()
808                    .any(|marker| rest.starts_with(marker) && marker_admitted)
809                {
810                    line.comment = true;
811                    rest_ignored = true;
812                    index = window.len();
813                    break 'scan;
814                }
815                if !final_piece
816                    && syntax
817                        .line_comments
818                        .iter()
819                        .any(|marker| prefix_of(rest, marker) && marker_admitted)
820                {
821                    break;
822                }
823                if let Some(block) = syntax.block {
824                    if rest.starts_with(block.open) {
825                        line.comment = true;
826                        line.whitespace_boundary = false;
827                        *state = State::BlockComment { depth: 1 };
828                        index += block.open.len();
829                        continue;
830                    }
831                    if !final_piece && prefix_of(rest, block.open) {
832                        break;
833                    }
834                }
835                if syntax.rust_raw_strings {
836                    if let Some((hashes, consumed)) = rust_raw_open(rest) {
837                        line.code = true;
838                        line.whitespace_boundary = false;
839                        *state = State::RustRaw { hashes };
840                        index += consumed;
841                        continue;
842                    }
843                    if !final_piece && byte == b'r' && rest[1..].iter().all(|b| *b == b'#') {
844                        break;
845                    }
846                }
847                if syntax.language == Language::Cpp {
848                    if let Some((close, consumed)) = cpp_raw_open(rest) {
849                        line.code = true;
850                        *state = State::Delimited { close };
851                        index += consumed;
852                        continue;
853                    }
854                    if !final_piece && cpp_raw_could_continue(rest) {
855                        break;
856                    }
857                }
858                if syntax.language == Language::Sql {
859                    if let Some((close, consumed)) = sql_dollar_open(rest) {
860                        line.code = true;
861                        *state = State::Delimited { close };
862                        index += consumed;
863                        continue;
864                    }
865                    if !final_piece && sql_dollar_could_continue(rest) {
866                        break;
867                    }
868                }
869                if syntax.language == Language::Ruby {
870                    if let Some((open, close, consumed)) = ruby_percent_open(rest) {
871                        line.code = true;
872                        *state = State::RubyPercent { open, close, depth: 1 };
873                        index += consumed;
874                        continue;
875                    }
876                    if !final_piece && byte == b'%' && rest.len() < 3 {
877                        break;
878                    }
879                }
880                if syntax.language != Language::Shell || shell.arithmetic_parens == 0 {
881                    // Before the opener is read: a tag the window's edge cuts would read
882                    // as a shorter tag.
883                    if !final_piece && heredoc_could_continue(syntax.language, rest) {
884                        break;
885                    }
886                    if let Some((terminator, indent, php)) = heredoc_open(syntax.language, rest) {
887                        line.code = true;
888                        *state = State::Heredoc { terminator, indent, php };
889                        rest_ignored = true;
890                        index = window.len();
891                        break 'scan;
892                    }
893                }
894                if syntax.language == Language::CSharp {
895                    if rest.starts_with(b"@\"") {
896                        line.code = true;
897                        *state = State::Quoted {
898                            quote: b'"',
899                            escaped: false,
900                            multiline: true,
901                            doubled: true,
902                        };
903                        index += 2;
904                        continue;
905                    }
906                    if !final_piece && rest == b"@" {
907                        break;
908                    }
909                }
910                if matches!(syntax.language, Language::CSharp | Language::Java) && byte == b'"' {
911                    if short(index, 3) {
912                        break;
913                    }
914                    if rest.starts_with(b"\"\"\"") {
915                        let run = rest.iter().take_while(|b| **b == b'"').count();
916                        if syntax.language == Language::CSharp && !final_piece && run == rest.len()
917                        {
918                            break;
919                        }
920                        line.code = true;
921                        let width = if syntax.language == Language::CSharp { run } else { 3 };
922                        *state = State::TripleQuoted { quote: b'"', width };
923                        index += width;
924                        continue;
925                    }
926                }
927                if syntax.triple_quotes && matches!(byte, b'\'' | b'"') {
928                    if short(index, 3) {
929                        break;
930                    }
931                    if rest.starts_with(&[byte, byte, byte]) {
932                        line.code = true;
933                        line.whitespace_boundary = false;
934                        *state = State::TripleQuoted { quote: byte, width: 3 };
935                        index += 3;
936                        continue;
937                    }
938                }
939                if matches!(byte, b'\'' | b'"') || (syntax.backtick_strings && byte == b'`') {
940                    if syntax.language == Language::Rust && byte == b'\'' {
941                        if short(index, 3) {
942                            break;
943                        }
944                        if rust_lifetime(rest) {
945                            line.code = true;
946                            index += 1;
947                            continue;
948                        }
949                    }
950                    line.code = true;
951                    line.whitespace_boundary = false;
952                    // A C string's `multiline` is whether the line ends with a backslash,
953                    // which the line's end supplies (`c_string_opened`).
954                    let c_string = syntax.language == Language::C && byte == b'"';
955                    line.c_string_opened |= c_string;
956                    let multiline = byte == b'`'
957                        || (byte == b'"' && syntax.language == Language::Rust)
958                        || matches!(
959                            syntax.language,
960                            Language::Ruby | Language::Shell | Language::Sql
961                        );
962                    *state = State::Quoted {
963                        quote: byte,
964                        escaped: false,
965                        multiline,
966                        doubled: syntax.language == Language::Sql,
967                    };
968                    index += 1;
969                    continue;
970                }
971                if syntax.language == Language::JavaScript {
972                    if byte.is_ascii_alphanumeric() || matches!(byte, b'_' | b'$') {
973                        let run = rest
974                            .iter()
975                            .take_while(|b| b.is_ascii_alphanumeric() || matches!(b, b'_' | b'$'))
976                            .count();
977                        if !final_piece && run == rest.len() {
978                            break;
979                        }
980                        let word = &rest[..run];
981                        javascript.pending_control_paren = !javascript.after_dot
982                            && matches!(word, b"if" | b"while" | b"for" | b"with");
983                        javascript.after_dot = false;
984                        javascript.regex_allowed = matches!(
985                            word,
986                            b"return"
987                                | b"else"
988                                | b"do"
989                                | b"throw"
990                                | b"case"
991                                | b"delete"
992                                | b"typeof"
993                                | b"void"
994                                | b"yield"
995                                | b"await"
996                                | b"instanceof"
997                                | b"in"
998                                | b"of"
999                        );
1000                        line.code = true;
1001                        index += run;
1002                        continue;
1003                    }
1004                    if byte == b'(' {
1005                        javascript.paren_control.push(javascript.pending_control_paren);
1006                        javascript.pending_control_paren = false;
1007                        javascript.after_dot = false;
1008                        javascript.regex_allowed = true;
1009                        line.code = true;
1010                        index += 1;
1011                        continue;
1012                    }
1013                    if byte == b')' {
1014                        javascript.regex_allowed = javascript.paren_control.pop().unwrap_or(false);
1015                        javascript.pending_control_paren = false;
1016                        javascript.after_dot = false;
1017                        line.code = true;
1018                        index += 1;
1019                        continue;
1020                    }
1021                    javascript.pending_control_paren = false;
1022                    javascript.after_dot = byte == b'.';
1023                    javascript.regex_allowed = matches!(
1024                        byte,
1025                        b'=' | b'('
1026                            | b'['
1027                            | b'{'
1028                            | b':'
1029                            | b','
1030                            | b';'
1031                            | b'!'
1032                            | b'?'
1033                            | b'&'
1034                            | b'|'
1035                            | b'+'
1036                            | b'-'
1037                            | b'*'
1038                            | b'%'
1039                            | b'^'
1040                            | b'~'
1041                            | b'<'
1042                            | b'>'
1043                            | b'/'
1044                    );
1045                }
1046                line.code = true;
1047                line.whitespace_boundary = syntax.language == Language::Shell
1048                    && matches!(byte, b';' | b'&' | b'|' | b'(' | b')' | b'<' | b'>');
1049                index += 1;
1050            }
1051        }
1052    }
1053
1054    PieceScan { consumed: index, rest_ignored }
1055}
1056
1057fn leading_utf8_character(input: &[u8]) -> Option<(char, usize)> {
1058    let width = match *input.first()? {
1059        0x00..=0x7f => 1,
1060        0xc2..=0xdf => 2,
1061        0xe0..=0xef => 3,
1062        0xf0..=0xf4 => 4,
1063        _ => return None,
1064    };
1065    let character = std::str::from_utf8(input.get(..width)?).ok()?.chars().next()?;
1066    Some((character, width))
1067}
1068
1069fn rust_raw_open(input: &[u8]) -> Option<(u8, usize)> {
1070    if input.first() != Some(&b'r') {
1071        return None;
1072    }
1073    let hashes = input[1..].iter().take_while(|byte| **byte == b'#').count();
1074    if hashes > usize::from(u8::MAX) || input.get(hashes + 1) != Some(&b'"') {
1075        return None;
1076    }
1077    Some((u8::try_from(hashes).expect("bounded above"), hashes + 2))
1078}
1079
1080fn rust_raw_close(input: &[u8], hashes: u8) -> bool {
1081    input.first() == Some(&b'"')
1082        && (hashes == 0
1083            || input
1084                .get(1..=usize::from(hashes))
1085                .is_some_and(|tail| tail.iter().all(|byte| *byte == b'#')))
1086}
1087
1088fn rust_lifetime(input: &[u8]) -> bool {
1089    let Some(first) = input.get(1) else { return false };
1090    if !first.is_ascii_alphabetic() && *first != b'_' {
1091        return false;
1092    }
1093    let name_len =
1094        input[1..].iter().take_while(|byte| byte.is_ascii_alphanumeric() || **byte == b'_').count();
1095    !(name_len == 1 && input.get(2) == Some(&b'\''))
1096}
1097
1098fn cpp_raw_open(input: &[u8]) -> Option<(Vec<u8>, usize)> {
1099    if !input.starts_with(b"R\"") {
1100        return None;
1101    }
1102    let end = input[2..].iter().position(|byte| *byte == b'(')? + 2;
1103    let delimiter = &input[2..end];
1104    if delimiter.len() > 16
1105        || delimiter.iter().any(|b| b.is_ascii_whitespace() || matches!(b, b'\\' | b'(' | b')'))
1106    {
1107        return None;
1108    }
1109    let mut close = Vec::with_capacity(delimiter.len() + 2);
1110    close.push(b')');
1111    close.extend_from_slice(delimiter);
1112    close.push(b'"');
1113    Some((close, end + 1))
1114}
1115
1116fn sql_dollar_open(input: &[u8]) -> Option<(Vec<u8>, usize)> {
1117    if input.first() != Some(&b'$') {
1118        return None;
1119    }
1120    let end = input[1..].iter().position(|byte| *byte == b'$')? + 1;
1121    let tag = &input[1..end];
1122    if !tag.is_empty()
1123        && (!tag[0].is_ascii_alphabetic() && tag[0] != b'_'
1124            || tag.iter().any(|b| !b.is_ascii_alphanumeric() && *b != b'_'))
1125    {
1126        return None;
1127    }
1128    Some((input[..=end].to_vec(), end + 1))
1129}
1130
1131fn ruby_percent_open(input: &[u8]) -> Option<(u8, u8, usize)> {
1132    if input.first() != Some(&b'%') {
1133        return None;
1134    }
1135    let delimiter_index = match input.get(1) {
1136        Some(b'q' | b'Q' | b'w' | b'W' | b'i' | b'I' | b'r' | b'x' | b's') => 2,
1137        Some(b'{' | b'[' | b'(' | b'<') => 1,
1138        _ => return None,
1139    };
1140    let open = *input.get(delimiter_index)?;
1141    let close = match open {
1142        b'{' => b'}',
1143        b'[' => b']',
1144        b'(' => b')',
1145        b'<' => b'>',
1146        b'/' | b'!' | b'|' => open,
1147        _ => return None,
1148    };
1149    Some((open, close, delimiter_index + 1))
1150}
1151
1152fn heredoc_open(language: Language, input: &[u8]) -> Option<(Vec<u8>, bool, bool)> {
1153    let php = language == Language::Php;
1154    if !matches!(language, Language::Ruby | Language::Shell | Language::Php) {
1155        return None;
1156    }
1157    let mut tail = if php { input.strip_prefix(b"<<<")? } else { input.strip_prefix(b"<<")? };
1158    let indent = if !php && matches!(tail.first(), Some(b'-' | b'~')) {
1159        tail = &tail[1..];
1160        true
1161    } else {
1162        false
1163    };
1164    let quote = if matches!(tail.first(), Some(b'\'' | b'"')) {
1165        let quote = tail[0];
1166        tail = &tail[1..];
1167        Some(quote)
1168    } else {
1169        None
1170    };
1171    let width = tail.iter().take_while(|b| b.is_ascii_alphanumeric() || **b == b'_').count();
1172    if width == 0 || !tail[0].is_ascii_alphabetic() && tail[0] != b'_' {
1173        return None;
1174    }
1175    if let Some(quote) = quote {
1176        if tail.get(width) != Some(&quote) {
1177            return None;
1178        }
1179    }
1180    Some((tail[..width].to_vec(), indent, php))
1181}
1182
1183#[cfg(test)]
1184mod tests {
1185    use super::*;
1186
1187    /// The whole-line classifier the streaming one replaced, verbatim, as the oracle
1188    /// every streaming answer is held to.
1189    #[allow(dead_code, clippy::pedantic)]
1190    mod reference {
1191        use super::super::{
1192            JavaScriptContext, Language, MetricValues, ShellContext, State, Syntax, cpp_raw_open,
1193            heredoc_open, leading_utf8_character, ruby_percent_open, rust_lifetime, rust_raw_close,
1194            rust_raw_open, sql_dollar_open,
1195        };
1196
1197        /// Streaming `code-sloc-v1` counter for a supported language (analyzer version 3).
1198        ///
1199        /// The counter retains the current logical line, its allocated capacity, and parser
1200        /// state. A one-line minified or generated source can therefore require file-sized
1201        /// memory per active worker. Mixed
1202        /// code/comment lines are code, blank lines inside block comments are comments, and
1203        /// multiline string lines are code. Line endings follow the same LF, CRLF, lone-CR,
1204        /// and unterminated-final-line contract as the basic analyzer.
1205        #[derive(Debug)]
1206        pub(super) struct WholeLineAccumulator {
1207            syntax: Syntax,
1208            state: State,
1209            line: Vec<u8>,
1210            previous_cr: bool,
1211            javascript: JavaScriptContext,
1212            shell: ShellContext,
1213            metrics: MetricValues,
1214        }
1215
1216        impl WholeLineAccumulator {
1217            /// Create a counter for one stable file-type ID, or return `None` when
1218            /// `code-sloc-v1` does not claim that language.
1219            pub(super) fn for_type(file_type: &str) -> Option<Self> {
1220                let syntax = Syntax::for_type(file_type)?;
1221                Some(Self {
1222                    syntax,
1223                    state: State::Normal,
1224                    line: Vec::new(),
1225                    previous_cr: false,
1226                    javascript: JavaScriptContext::default(),
1227                    shell: ShellContext::default(),
1228                    metrics: MetricValues::default(),
1229                })
1230            }
1231
1232            /// Consume an arbitrary byte chunk.
1233            pub(super) fn push(&mut self, chunk: &[u8]) {
1234                for &byte in chunk {
1235                    if self.previous_cr {
1236                        self.previous_cr = false;
1237                        if byte == b'\n' {
1238                            continue;
1239                        }
1240                    }
1241                    match byte {
1242                        b'\r' => {
1243                            self.finish_line();
1244                            self.previous_cr = true;
1245                        }
1246                        b'\n' => self.finish_line(),
1247                        _ => self.line.push(byte),
1248                    }
1249                }
1250            }
1251
1252            /// Finish an unterminated final line and return its additive metrics.
1253            pub(super) fn finish(mut self) -> MetricValues {
1254                if !self.line.is_empty() && self.line.as_slice() != [0xef, 0xbb, 0xbf] {
1255                    self.finish_line();
1256                }
1257                self.metrics
1258            }
1259
1260            fn finish_line(&mut self) {
1261                let class = classify_line(
1262                    self.syntax,
1263                    &mut self.state,
1264                    &mut self.javascript,
1265                    &mut self.shell,
1266                    &self.line,
1267                );
1268                self.metrics.physical_lines = self.metrics.physical_lines.saturating_add(1);
1269                match class {
1270                    LineClass::Code => {
1271                        self.metrics.code_lines = self.metrics.code_lines.saturating_add(1);
1272                    }
1273                    LineClass::Comment => {
1274                        self.metrics.comment_lines = self.metrics.comment_lines.saturating_add(1);
1275                    }
1276                    LineClass::Blank => {
1277                        self.metrics.code_blank_lines =
1278                            self.metrics.code_blank_lines.saturating_add(1);
1279                    }
1280                }
1281                self.line.clear();
1282            }
1283        }
1284
1285        #[derive(Clone, Copy, Debug)]
1286        enum LineClass {
1287            Code,
1288            Comment,
1289            Blank,
1290        }
1291
1292        fn classify_line(
1293            syntax: Syntax,
1294            state: &mut State,
1295            javascript: &mut JavaScriptContext,
1296            shell: &mut ShellContext,
1297            line: &[u8],
1298        ) -> LineClass {
1299            let mut index = usize::from(line.starts_with(&[0xef, 0xbb, 0xbf]));
1300            index = index.saturating_mul(3);
1301            let mut whitespace_boundary = matches!(state, State::Normal);
1302            let mut code = matches!(
1303                state,
1304                State::Quoted { .. }
1305                    | State::TripleQuoted { .. }
1306                    | State::RustRaw { .. }
1307                    | State::Delimited { .. }
1308                    | State::Heredoc { .. }
1309                    | State::RubyPercent { .. }
1310                    | State::Regex { .. }
1311            );
1312            let mut comment = matches!(state, State::BlockComment { .. });
1313
1314            if let State::Heredoc { terminator, indent, php } = state {
1315                let candidate = if *indent {
1316                    let indentation = line
1317                        .iter()
1318                        .take_while(|byte| {
1319                            if syntax.language == Language::Shell {
1320                                **byte == b'\t'
1321                            } else {
1322                                byte.is_ascii_whitespace()
1323                            }
1324                        })
1325                        .count();
1326                    &line[indentation..]
1327                } else {
1328                    line
1329                };
1330                if candidate == terminator
1331                    || (*php && candidate.strip_suffix(b";") == Some(terminator.as_slice()))
1332                {
1333                    *state = State::Normal;
1334                }
1335                return LineClass::Code;
1336            }
1337
1338            if matches!(state, State::RubyBlock) {
1339                if line.starts_with(b"=end") {
1340                    *state = State::Normal;
1341                }
1342                return LineClass::Comment;
1343            }
1344            if syntax.ruby_blocks && matches!(state, State::Normal) && line.starts_with(b"=begin") {
1345                *state = State::RubyBlock;
1346                return LineClass::Comment;
1347            }
1348
1349            while index < line.len() {
1350                if let State::Delimited { close } = state {
1351                    code = true;
1352                    if line[index..].starts_with(close) {
1353                        index += close.len();
1354                        *state = State::Normal;
1355                    } else {
1356                        index += 1;
1357                    }
1358                    continue;
1359                }
1360                match state.clone() {
1361                    State::BlockComment { mut depth } => {
1362                        comment = true;
1363                        let block = syntax.block.expect("block state requires block syntax");
1364                        if syntax.nested_blocks && line[index..].starts_with(block.open) {
1365                            depth = depth.saturating_add(1);
1366                            *state = State::BlockComment { depth };
1367                            index += block.open.len();
1368                        } else if line[index..].starts_with(block.close) {
1369                            depth = depth.saturating_sub(1);
1370                            *state = if depth == 0 {
1371                                State::Normal
1372                            } else {
1373                                State::BlockComment { depth }
1374                            };
1375                            index += block.close.len();
1376                        } else {
1377                            index += 1;
1378                        }
1379                    }
1380                    State::Quoted { quote, mut escaped, multiline, doubled } => {
1381                        code = true;
1382                        let byte = line[index];
1383                        if escaped {
1384                            escaped = false;
1385                        } else if byte == b'\\' {
1386                            escaped = true;
1387                        } else if byte == quote {
1388                            if doubled && line.get(index + 1) == Some(&quote) {
1389                                index += 2;
1390                                continue;
1391                            }
1392                            *state = State::Normal;
1393                            if syntax.language == Language::JavaScript {
1394                                javascript.regex_allowed = false;
1395                            }
1396                            index += 1;
1397                            continue;
1398                        }
1399                        *state = State::Quoted { quote, escaped, multiline, doubled };
1400                        index += 1;
1401                    }
1402                    State::TripleQuoted { quote, width } => {
1403                        code = true;
1404                        if line[index..].iter().take(width).all(|b| *b == quote)
1405                            && line.len() - index >= width
1406                        {
1407                            *state = State::Normal;
1408                            if syntax.language == Language::JavaScript {
1409                                javascript.regex_allowed = false;
1410                            }
1411                            index += width;
1412                        } else {
1413                            index += 1;
1414                        }
1415                    }
1416                    State::RustRaw { hashes } => {
1417                        code = true;
1418                        if rust_raw_close(&line[index..], hashes) {
1419                            *state = State::Normal;
1420                            index += usize::from(hashes) + 1;
1421                        } else {
1422                            index += 1;
1423                        }
1424                    }
1425                    State::Delimited { .. } => {
1426                        unreachable!("delimited strings are handled before matching")
1427                    }
1428                    State::RubyPercent { open, close, mut depth } => {
1429                        code = true;
1430                        match line[index] {
1431                            b'\\' => index += usize::min(2, line.len() - index),
1432                            byte if byte == open && open != close => {
1433                                depth += 1;
1434                                *state = State::RubyPercent { open, close, depth };
1435                                index += 1;
1436                            }
1437                            byte if byte == close => {
1438                                depth -= 1;
1439                                *state = if depth == 0 {
1440                                    State::Normal
1441                                } else {
1442                                    State::RubyPercent { open, close, depth }
1443                                };
1444                                index += 1;
1445                            }
1446                            _ => index += 1,
1447                        }
1448                    }
1449                    State::Regex { mut escaped, mut in_class } => {
1450                        code = true;
1451                        let byte = line[index];
1452                        if escaped {
1453                            escaped = false;
1454                        } else if byte == b'\\' {
1455                            escaped = true;
1456                        } else if byte == b'[' {
1457                            in_class = true;
1458                        } else if byte == b']' {
1459                            in_class = false;
1460                        } else if byte == b'/' && !in_class {
1461                            *state = State::Normal;
1462                            index += 1;
1463                            javascript.regex_allowed = false;
1464                            continue;
1465                        }
1466                        *state = State::Regex { escaped, in_class };
1467                        index += 1;
1468                    }
1469                    State::Heredoc { .. } => unreachable!("heredocs return before byte scanning"),
1470                    State::RubyBlock => unreachable!("Ruby blocks return before byte scanning"),
1471                    State::Normal => {
1472                        let byte = line[index];
1473                        if byte.is_ascii_whitespace() {
1474                            whitespace_boundary = true;
1475                            index += 1;
1476                            continue;
1477                        }
1478                        if let Some((character, width)) = leading_utf8_character(&line[index..]) {
1479                            if crate::content::content_basic_metrics::is_content_whitespace(
1480                                character,
1481                            ) {
1482                                whitespace_boundary = true;
1483                                index += width;
1484                                continue;
1485                            }
1486                        }
1487                        if syntax.language == Language::Shell {
1488                            if shell.arithmetic_parens == 0 && line[index..].starts_with(b"$((") {
1489                                shell.arithmetic_parens = 2;
1490                                code = true;
1491                                whitespace_boundary = false;
1492                                index += 3;
1493                                continue;
1494                            }
1495                            if shell.arithmetic_parens == 0
1496                                && whitespace_boundary
1497                                && line[index..].starts_with(b"((")
1498                            {
1499                                shell.arithmetic_parens = 2;
1500                                code = true;
1501                                whitespace_boundary = false;
1502                                index += 2;
1503                                continue;
1504                            }
1505                            if shell.arithmetic_parens > 0 {
1506                                match byte {
1507                                    b'(' => {
1508                                        shell.arithmetic_parens =
1509                                            shell.arithmetic_parens.saturating_add(1);
1510                                    }
1511                                    b')' => shell.arithmetic_parens -= 1,
1512                                    _ => {}
1513                                }
1514                            }
1515                        }
1516                        if syntax.language == Language::JavaScript
1517                            && byte == b'/'
1518                            && javascript.regex_allowed
1519                            && !line[index..].starts_with(b"//")
1520                            && !line[index..].starts_with(b"/*")
1521                        {
1522                            code = true;
1523                            whitespace_boundary = false;
1524                            *state = State::Regex { escaped: false, in_class: false };
1525                            index += 1;
1526                            continue;
1527                        }
1528                        if syntax.line_comments.iter().any(|marker| {
1529                            line[index..].starts_with(marker)
1530                                && (!syntax.shell_hash_boundary || whitespace_boundary)
1531                        }) {
1532                            comment = true;
1533                            break;
1534                        }
1535                        if let Some(block) = syntax.block {
1536                            if line[index..].starts_with(block.open) {
1537                                comment = true;
1538                                whitespace_boundary = false;
1539                                *state = State::BlockComment { depth: 1 };
1540                                index += block.open.len();
1541                                continue;
1542                            }
1543                        }
1544                        if syntax.rust_raw_strings {
1545                            if let Some((hashes, consumed)) = rust_raw_open(&line[index..]) {
1546                                code = true;
1547                                whitespace_boundary = false;
1548                                *state = State::RustRaw { hashes };
1549                                index += consumed;
1550                                continue;
1551                            }
1552                        }
1553                        if syntax.language == Language::Cpp {
1554                            if let Some((close, consumed)) = cpp_raw_open(&line[index..]) {
1555                                code = true;
1556                                *state = State::Delimited { close };
1557                                index += consumed;
1558                                continue;
1559                            }
1560                        }
1561                        if syntax.language == Language::Sql {
1562                            if let Some((close, consumed)) = sql_dollar_open(&line[index..]) {
1563                                code = true;
1564                                *state = State::Delimited { close };
1565                                index += consumed;
1566                                continue;
1567                            }
1568                        }
1569                        if syntax.language == Language::Ruby {
1570                            if let Some((open, close, consumed)) = ruby_percent_open(&line[index..])
1571                            {
1572                                code = true;
1573                                *state = State::RubyPercent { open, close, depth: 1 };
1574                                index += consumed;
1575                                continue;
1576                            }
1577                        }
1578                        if let Some((terminator, indent, php)) =
1579                            (syntax.language != Language::Shell || shell.arithmetic_parens == 0)
1580                                .then(|| heredoc_open(syntax.language, &line[index..]))
1581                                .flatten()
1582                        {
1583                            code = true;
1584                            *state = State::Heredoc { terminator, indent, php };
1585                            break;
1586                        }
1587                        if syntax.language == Language::CSharp && line[index..].starts_with(b"@\"")
1588                        {
1589                            code = true;
1590                            *state = State::Quoted {
1591                                quote: b'"',
1592                                escaped: false,
1593                                multiline: true,
1594                                doubled: true,
1595                            };
1596                            index += 2;
1597                            continue;
1598                        }
1599                        if matches!(syntax.language, Language::CSharp | Language::Java)
1600                            && line[index..].starts_with(b"\"\"\"")
1601                        {
1602                            code = true;
1603                            let width = if syntax.language == Language::CSharp {
1604                                line[index..].iter().take_while(|b| **b == b'"').count()
1605                            } else {
1606                                3
1607                            };
1608                            *state = State::TripleQuoted { quote: b'"', width };
1609                            index += width;
1610                            continue;
1611                        }
1612                        if syntax.triple_quotes
1613                            && matches!(byte, b'\'' | b'"')
1614                            && line[index..].starts_with(&[byte, byte, byte])
1615                        {
1616                            code = true;
1617                            whitespace_boundary = false;
1618                            *state = State::TripleQuoted { quote: byte, width: 3 };
1619                            index += 3;
1620                            continue;
1621                        }
1622                        if matches!(byte, b'\'' | b'"') || (syntax.backtick_strings && byte == b'`')
1623                        {
1624                            if syntax.language == Language::Rust
1625                                && byte == b'\''
1626                                && rust_lifetime(&line[index..])
1627                            {
1628                                code = true;
1629                                index += 1;
1630                                continue;
1631                            }
1632                            code = true;
1633                            whitespace_boundary = false;
1634                            let multiline = byte == b'`'
1635                                || (byte == b'"' && syntax.language == Language::Rust)
1636                                || matches!(
1637                                    syntax.language,
1638                                    Language::Ruby | Language::Shell | Language::Sql
1639                                )
1640                                || (syntax.language == Language::C
1641                                    && byte == b'"'
1642                                    && line.ends_with(b"\\"));
1643                            *state = State::Quoted {
1644                                quote: byte,
1645                                escaped: false,
1646                                multiline,
1647                                doubled: syntax.language == Language::Sql,
1648                            };
1649                            index += 1;
1650                            continue;
1651                        }
1652                        if syntax.language == Language::JavaScript {
1653                            if byte.is_ascii_alphanumeric() || matches!(byte, b'_' | b'$') {
1654                                let end = line[index..]
1655                                    .iter()
1656                                    .take_while(|b| {
1657                                        b.is_ascii_alphanumeric() || matches!(b, b'_' | b'$')
1658                                    })
1659                                    .count()
1660                                    + index;
1661                                let word = &line[index..end];
1662                                javascript.pending_control_paren = !javascript.after_dot
1663                                    && matches!(word, b"if" | b"while" | b"for" | b"with");
1664                                javascript.after_dot = false;
1665                                javascript.regex_allowed = matches!(
1666                                    word,
1667                                    b"return"
1668                                        | b"else"
1669                                        | b"do"
1670                                        | b"throw"
1671                                        | b"case"
1672                                        | b"delete"
1673                                        | b"typeof"
1674                                        | b"void"
1675                                        | b"yield"
1676                                        | b"await"
1677                                        | b"instanceof"
1678                                        | b"in"
1679                                        | b"of"
1680                                );
1681                                code = true;
1682                                index = end;
1683                                continue;
1684                            }
1685                            if byte == b'(' {
1686                                javascript.paren_control.push(javascript.pending_control_paren);
1687                                javascript.pending_control_paren = false;
1688                                javascript.after_dot = false;
1689                                javascript.regex_allowed = true;
1690                                code = true;
1691                                index += 1;
1692                                continue;
1693                            }
1694                            if byte == b')' {
1695                                javascript.regex_allowed =
1696                                    javascript.paren_control.pop().unwrap_or(false);
1697                                javascript.pending_control_paren = false;
1698                                javascript.after_dot = false;
1699                                code = true;
1700                                index += 1;
1701                                continue;
1702                            }
1703                            javascript.pending_control_paren = false;
1704                            javascript.after_dot = byte == b'.';
1705                            javascript.regex_allowed = matches!(
1706                                byte,
1707                                b'=' | b'('
1708                                    | b'['
1709                                    | b'{'
1710                                    | b':'
1711                                    | b','
1712                                    | b';'
1713                                    | b'!'
1714                                    | b'?'
1715                                    | b'&'
1716                                    | b'|'
1717                                    | b'+'
1718                                    | b'-'
1719                                    | b'*'
1720                                    | b'%'
1721                                    | b'^'
1722                                    | b'~'
1723                                    | b'<'
1724                                    | b'>'
1725                                    | b'/'
1726                            );
1727                        }
1728                        code = true;
1729                        whitespace_boundary = syntax.language == Language::Shell
1730                            && matches!(byte, b';' | b'&' | b'|' | b'(' | b')' | b'<' | b'>');
1731                        index += 1;
1732                    }
1733                }
1734            }
1735
1736            match state {
1737                State::Quoted { multiline: false, escaped: true, .. }
1738                    if matches!(syntax.language, Language::C | Language::Cpp) =>
1739                {
1740                    *state = State::Quoted {
1741                        quote: b'"',
1742                        escaped: false,
1743                        multiline: false,
1744                        doubled: false,
1745                    };
1746                }
1747                State::Quoted { multiline: false, .. } | State::Regex { .. } => {
1748                    *state = State::Normal
1749                }
1750                State::Quoted { multiline: true, escaped, .. } => *escaped = false,
1751                _ => {}
1752            }
1753            if code {
1754                LineClass::Code
1755            } else if comment {
1756                LineClass::Comment
1757            } else {
1758                LineClass::Blank
1759            }
1760        }
1761    }
1762
1763    fn whole(language: &str, chunks: &[&[u8]]) -> MetricValues {
1764        let mut counter =
1765            reference::WholeLineAccumulator::for_type(language).expect("supported language");
1766        for chunk in chunks {
1767            counter.push(chunk);
1768        }
1769        counter.finish()
1770    }
1771
1772    /// The streaming scan's metrics after each line it finished, and its peak window.
1773    fn streaming_by_line(
1774        language: &str,
1775        window: usize,
1776        chunks: &[&[u8]],
1777    ) -> (Vec<MetricValues>, usize) {
1778        let mut counter =
1779            CodeAccumulator::with_window_bytes(language, window).expect("supported language");
1780        for chunk in chunks {
1781            counter.push(chunk);
1782        }
1783        let peak = counter.peak_window();
1784        let (total, by_line) = counter.finish_by_line();
1785        assert_eq!(by_line.last().copied().unwrap_or_default(), total, "{language}");
1786        (by_line, peak)
1787    }
1788
1789    /// The reference's metrics after each line it finishes. The oracle stays verbatim, so
1790    /// it keeps no history: each entry is a fresh reference fed the source through that
1791    /// line's terminator, where the reference finishes a line (a CR, or an LF no CR
1792    /// precedes), and a last entry is its total when `finish` adds an unterminated line.
1793    fn whole_by_line(language: &str, source: &[u8]) -> Vec<MetricValues> {
1794        let mut by_line = Vec::new();
1795        for (at, &byte) in source.iter().enumerate() {
1796            if byte == b'\r' || (byte == b'\n' && (at == 0 || source[at - 1] != b'\r')) {
1797                by_line.push(whole(language, &[&source[..=at]]));
1798            }
1799        }
1800        let total = whole(language, &[source]);
1801        if total.physical_lines > by_line.last().map_or(0, |last| last.physical_lines) {
1802            by_line.push(total);
1803        }
1804        by_line
1805    }
1806
1807    /// The streaming and the reference metrics agree after every line, not only in total:
1808    /// the metrics are cumulative, so a code line counted as a comment cannot be hidden by
1809    /// a comment line counted as code later. A disagreement names its first line.
1810    fn assert_same_lines(actual: &[MetricValues], expected: &[MetricValues], label: &str) {
1811        if let Some(line) = actual.iter().zip(expected).position(|(left, right)| left != right) {
1812            panic!(
1813                "{label}: line {} disagrees: streaming {:?}, whole-line {:?}",
1814                line + 1,
1815                actual[line],
1816                expected[line]
1817            );
1818        }
1819        assert_eq!(actual.len(), expected.len(), "{label}: lines finished");
1820    }
1821
1822    const LANGUAGES: [&str; 15] = [
1823        "rust",
1824        "javascript",
1825        "typescript",
1826        "go",
1827        "c",
1828        "cpp",
1829        "csharp",
1830        "java",
1831        "kotlin",
1832        "swift",
1833        "php",
1834        "python",
1835        "ruby",
1836        "shell",
1837        "sql",
1838    ];
1839
1840    /// Every construct the classifier knows, in one source fed to every language: what a
1841    /// construct means to a language the reference decides, and the streaming scan has
1842    /// to agree byte for byte. Line endings, splices, a byte-order mark on a later line,
1843    /// invalid UTF-8, Unicode whitespace, and no trailing newline are all in it.
1844    fn everything() -> Vec<u8> {
1845        let mut source = Vec::new();
1846        let lines: &[&[u8]] = &[
1847            b"// line comment\r\n",
1848            b"# hash comment\n",
1849            b"-- dash comment\r",
1850            b"code(); // trailing\n",
1851            b"/* block\n",
1852            b"still block */ code /* again */\n",
1853            b"/* outer /* inner */ still */ done\n",
1854            b"let s = \"quoted \\\" escape // not comment\"; // yes\n",
1855            b"let c = 'x'; let d = '\\''; /* tail\n",
1856            b"*/\n",
1857            b"SELECT 'it''s' -- doubled\n",
1858            b"x = \"\"\"triple\n",
1859            b"# inside\n",
1860            b"\"\"\" # after\n",
1861            b"y = '''one line''' # after\n",
1862            b"let r = r#\"raw \"# // not\"#; // comment\n",
1863            b"let r2 = r####\"deep\"### still\"####; // comment\n",
1864            b"auto s = R\"x(raw // not\n",
1865            b")x\"; // comment\n",
1866            b"auto t = R\"toolongdelimiterforcpp(nope)\"; // comment\n",
1867            b"SELECT $tag$dollar -- not\n",
1868            b"$tag$; -- comment\n",
1869            b"SELECT $$anon$$; -- comment\n",
1870            b"SELECT $bad-tag$; -- comment\n",
1871            b"s = %q{outer {inner} # not\n",
1872            b"} # comment\n",
1873            b"w = %w[a b] # comment\n",
1874            b"cat <<TEXT\n",
1875            b"# heredoc body\n",
1876            b"TEXT\n",
1877            b"cat <<-TEXT\n",
1878            b"\t# indented body\n",
1879            b"\tTEXT\n",
1880            b"cat <<~TEXT # ruby\n",
1881            b"  TEXT\n",
1882            b"cat <<'QUOTED' # x\n",
1883            b"QUOTED\n",
1884            b"$s = <<<PHP\n",
1885            b"// body\n",
1886            b"PHP;\n",
1887            b"x=$((1<<2)) # comment\n",
1888            b"((1<<3)) # comment\n",
1889            b"a=1;# comment after semicolon\n",
1890            b"printf foo#bar # comment\n",
1891            b"=begin\n",
1892            b"ruby block\n",
1893            b"=end\n",
1894            b"const re = /[/*]/; // comment\n",
1895            b"const q = total / count; /* comment */\n",
1896            b"if (ok) /[/*]/.test(v); next();\n",
1897            b"return /\\/\\* x [/] /; // c\n",
1898            b"var t = `tick // not\n",
1899            b"` // comment\n",
1900            b"var v = @\"verbatim \"\" // not\n",
1901            b"\"; // comment\n",
1902            b"var j = \"\"\"\n",
1903            b"// java text block\n",
1904            b"\"\"\"; // comment\n",
1905            b"var cs = \"\"\"\"\"\n",
1906            b"// five quotes\n",
1907            b"\"\"\"\"\"; // comment\n",
1908            b"fn f<'a>(x: &'a str) -> &'a str { x } // lifetime\n",
1909            b"let ch = 'a'; // char\n",
1910            b"const char *s = \"spliced \\\n",
1911            b"// continued\";\n",
1912            b"const char *t = \"open\\\n",
1913            b"\n",
1914            b"\xef\xbb\xbfbom line // comment\n",
1915            b"\xef\xbb\xbf\n",
1916            b"\xff\xfe invalid utf8 // comment\n",
1917            b"\xe3\x80\x80// ideographic space then comment\n",
1918            b"\xe2\x80\x83code after em space\n",
1919            b"\xe3\x80\n",
1920            b"\r\n",
1921            b"\n",
1922            b"   \n",
1923            b"trailing backslash \\\n",
1924            b"last line without newline",
1925        ];
1926        for line in lines {
1927            source.extend_from_slice(line);
1928        }
1929        source
1930    }
1931
1932    /// Tokens the window's edge can cut mid-way, each longer than any small window:
1933    /// raw-string hashes, a dollar-quote tag, a heredoc tag, a JavaScript identifier, a
1934    /// C# quote run, and a delimited closer the scan must wait for whole.
1935    fn long_tokens() -> Vec<u8> {
1936        let mut source = Vec::new();
1937        let hashes = "#".repeat(200);
1938        source.extend_from_slice(format!("let r = r{hashes}\"x // y\"{hashes}; // c\n").as_bytes());
1939        let tag = "t".repeat(300);
1940        source.extend_from_slice(format!("SELECT ${tag}$body -- not\n").as_bytes());
1941        source.extend_from_slice(format!("more ${tag}$; -- comment\n").as_bytes());
1942        source.extend_from_slice(format!("cat <<{tag}\n# body\n{tag}\n").as_bytes());
1943        source.extend_from_slice(format!("cat <<'{tag}'\n# body\n{tag}\n").as_bytes());
1944        let word = "w".repeat(400);
1945        source.extend_from_slice(format!("if ({word}) /[/*]/.test(v); // c\n").as_bytes());
1946        source.extend_from_slice(format!("{word}do /re/ // c\n").as_bytes());
1947        let quotes = "\"".repeat(40);
1948        source.extend_from_slice(format!("var s = {quotes}\n// text\n{quotes}; // c\n").as_bytes());
1949        source.extend_from_slice(b"auto s = R\"abcdefghijklmnop(raw)abcdefghijklmnop\"; // c\n");
1950        source.extend_from_slice(b"%q{ // tail\n");
1951        source
1952    }
1953
1954    /// The streaming scan agrees with the whole-line classifier for every language, line
1955    /// by line: at nine window sizes, including a window of one byte, which puts a piece
1956    /// edge at every byte; and at windows of one and three bytes, with the source split
1957    /// in two at every seventh byte.
1958    #[test]
1959    fn piecewise_classification_agrees_with_the_whole_line_classifier_everywhere() {
1960        let mut both = everything();
1961        both.extend_from_slice(&long_tokens());
1962        let sources = [everything(), long_tokens(), both];
1963        for language in LANGUAGES {
1964            for source in &sources {
1965                let expected = whole_by_line(language, source);
1966                for window in [1, 2, 3, 5, 7, 16, 61, 4096, LINE_WINDOW_BYTES] {
1967                    let (actual, _) = streaming_by_line(language, window, &[source]);
1968                    assert_same_lines(&actual, &expected, &format!("{language}: window {window}"));
1969                }
1970                for split in (0..=source.len()).step_by(7) {
1971                    let chunks: [&[u8]; 2] = [&source[..split], &source[split..]];
1972                    for window in [1, 3] {
1973                        let (actual, _) = streaming_by_line(language, window, &chunks);
1974                        let label = format!("{language}: split {split}, window {window}");
1975                        assert_same_lines(&actual, &expected, &label);
1976                    }
1977                }
1978            }
1979        }
1980    }
1981
1982    /// What the line-by-line comparison is for: a code line counted as a comment and a
1983    /// comment counted as code leave equal totals, and still disagree on the first line.
1984    #[test]
1985    #[should_panic(expected = "swapped: line 1 disagrees")]
1986    fn a_swap_that_leaves_equal_totals_still_disagrees_line_by_line() {
1987        let code = MetricValues { physical_lines: 1, code_lines: 1, ..MetricValues::default() };
1988        let comment =
1989            MetricValues { physical_lines: 1, comment_lines: 1, ..MetricValues::default() };
1990        let both = MetricValues {
1991            physical_lines: 2,
1992            code_lines: 1,
1993            comment_lines: 1,
1994            ..MetricValues::default()
1995        };
1996        assert_same_lines(&[code, both], &[comment, both], "swapped");
1997    }
1998
1999    /// Every two- and three-way chunking of a small mixed source, at window 1, for every
2000    /// language, line by line.
2001    #[test]
2002    fn every_chunk_boundary_of_a_mixed_source_agrees() {
2003        let source = b"a = \"q\\\"\" /* c\n */ r#\"x\"# 'y' $t$z$t$ <<T\nT\n// d\n\xef\xbb\xbf%q{w}\r\n\xe3\x80\x80e";
2004        for language in LANGUAGES {
2005            let expected = whole_by_line(language, source);
2006            for first in 0..=source.len() {
2007                for second in first..=source.len() {
2008                    let chunks: [&[u8]; 3] =
2009                        [&source[..first], &source[first..second], &source[second..]];
2010                    let (actual, _) = streaming_by_line(language, 1, &chunks);
2011                    let label = format!("{language}: splits {first}, {second}");
2012                    assert_same_lines(&actual, &expected, &label);
2013                }
2014            }
2015        }
2016    }
2017
2018    /// A long line is classified in pieces as it arrives, and the window never holds
2019    /// more than its bound plus what a piece cannot yet decide.
2020    #[test]
2021    fn a_long_line_costs_the_window_not_the_line() {
2022        let segment = everything();
2023        let mut line: Vec<u8> = Vec::new();
2024        while line.len() < 3 * LINE_WINDOW_BYTES {
2025            line.extend(
2026                segment.iter().map(
2027                    |byte| {
2028                        if matches!(*byte, b'\n' | b'\r') { b' ' } else { *byte }
2029                    },
2030                ),
2031            );
2032        }
2033        line.push(b'\n');
2034        for language in LANGUAGES {
2035            let expected = whole_by_line(language, &line);
2036            let (actual, peak) = streaming_by_line(language, LINE_WINDOW_BYTES, &[&line]);
2037            assert_same_lines(&actual, &expected, language);
2038            assert!(
2039                peak < 2 * LINE_WINDOW_BYTES,
2040                "{language}: the window held {peak} bytes of a {}-byte line",
2041                line.len()
2042            );
2043        }
2044    }
2045
2046    /// The bound this exists for: a single 64 MiB line of code holds at most the window
2047    /// and one read chunk, not the line (fdu-1zb6).
2048    #[test]
2049    fn a_sixty_four_mebibyte_line_holds_at_most_the_window_and_a_chunk() {
2050        const CHUNK: usize = 64 * 1024;
2051        let piece: &[u8] = b"var a = 1; /* c */ b = \"s\"; ";
2052        let mut chunk = Vec::with_capacity(CHUNK + piece.len());
2053        while chunk.len() < CHUNK {
2054            chunk.extend_from_slice(piece);
2055        }
2056        chunk.truncate(CHUNK);
2057        let mut counter = CodeAccumulator::for_type("javascript").expect("javascript is supported");
2058        let chunks = (64 * 1024 * 1024) / CHUNK;
2059        for _ in 0..chunks {
2060            counter.push(&chunk);
2061        }
2062        counter.push(b"\n");
2063        let peak = counter.peak_window();
2064        let metrics = counter.finish();
2065        assert_eq!((metrics.physical_lines, metrics.code_lines), (1, 1));
2066        assert!(
2067            peak <= LINE_WINDOW_BYTES + CHUNK,
2068            "the window held {peak} bytes of a {} MiB line",
2069            chunks * CHUNK / (1024 * 1024)
2070        );
2071    }
2072
2073    type PartitionCase<'a> = (&'a str, &'a [u8], (u64, u64, u64));
2074
2075    fn count(language: &str, chunks: &[&[u8]]) -> MetricValues {
2076        let mut counter = CodeAccumulator::for_type(language).expect("supported language");
2077        for chunk in chunks {
2078            counter.push(chunk);
2079        }
2080        counter.finish()
2081    }
2082
2083    #[test]
2084    fn partitions_c_like_source_and_counts_mixed_lines_as_code() {
2085        let metrics = count(
2086            "javascript",
2087            &[b"// first\r\nlet url = \"https://example.test\"; // tail\r/* block\n\nend */\n`// text\nmore`;"],
2088        );
2089        assert_eq!(metrics.physical_lines, 7);
2090        assert_eq!(metrics.code_lines, 3);
2091        assert_eq!(metrics.comment_lines, 4);
2092        assert_eq!(metrics.code_blank_lines, 0);
2093    }
2094
2095    #[test]
2096    fn rust_nested_comments_and_raw_strings_ignore_comment_markers() {
2097        let source =
2098            b"/* outer\n/* inner */\n*/\nlet raw = r##\"/* text */\n// still text\"##;\n\n";
2099        let expected = count("rust", &[source]);
2100        assert_eq!(expected.physical_lines, 6);
2101        assert_eq!(expected.code_lines, 2);
2102        assert_eq!(expected.comment_lines, 3);
2103        assert_eq!(expected.code_blank_lines, 1);
2104        for split in 0..=source.len() {
2105            assert_eq!(count("rust", &[&source[..split], &source[split..]]), expected);
2106        }
2107    }
2108
2109    #[test]
2110    fn triple_quoted_docstrings_are_code_in_v1() {
2111        let metrics = count("python", &[b"\"\"\"docs\n# text\n\"\"\"\n# comment\npass\n"]);
2112        assert_eq!(metrics.physical_lines, 5);
2113        assert_eq!(metrics.code_lines, 4);
2114        assert_eq!(metrics.comment_lines, 1);
2115        assert_eq!(metrics.code_blank_lines, 0);
2116    }
2117
2118    #[test]
2119    fn every_line_ending_convention_has_the_same_partition() {
2120        for source in [
2121            "// comment\nlet value = 1;\n\n",
2122            "// comment\r\nlet value = 1;\r\n\r\n",
2123            "// comment\rlet value = 1;\r\r",
2124            "// comment\r\nlet value = 1;\r\n",
2125        ] {
2126            let metrics = count("rust", &[source.as_bytes()]);
2127            let expected_lines = if source.ends_with("value = 1;\r\n") { 2 } else { 3 };
2128            assert_eq!(metrics.physical_lines, expected_lines, "{source:?}");
2129            assert_eq!(metrics.code_lines, 1, "{source:?}");
2130            assert_eq!(metrics.comment_lines, 1, "{source:?}");
2131            assert_eq!(metrics.code_blank_lines, expected_lines - 2, "{source:?}");
2132        }
2133    }
2134
2135    #[test]
2136    fn a_leading_utf8_bom_is_not_an_invented_line() {
2137        let empty = count("rust", &[b"\xef\xbb\xbf"]);
2138        assert_eq!(empty.physical_lines, 0);
2139
2140        let blank = count("rust", &[b"\xef\xbb\xbf\n"]);
2141        assert_eq!(blank.physical_lines, 1);
2142        assert_eq!(blank.code_blank_lines, 1);
2143    }
2144
2145    #[test]
2146    fn unicode_whitespace_uses_the_basic_analyzers_pinned_table() {
2147        let metrics = count("rust", &["\u{3000}\n\u{2003}// comment\n".as_bytes()]);
2148        assert_eq!(metrics.physical_lines, 2);
2149        assert_eq!(metrics.code_blank_lines, 1);
2150        assert_eq!(metrics.comment_lines, 1);
2151        assert_eq!(metrics.code_lines, 0);
2152
2153        let shell = count("shell", &["\u{3000}# comment\n".as_bytes()]);
2154        assert_eq!(shell.comment_lines, 1);
2155        assert_eq!(shell.code_lines, 0);
2156    }
2157
2158    #[test]
2159    fn unsupported_languages_are_explicit() {
2160        assert!(CodeAccumulator::for_type("haskell").is_none());
2161    }
2162
2163    #[test]
2164    fn rust_multiline_string_and_lifetime_preserve_following_comment_state() {
2165        let string = b"const S: &str = \"first\n// text\nlast\";\n";
2166        let lifetime = b"fn x<'a>() { /*\ncomment\n*/ }\n";
2167        for (source, code, comment) in [(string.as_slice(), 3, 0), (lifetime.as_slice(), 2, 1)] {
2168            for split in 0..=source.len() {
2169                let metrics = count("rust", &[&source[..split], &source[split..]]);
2170                assert_eq!(
2171                    (metrics.code_lines, metrics.comment_lines),
2172                    (code, comment),
2173                    "split {split}"
2174                );
2175            }
2176        }
2177    }
2178
2179    #[test]
2180    fn multiline_literal_families_preserve_comment_markers() {
2181        let cases: &[PartitionCase<'_>] = &[
2182            ("cpp", b"const char *s = R\"tag(first\n// text\nlast)tag\";\n", (3, 0, 0)),
2183            ("c", b"const char *s = \"first\\\n// text\";\n", (2, 0, 0)),
2184            ("java", b"class C { String s = \"\"\"\n// text\nlast\n\"\"\"; }\n", (4, 0, 0)),
2185            ("csharp", b"class C { string s = @\"first\n// text\nlast\"; }\n", (3, 0, 0)),
2186            ("csharp", b"class C { string s = \"\"\"\n// text\nlast\n\"\"\"; }\n", (4, 0, 0)),
2187            ("ruby", b"s = %q{first\n# text\nlast}\n", (3, 0, 0)),
2188            ("ruby", b"s = <<~TEXT\n# text\nlast\nTEXT\n", (4, 0, 0)),
2189            ("ruby", b"s = \"first\n# text\nlast\"\n", (3, 0, 0)),
2190            ("shell", b"cat <<'TEXT'\n# text\nlast\nTEXT\n", (4, 0, 0)),
2191            ("shell", b"value='first\n# text\nlast'\n", (3, 0, 0)),
2192            ("sql", b"SELECT $tag$first\n-- text\nlast$tag$;\n", (3, 0, 0)),
2193            ("sql", b"SELECT 'first\n-- text\nlast';\n", (3, 0, 0)),
2194            ("php", b"<?php\n$s = <<<TEXT\n// text\nlast\nTEXT;\n", (5, 0, 0)),
2195        ];
2196        for (language, source, expected) in cases {
2197            for split in 0..=source.len() {
2198                let metrics = count(language, &[&source[..split], &source[split..]]);
2199                assert_eq!(
2200                    (metrics.code_lines, metrics.comment_lines, metrics.code_blank_lines),
2201                    *expected,
2202                    "{language} split {split}"
2203                );
2204            }
2205        }
2206    }
2207
2208    #[test]
2209    fn multiline_delimiters_restore_comment_recognition_after_closing() {
2210        let cases: &[(&str, &[u8], (u64, u64))] = &[
2211            ("cpp", b"auto s = R\"x(/*\n// body\n)x\";\n// comment\n", (3, 1)),
2212            ("csharp", b"var s = @\"first \"\" quote\n// body\nlast\";\n// comment\n", (3, 1)),
2213            ("ruby", b"s = %q{outer {inner\n# body\n}}\n# comment\n", (3, 1)),
2214            ("ruby", b"s = <<~TEXT\n# body\n  TEXT\n# comment\n", (3, 1)),
2215            ("shell", b"cat <<'TEXT'\n# body\nTEXT\n# comment\n", (3, 1)),
2216            ("shell", b"cat <<-TEXT\n TEXT\n# body\nTEXT\n# comment\n", (4, 1)),
2217            ("shell", b"cat <<-TEXT\n\tTEXT\n# comment\n", (2, 1)),
2218            ("sql", b"SELECT $tag$first\n-- body\nlast$tag$;\n-- comment\n", (3, 1)),
2219            ("php", b"$s = <<<TEXT\n// body\nTEXT;\n// comment\n", (3, 1)),
2220        ];
2221        for (language, source, expected) in cases {
2222            let metrics = count(language, &[source]);
2223            assert_eq!((metrics.code_lines, metrics.comment_lines), *expected, "{language}");
2224        }
2225    }
2226
2227    #[test]
2228    fn shell_comments_and_arithmetic_shifts_do_not_hold_lexer_state() {
2229        let cases: &[PartitionCase<'_>] = &[
2230            // A command separator starts a new shell word, so the unmatched quote is
2231            // comment text and cannot turn the next line into a multiline string.
2232            ("shell", b"true;# \"unterminated\n# following\nprintf ok\n", (2, 1, 0)),
2233            // In arithmetic expansion and arithmetic commands, `<<` shifts a value.
2234            // Neither spelling opens a heredoc that consumes the following comment.
2235            ("shell", b"N=2\nx=$((1<<N))\n# following\n", (2, 1, 0)),
2236            ("shell", b"N=2\n((1<<N))\n# following\n", (2, 1, 0)),
2237            // A hash within a shell word is literal, while a real heredoc body is code.
2238            ("shell", b"printf '%s' foo#bar\ncat <<TEXT\n# literal\nTEXT\n# comment\n", (4, 1, 0)),
2239        ];
2240        for (language, source, expected) in cases {
2241            for split in 0..=source.len() {
2242                let metrics = count(language, &[&source[..split], &source[split..]]);
2243                assert_eq!(
2244                    (metrics.code_lines, metrics.comment_lines, metrics.code_blank_lines),
2245                    *expected,
2246                    "source {source:?}, split {split}"
2247                );
2248            }
2249        }
2250    }
2251
2252    #[test]
2253    fn javascript_regex_and_division_leave_comment_state_correct() {
2254        let cases: &[PartitionCase<'_>] = &[
2255            ("javascript", b"const re = /[/*]/;\nconst answer = 42;\n", (2, 0, 0)),
2256            ("typescript", b"const re = /\\/\\* inside [/] /;\nconst n = 9;\n", (2, 0, 0)),
2257            ("javascript", b"const ratio = total / count;\n/* comment */\nconst re = /[//]/;\n", (2, 1, 0)),
2258            ("typescript", b"return /[/*]/.test(value);\n// comment\nnext();\n", (2, 1, 0)),
2259            ("javascript", b"const quotient = total\n / count; /* real comment */\nconst re = /[/*]/;\nnext();\n", (4, 0, 0)),
2260            ("javascript", b"const text = \"plain\" / count;\nconst re = /[/*]/;\nnext();\n", (3, 0, 0)),
2261            ("javascript", b"if (ok) /[/*]/.test(value);\nconst next = 1;\n", (2, 0, 0)),
2262            ("javascript", b"if (first) {}\nif (ok) /[/*]/.test(value);\nnext();\n", (3, 0, 0)),
2263            ("typescript", b"while (ready && check(x)) /[/*]/.test(value);\nnext();\n", (2, 0, 0)),
2264            ("javascript", b"if (ok &&\n check(x)) /[/*]/.test(value);\nnext();\n", (3, 0, 0)),
2265            ("javascript", b"if (ok) fn(value) / count;\n/* comment */\n", (1, 1, 0)),
2266            ("javascript", b"const ratio = object.if(value) / count;\n/* comment */\n", (1, 1, 0)),
2267        ];
2268        for (language, source, expected) in cases {
2269            for split in 0..=source.len() {
2270                let metrics = count(language, &[&source[..split], &source[split..]]);
2271                assert_eq!(
2272                    (metrics.code_lines, metrics.comment_lines, metrics.code_blank_lines),
2273                    *expected,
2274                    "{language} split {split}"
2275                );
2276            }
2277        }
2278    }
2279}