Skip to main content

fdu_core/content/
content_basic_metrics.rs

1//! Fused streaming admission, line, word, paragraph, and logical-word statistics.
2
3use super::MetricValues;
4
5/// Outcome of the basic text-admission pass.
6#[derive(Clone, Copy, PartialEq, Eq, Debug)]
7pub enum TextAdmission {
8    /// Valid UTF-8 without a NUL byte.
9    Accepted(MetricValues),
10    /// A NUL byte appeared anywhere in the stream.
11    Binary,
12    /// The byte stream was not valid UTF-8.
13    InvalidUtf8,
14}
15
16/// Stateful fused counter that accepts arbitrary byte chunks.
17#[derive(Debug, Default)]
18#[allow(clippy::struct_excessive_bools)]
19pub struct BasicAccumulator {
20    metrics: MetricValues,
21    utf8_carry: Vec<u8>,
22    collect_logical: bool,
23    binary: bool,
24    invalid_utf8: bool,
25    first_character: bool,
26    current_line_exists: bool,
27    current_line_nonblank: bool,
28    previous_cr: bool,
29    raw_word_open: bool,
30    nonwide_word_open: bool,
31    paragraph_open: bool,
32}
33
34impl BasicAccumulator {
35    /// Create an empty streaming counter.
36    pub fn new() -> Self {
37        Self::with_logical_metrics(false)
38    }
39
40    /// Select the optional logical-word and paragraph collectors.
41    pub(crate) fn with_logical_metrics(collect_logical: bool) -> Self {
42        Self { collect_logical, first_character: true, ..Self::default() }
43    }
44
45    /// Consume another byte chunk.
46    pub fn push(&mut self, chunk: &[u8]) {
47        if self.binary || self.invalid_utf8 {
48            return;
49        }
50        if chunk.contains(&0) {
51            self.binary = true;
52            self.utf8_carry.clear();
53            return;
54        }
55        if self.utf8_carry.is_empty() {
56            self.push_utf8(chunk);
57            return;
58        }
59
60        let mut joined = Vec::with_capacity(self.utf8_carry.len() + chunk.len());
61        joined.extend_from_slice(&self.utf8_carry);
62        joined.extend_from_slice(chunk);
63        self.utf8_carry.clear();
64        self.push_utf8(&joined);
65    }
66
67    fn push_utf8(&mut self, bytes: &[u8]) {
68        match std::str::from_utf8(bytes) {
69            Ok(text) => self.push_text(text),
70            Err(error) => {
71                let valid = error.valid_up_to();
72                if let Ok(text) = std::str::from_utf8(&bytes[..valid]) {
73                    self.push_text(text);
74                }
75                if error.error_len().is_some() {
76                    self.invalid_utf8 = true;
77                } else {
78                    self.utf8_carry.extend_from_slice(&bytes[valid..]);
79                }
80            }
81        }
82    }
83
84    /// Finish the stream, accounting for an unterminated final line and word.
85    pub fn finish(mut self) -> TextAdmission {
86        if self.binary {
87            return TextAdmission::Binary;
88        }
89        if self.invalid_utf8 || !self.utf8_carry.is_empty() {
90            return TextAdmission::InvalidUtf8;
91        }
92        if self.current_line_exists {
93            self.finish_line();
94        }
95        if self.raw_word_open {
96            self.metrics.raw_words = self.metrics.raw_words.saturating_add(1);
97        }
98        TextAdmission::Accepted(self.metrics)
99    }
100
101    fn push_text(&mut self, text: &str) {
102        for character in text.chars() {
103            if self.first_character {
104                self.first_character = false;
105                if character == '\u{feff}' {
106                    continue;
107                }
108            }
109
110            if self.previous_cr {
111                self.previous_cr = false;
112                if character == '\n' {
113                    continue;
114                }
115            }
116            match character {
117                '\r' => {
118                    self.current_line_exists = true;
119                    self.finish_line();
120                    self.previous_cr = true;
121                }
122                '\n' => {
123                    self.current_line_exists = true;
124                    self.finish_line();
125                }
126                _ => self.push_character(character),
127            }
128        }
129    }
130
131    fn push_character(&mut self, character: char) {
132        self.current_line_exists = true;
133        if is_content_whitespace(character) {
134            if self.raw_word_open {
135                self.metrics.raw_words = self.metrics.raw_words.saturating_add(1);
136                self.raw_word_open = false;
137            }
138            self.nonwide_word_open = false;
139            return;
140        }
141
142        self.current_line_nonblank = true;
143        self.raw_word_open = true;
144        if !self.collect_logical {
145            return;
146        }
147        if is_wide(character) {
148            self.metrics.logical_word_stats.wide_chars =
149                self.metrics.logical_word_stats.wide_chars.saturating_add(1);
150            self.nonwide_word_open = false;
151        } else {
152            self.metrics.logical_word_stats.nonwide_chars =
153                self.metrics.logical_word_stats.nonwide_chars.saturating_add(1);
154            if !self.nonwide_word_open {
155                self.metrics.logical_word_stats.nonwide_tokens =
156                    self.metrics.logical_word_stats.nonwide_tokens.saturating_add(1);
157                self.nonwide_word_open = true;
158            }
159        }
160    }
161
162    fn finish_line(&mut self) {
163        if self.raw_word_open {
164            self.metrics.raw_words = self.metrics.raw_words.saturating_add(1);
165            self.raw_word_open = false;
166        }
167        self.metrics.physical_lines = self.metrics.physical_lines.saturating_add(1);
168        if self.current_line_nonblank {
169            self.metrics.nonblank_lines = self.metrics.nonblank_lines.saturating_add(1);
170            if self.collect_logical && !self.paragraph_open {
171                self.metrics.paragraphs = self.metrics.paragraphs.saturating_add(1);
172                self.paragraph_open = true;
173            }
174        } else {
175            self.metrics.blank_lines = self.metrics.blank_lines.saturating_add(1);
176            if self.collect_logical {
177                self.paragraph_open = false;
178            }
179        }
180        self.current_line_exists = false;
181        self.current_line_nonblank = false;
182        self.nonwide_word_open = false;
183    }
184}
185
186pub(super) fn is_content_whitespace(character: char) -> bool {
187    matches!(
188        character,
189        '\u{0009}'..='\u{000d}'
190            | '\u{0020}'
191            | '\u{0085}'
192            | '\u{00a0}'
193            | '\u{1680}'
194            | '\u{2000}'..='\u{200a}'
195            | '\u{2028}'
196            | '\u{2029}'
197            | '\u{202f}'
198            | '\u{205f}'
199            | '\u{3000}'
200    )
201}
202
203fn is_wide(character: char) -> bool {
204    matches!(
205        character as u32,
206        0x1100..=0x115f
207            | 0x2329..=0x232a
208            | 0x2e80..=0xa4cf
209            | 0xac00..=0xd7a3
210            | 0xf900..=0xfaff
211            | 0xfe10..=0xfe19
212            | 0xfe30..=0xfe6f
213            | 0xff00..=0xff60
214            | 0xffe0..=0xffe6
215            | 0x1f300..=0x1faff
216            | 0x20000..=0x3fffd
217    )
218}
219
220#[cfg(test)]
221mod tests {
222    use super::*;
223
224    fn accepted(chunks: &[&[u8]]) -> MetricValues {
225        let mut accumulator = BasicAccumulator::with_logical_metrics(true);
226        for chunk in chunks {
227            accumulator.push(chunk);
228        }
229        match accumulator.finish() {
230            TextAdmission::Accepted(metrics) => metrics,
231            other => panic!("expected accepted text, got {other:?}"),
232        }
233    }
234
235    #[test]
236    fn line_endings_and_terminal_boundaries_have_one_contract() {
237        for input in ["a\nb\n", "a\r\nb\r\n", "a\rb\r", "a\r\nb\rc\n"] {
238            let metrics = accepted(&[input.as_bytes()]);
239            let expected = if input.contains('c') { 3 } else { 2 };
240            assert_eq!(metrics.physical_lines, expected, "{input:?}");
241            assert_eq!(metrics.nonblank_lines, expected, "{input:?}");
242            assert_eq!(metrics.blank_lines, 0, "{input:?}");
243        }
244        assert_eq!(accepted(&[b""]).physical_lines, 0);
245        assert_eq!(accepted(&[b"one line"]).physical_lines, 1);
246        assert_eq!(accepted(&[b"one line\n"]).physical_lines, 1);
247        assert_eq!(accepted(&[b"\n"]).blank_lines, 1);
248        assert_eq!(accepted(&["\u{feff}".as_bytes()]).physical_lines, 0);
249        assert_eq!(accepted(&["\u{feff}\n".as_bytes()]).blank_lines, 1);
250    }
251
252    #[test]
253    fn every_chunk_boundary_matches_the_one_chunk_result() {
254        let input = "\u{feff}alpha\r\n\u{3000}\r中文 beta\nlast".as_bytes();
255        let expected = accepted(&[input]);
256        for split in 0..=input.len() {
257            assert_eq!(accepted(&[&input[..split], &input[split..]]), expected, "split {split}");
258        }
259        for first in 0..=input.len() {
260            for second in first..=input.len() {
261                assert_eq!(
262                    accepted(&[&input[..first], &input[first..second], &input[second..]]),
263                    expected,
264                    "splits {first}, {second}"
265                );
266            }
267        }
268    }
269
270    #[test]
271    fn unicode_whitespace_words_paragraphs_and_logical_stats_are_additive() {
272        let metrics = accepted(&["one two\n\n中文\u{3000}longtoken\n".as_bytes()]);
273        assert_eq!(metrics.physical_lines, 3);
274        assert_eq!(metrics.blank_lines, 1);
275        assert_eq!(metrics.nonblank_lines, 2);
276        assert_eq!(metrics.raw_words, 4);
277        assert_eq!(metrics.paragraphs, 2);
278        assert_eq!(
279            metrics.logical_word_stats,
280            super::super::LogicalWordStats { wide_chars: 2, nonwide_tokens: 3, nonwide_chars: 15 }
281        );
282    }
283
284    #[test]
285    fn analyzer_v1_pins_the_unicode_white_space_table() {
286        let white_space = [
287            '\u{0009}', '\u{000a}', '\u{000b}', '\u{000c}', '\u{000d}', '\u{0020}', '\u{0085}',
288            '\u{00a0}', '\u{1680}', '\u{2000}', '\u{2001}', '\u{2002}', '\u{2003}', '\u{2004}',
289            '\u{2005}', '\u{2006}', '\u{2007}', '\u{2008}', '\u{2009}', '\u{200a}', '\u{2028}',
290            '\u{2029}', '\u{202f}', '\u{205f}', '\u{3000}',
291        ];
292        assert!(white_space.into_iter().all(is_content_whitespace));
293        for non_whitespace in ['\u{0008}', '\u{200b}', '\u{2060}', '\u{feff}'] {
294            assert!(!is_content_whitespace(non_whitespace), "{non_whitespace:?}");
295        }
296    }
297
298    #[test]
299    fn early_or_late_nul_and_invalid_utf8_discard_provisional_metrics() {
300        let mut early = BasicAccumulator::new();
301        early.push(b"\0text");
302        assert_eq!(early.finish(), TextAdmission::Binary);
303
304        let mut late = BasicAccumulator::new();
305        late.push(b"valid\nlines\n");
306        late.push(b"late\0binary");
307        assert_eq!(late.finish(), TextAdmission::Binary);
308
309        let mut invalid = BasicAccumulator::new();
310        invalid.push(&[b'a', 0xff]);
311        assert_eq!(invalid.finish(), TextAdmission::InvalidUtf8);
312    }
313
314    #[test]
315    fn basic_mode_keeps_raw_words_but_omits_deeper_document_metrics() {
316        let mut accumulator = BasicAccumulator::new();
317        accumulator.push(b"one two\n\nthree\n");
318        let TextAdmission::Accepted(metrics) = accumulator.finish() else {
319            panic!("expected accepted text");
320        };
321        assert_eq!(metrics.raw_words, 3);
322        assert_eq!(metrics.paragraphs, 0);
323        assert_eq!(metrics.logical_word_stats, super::super::LogicalWordStats::default());
324    }
325}