Skip to main content

cpd_tokenizer/
generic.rs

1// cpd-tokenizer: generic whitespace-and-punctuation tokenizer for non-JS/TS formats.
2// Handles comment styles, ignore regions, and per-line token scanning without regex.
3
4use cpd_core::models::{Location, Token, TokenKind};
5
6#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7enum CommentStyle {
8    /// Single-line `//`, block `/* */`
9    CStyle,
10    /// Single-line `#`
11    Hash,
12    /// Single-line `--`
13    DoubleDash,
14    /// Single-line `--`, block `--[[ ]]`
15    Lua,
16    /// Single-line `;`
17    Semicolon,
18    /// Single-line `'`
19    VisualBasic,
20    /// No comments (fallback for unrecognised formats)
21    #[allow(dead_code)]
22    None,
23}
24
25fn comment_style(format: &str) -> CommentStyle {
26    match format {
27        "c" | "c-header" | "cpp" | "cpp-header" | "csharp" | "java" | "go" | "rust" | "swift"
28        | "kotlin" | "scala" | "dart" | "php" | "typescript" | "jsx" | "tsx" | "javascript"
29        | "groovy" | "d" | "glsl" | "hlsl" | "wgsl" | "openqasm" | "solidity" | "bicep" | "hcl"
30        | "json5" | "less" | "scss" | "css" | "objectivec" | "protobuf" | "apex" | "verilog"
31        | "zig" | "odin" | "fsharp" | "actionscript" | "cfscript" => CommentStyle::CStyle,
32
33        "python" | "ruby" | "perl" | "bash" | "sh" | "zsh" | "fish" | "r" | "julia" | "yaml"
34        | "toml" | "dockerfile" | "makefile" | "cmake" | "coffeescript" | "crystal" | "nim"
35        | "gdscript" | "elixir" | "awk" | "tcl" | "powershell" | "puppet" | "ignore" => {
36            CommentStyle::Hash
37        }
38
39        "sql" | "haskell" | "elm" | "ada" | "plsql" => CommentStyle::DoubleDash,
40
41        "lua" => CommentStyle::Lua,
42
43        "ini" | "properties" | "asm6502" | "nasm" | "lisp" | "clojure" | "scheme" | "racket" => {
44            CommentStyle::Semicolon
45        }
46
47        "vb" | "vbs" | "basic" | "vbnet" | "visual-basic" => CommentStyle::VisualBasic,
48
49        _ => CommentStyle::CStyle,
50    }
51}
52
53fn is_ignore_start(text: &str) -> bool {
54    text.contains("jscpd:ignore-start")
55}
56
57fn is_ignore_end(text: &str) -> bool {
58    text.contains("jscpd:ignore-end")
59}
60
61fn comment_kind(in_ignore: bool) -> TokenKind {
62    if in_ignore {
63        TokenKind::Ignore
64    } else {
65        TokenKind::Comment
66    }
67}
68
69fn make_token(kind: TokenKind, value: &str, line: u32, col: u32, offset: u32) -> Token {
70    let len = value.len() as u32;
71    Token {
72        kind,
73        value: value.to_string(),
74        start: Location {
75            line,
76            column: col,
77            offset,
78        },
79        end: Location {
80            line,
81            column: col + len,
82            offset: offset + len,
83        },
84    }
85}
86
87fn classify_word(word: &str) -> TokenKind {
88    if word.chars().all(|c| c.is_ascii_digit()) {
89        return TokenKind::Literal;
90    }
91    if word.chars().all(|c| c.is_ascii_punctuation()) {
92        return TokenKind::Punctuation;
93    }
94    TokenKind::Identifier
95}
96
97fn tokenize_line_content(
98    line: &str,
99    line_num: u32,
100    line_offset: u32,
101    style: CommentStyle,
102    in_ignore: bool,
103    in_block_comment: &mut bool,
104) -> Vec<Token> {
105    let mut tokens = Vec::new();
106
107    // Collect (byte_offset, char) pairs once — zero heap allocation vs chars().collect().
108    // `char_indices()` returns (byte_index, char) which gives us correct UTF-8 byte offsets
109    // for column accounting while avoiding a Vec<char> heap allocation per line.
110    let chars: Vec<(usize, char)> = line.char_indices().collect();
111    let n = chars.len();
112    let mut i = 0usize;
113
114    // col is in bytes (UTF-8 units), consistent with char.len_utf8() increments below.
115    let mut col = 0u32;
116
117    macro_rules! offset {
118        () => {
119            line_offset + col
120        };
121    }
122
123    while i < n {
124        let (_, ch) = chars[i];
125
126        // Handle block comment end
127        if *in_block_comment {
128            let kind = comment_kind(in_ignore);
129            if matches!(style, CommentStyle::CStyle)
130                && i + 1 < n
131                && ch == '*'
132                && chars[i + 1].1 == '/'
133            {
134                let start_col = col;
135                let start_off = offset!();
136                col += 2;
137                i += 2;
138                tokens.push(make_token(kind, "*/", line_num, start_col, start_off));
139                *in_block_comment = false;
140                continue;
141            }
142            // Still inside block comment — consume char
143            let start_col = col;
144            let start_off = offset!();
145            let mut s = String::new();
146            s.push(ch);
147            col += ch.len_utf8() as u32;
148            i += 1;
149            tokens.push(make_token(kind, &s, line_num, start_col, start_off));
150            continue;
151        }
152
153        // Lua long block comment --[[
154        if matches!(style, CommentStyle::Lua)
155            && i + 3 < n
156            && ch == '-'
157            && chars[i + 1].1 == '-'
158            && chars[i + 2].1 == '['
159            && chars[i + 3].1 == '['
160        {
161            let rest = &line[chars[i].0..];
162            tokens.push(make_token(
163                comment_kind(in_ignore),
164                rest,
165                line_num,
166                col,
167                offset!(),
168            ));
169            break;
170        }
171
172        // C-style block comment open /*
173        if matches!(style, CommentStyle::CStyle) && i + 1 < n && ch == '/' && chars[i + 1].1 == '*'
174        {
175            *in_block_comment = true;
176            let start_col = col;
177            let start_off = offset!();
178            col += 2;
179            i += 2;
180            tokens.push(make_token(
181                comment_kind(in_ignore),
182                "/*",
183                line_num,
184                start_col,
185                start_off,
186            ));
187            continue;
188        }
189
190        // Line comment — check current position directly without allocating
191        let is_comment = match style {
192            CommentStyle::CStyle => i + 1 < n && ch == '/' && chars[i + 1].1 == '/',
193            CommentStyle::Hash => ch == '#',
194            CommentStyle::DoubleDash | CommentStyle::Lua => {
195                i + 1 < n && ch == '-' && chars[i + 1].1 == '-'
196            }
197            CommentStyle::Semicolon => ch == ';',
198            CommentStyle::VisualBasic => ch == '\'',
199            CommentStyle::None => false,
200        };
201
202        if is_comment {
203            let rest = &line[chars[i].0..];
204            tokens.push(make_token(
205                comment_kind(in_ignore),
206                rest,
207                line_num,
208                col,
209                offset!(),
210            ));
211            break;
212        }
213
214        // String literals (double-quote or single-quote)
215        if ch == '"' || ch == '\'' {
216            let quote = ch;
217            let start_col = col;
218            let start_off = offset!();
219            let mut j = chars[i].0; // byte start of string in `line`
220            let str_start = j;
221            col += 1;
222            i += 1;
223            j += 1;
224            while i < n && chars[i].1 != quote {
225                if chars[i].1 == '\\' && i + 1 < n {
226                    col += chars[i].1.len_utf8() as u32 + chars[i + 1].1.len_utf8() as u32;
227                    i += 2;
228                } else {
229                    col += chars[i].1.len_utf8() as u32;
230                    i += 1;
231                }
232            }
233            if i < n {
234                col += 1;
235                i += 1;
236            }
237            let str_end = if i < n {
238                chars[i - 1].0 + chars[i - 1].1.len_utf8()
239            } else {
240                line.len()
241            };
242            let _ = (j, str_start); // byte indices computed above but using slice below
243            let s = &line[str_start..str_end];
244            let kind = if in_ignore {
245                TokenKind::Ignore
246            } else {
247                TokenKind::Literal
248            };
249            tokens.push(make_token(kind, s, line_num, start_col, start_off));
250            continue;
251        }
252
253        // Whitespace
254        if ch.is_whitespace() {
255            let start_col = col;
256            let start_off = offset!();
257            let byte_start = chars[i].0;
258            while i < n && chars[i].1.is_whitespace() {
259                col += chars[i].1.len_utf8() as u32;
260                i += 1;
261            }
262            let byte_end = if i < n { chars[i].0 } else { line.len() };
263            let kind = if in_ignore {
264                TokenKind::Ignore
265            } else {
266                TokenKind::Whitespace
267            };
268            tokens.push(make_token(
269                kind,
270                &line[byte_start..byte_end],
271                line_num,
272                start_col,
273                start_off,
274            ));
275            continue;
276        }
277
278        // Numbers
279        if ch.is_ascii_digit() {
280            let start_col = col;
281            let start_off = offset!();
282            let byte_start = chars[i].0;
283            while i < n && (chars[i].1.is_ascii_digit() || chars[i].1 == '.') {
284                col += 1;
285                i += 1;
286            }
287            let byte_end = if i < n { chars[i].0 } else { line.len() };
288            let kind = if in_ignore {
289                TokenKind::Ignore
290            } else {
291                TokenKind::Literal
292            };
293            tokens.push(make_token(
294                kind,
295                &line[byte_start..byte_end],
296                line_num,
297                start_col,
298                start_off,
299            ));
300            continue;
301        }
302
303        // Identifiers / keywords
304        if ch.is_alphabetic() || ch == '_' {
305            let start_col = col;
306            let start_off = offset!();
307            let byte_start = chars[i].0;
308            while i < n && (chars[i].1.is_alphanumeric() || chars[i].1 == '_') {
309                col += chars[i].1.len_utf8() as u32;
310                i += 1;
311            }
312            let byte_end = if i < n { chars[i].0 } else { line.len() };
313            let s = &line[byte_start..byte_end];
314            let kind = if in_ignore {
315                TokenKind::Ignore
316            } else {
317                classify_word(s)
318            };
319            tokens.push(make_token(kind, s, line_num, start_col, start_off));
320            continue;
321        }
322
323        // Operators / punctuation (single char)
324        let start_col = col;
325        let start_off = offset!();
326        let byte_start = chars[i].0;
327        col += ch.len_utf8() as u32;
328        i += 1;
329        let byte_end = if i < n { chars[i].0 } else { line.len() };
330        let kind = if in_ignore {
331            TokenKind::Ignore
332        } else {
333            TokenKind::Punctuation
334        };
335        tokens.push(make_token(
336            kind,
337            &line[byte_start..byte_end],
338            line_num,
339            start_col,
340            start_off,
341        ));
342    }
343
344    tokens
345}
346
347/// Tokenize source in the given format. Never panics on empty input.
348pub fn tokenize_generic(source: &str, format: &str) -> Vec<Token> {
349    if source.is_empty() {
350        return Vec::new();
351    }
352
353    let style = comment_style(format);
354    let mut tokens = Vec::new();
355    let mut in_ignore = false;
356    let mut in_block_comment = false;
357    let mut offset = 0u32;
358
359    for (line_idx, line) in source.lines().enumerate() {
360        let line_num = line_idx as u32 + 1;
361        let trimmed = line.trim();
362
363        if is_ignore_start(trimmed) {
364            in_ignore = true;
365        }
366        if is_ignore_end(trimmed) {
367            in_ignore = false;
368            // Advance offset past this line and continue
369            offset += line.len() as u32 + 1;
370            continue;
371        }
372
373        let line_tokens = tokenize_line_content(
374            line,
375            line_num,
376            offset,
377            style,
378            in_ignore,
379            &mut in_block_comment,
380        );
381        tokens.extend(line_tokens);
382        offset += line.len() as u32 + 1;
383    }
384
385    tokens
386}
387
388#[cfg(test)]
389mod tests {
390    use super::*;
391
392    #[test]
393    fn python_produces_tokens() {
394        let tokens = tokenize_generic("def hello():\n    return 42\n", "python");
395        assert!(!tokens.is_empty());
396    }
397
398    #[test]
399    fn python_hash_comment_marked_as_comment() {
400        let tokens = tokenize_generic("# this is a comment\nx = 1\n", "python");
401        let has_comment = tokens.iter().any(|t| t.kind == TokenKind::Comment);
402        assert!(has_comment, "Python # comments must be Comment kind");
403    }
404
405    #[test]
406    fn lisp_family_semicolon_comment_marked_as_comment() {
407        for format in ["lisp", "clojure", "scheme", "racket"] {
408            let tokens = tokenize_generic(";; a comment\n(def x 1)\n", format);
409            let has_comment = tokens.iter().any(|t| t.kind == TokenKind::Comment);
410            assert!(has_comment, "{format} ; comments must be Comment kind");
411        }
412    }
413
414    #[test]
415    fn go_c_style_comment_recognized() {
416        let tokens = tokenize_generic("// hello\nfunc main() {}\n", "go");
417        let has_comment = tokens.iter().any(|t| t.kind == TokenKind::Comment);
418        assert!(has_comment);
419    }
420
421    #[test]
422    fn empty_input_returns_empty() {
423        let tokens = tokenize_generic("", "python");
424        assert!(
425            tokens.is_empty(),
426            "empty input must return empty vec, not panic"
427        );
428    }
429
430    #[test]
431    fn unknown_format_does_not_panic() {
432        let result =
433            std::panic::catch_unwind(|| tokenize_generic("hello world", "unknown_format_xyz"));
434        assert!(result.is_ok());
435    }
436
437    #[test]
438    fn ignore_region_tokens_marked_as_ignore() {
439        let source = "x = 1\n# jscpd:ignore-start\ny = 2\n# jscpd:ignore-end\nz = 3\n";
440        let tokens = tokenize_generic(source, "python");
441        let has_ignore = tokens.iter().any(|t| t.kind == TokenKind::Ignore);
442        assert!(has_ignore, "tokens in ignore region must be Ignore kind");
443    }
444
445    #[test]
446    fn sql_double_dash_comment_recognized() {
447        let tokens = tokenize_generic("-- a comment\nSELECT * FROM foo;\n", "sql");
448        let has_comment = tokens.iter().any(|t| t.kind == TokenKind::Comment);
449        assert!(has_comment);
450    }
451
452    #[test]
453    fn c_block_comment_recognized() {
454        let tokens = tokenize_generic("/* block */\nint x = 1;\n", "c");
455        let has_comment = tokens.iter().any(|t| t.kind == TokenKind::Comment);
456        assert!(has_comment);
457    }
458
459    #[test]
460    fn location_line_numbers_are_1_based() {
461        let tokens = tokenize_generic("x = 1\ny = 2\n", "python");
462        let first = tokens.first().expect("at least one token");
463        assert_eq!(first.start.line, 1);
464    }
465}