1use cpd_core::models::{Location, Token, TokenKind};
5
6#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7enum CommentStyle {
8 CStyle,
10 Hash,
12 DoubleDash,
14 Lua,
16 Semicolon,
18 VisualBasic,
20 #[allow(dead_code)]
22 None,
23}
24
25fn comment_style(format: &str) -> CommentStyle {
26 match format {
27 "c" | "c-header" | "cpp" | "cpp-header" | "csharp" | "java" | "go" | "rust" | "swift"
28 | "kotlin" | "scala" | "dart" | "php" | "typescript" | "jsx" | "tsx" | "javascript"
29 | "groovy" | "d" | "glsl" | "hlsl" | "wgsl" | "openqasm" | "solidity" | "bicep" | "hcl"
30 | "json5" | "less" | "scss" | "css" | "objectivec" | "protobuf" | "apex" | "verilog"
31 | "zig" | "odin" | "fsharp" | "actionscript" | "cfscript" => CommentStyle::CStyle,
32
33 "python" | "ruby" | "perl" | "bash" | "sh" | "zsh" | "fish" | "r" | "julia" | "yaml"
34 | "toml" | "dockerfile" | "makefile" | "cmake" | "coffeescript" | "crystal" | "nim"
35 | "gdscript" | "elixir" | "awk" | "tcl" | "powershell" | "puppet" | "ignore" => {
36 CommentStyle::Hash
37 }
38
39 "sql" | "haskell" | "elm" | "ada" | "plsql" => CommentStyle::DoubleDash,
40
41 "lua" => CommentStyle::Lua,
42
43 "ini" | "properties" | "asm6502" | "nasm" | "lisp" | "clojure" | "scheme" | "racket" => {
44 CommentStyle::Semicolon
45 }
46
47 "vb" | "vbs" | "basic" | "vbnet" | "visual-basic" => CommentStyle::VisualBasic,
48
49 _ => CommentStyle::CStyle,
50 }
51}
52
53fn is_ignore_start(text: &str) -> bool {
54 text.contains("jscpd:ignore-start")
55}
56
57fn is_ignore_end(text: &str) -> bool {
58 text.contains("jscpd:ignore-end")
59}
60
61fn comment_kind(in_ignore: bool) -> TokenKind {
62 if in_ignore {
63 TokenKind::Ignore
64 } else {
65 TokenKind::Comment
66 }
67}
68
69fn make_token(kind: TokenKind, value: &str, line: u32, col: u32, offset: u32) -> Token {
70 let len = value.len() as u32;
71 Token {
72 kind,
73 value: value.to_string(),
74 start: Location {
75 line,
76 column: col,
77 offset,
78 },
79 end: Location {
80 line,
81 column: col + len,
82 offset: offset + len,
83 },
84 }
85}
86
87fn classify_word(word: &str) -> TokenKind {
88 if word.chars().all(|c| c.is_ascii_digit()) {
89 return TokenKind::Literal;
90 }
91 if word.chars().all(|c| c.is_ascii_punctuation()) {
92 return TokenKind::Punctuation;
93 }
94 TokenKind::Identifier
95}
96
97fn tokenize_line_content(
98 line: &str,
99 line_num: u32,
100 line_offset: u32,
101 style: CommentStyle,
102 in_ignore: bool,
103 in_block_comment: &mut bool,
104) -> Vec<Token> {
105 let mut tokens = Vec::new();
106
107 let chars: Vec<(usize, char)> = line.char_indices().collect();
111 let n = chars.len();
112 let mut i = 0usize;
113
114 let mut col = 0u32;
116
117 macro_rules! offset {
118 () => {
119 line_offset + col
120 };
121 }
122
123 while i < n {
124 let (_, ch) = chars[i];
125
126 if *in_block_comment {
128 let kind = comment_kind(in_ignore);
129 if matches!(style, CommentStyle::CStyle)
130 && i + 1 < n
131 && ch == '*'
132 && chars[i + 1].1 == '/'
133 {
134 let start_col = col;
135 let start_off = offset!();
136 col += 2;
137 i += 2;
138 tokens.push(make_token(kind, "*/", line_num, start_col, start_off));
139 *in_block_comment = false;
140 continue;
141 }
142 let start_col = col;
144 let start_off = offset!();
145 let mut s = String::new();
146 s.push(ch);
147 col += ch.len_utf8() as u32;
148 i += 1;
149 tokens.push(make_token(kind, &s, line_num, start_col, start_off));
150 continue;
151 }
152
153 if matches!(style, CommentStyle::Lua)
155 && i + 3 < n
156 && ch == '-'
157 && chars[i + 1].1 == '-'
158 && chars[i + 2].1 == '['
159 && chars[i + 3].1 == '['
160 {
161 let rest = &line[chars[i].0..];
162 tokens.push(make_token(
163 comment_kind(in_ignore),
164 rest,
165 line_num,
166 col,
167 offset!(),
168 ));
169 break;
170 }
171
172 if matches!(style, CommentStyle::CStyle) && i + 1 < n && ch == '/' && chars[i + 1].1 == '*'
174 {
175 *in_block_comment = true;
176 let start_col = col;
177 let start_off = offset!();
178 col += 2;
179 i += 2;
180 tokens.push(make_token(
181 comment_kind(in_ignore),
182 "/*",
183 line_num,
184 start_col,
185 start_off,
186 ));
187 continue;
188 }
189
190 let is_comment = match style {
192 CommentStyle::CStyle => i + 1 < n && ch == '/' && chars[i + 1].1 == '/',
193 CommentStyle::Hash => ch == '#',
194 CommentStyle::DoubleDash | CommentStyle::Lua => {
195 i + 1 < n && ch == '-' && chars[i + 1].1 == '-'
196 }
197 CommentStyle::Semicolon => ch == ';',
198 CommentStyle::VisualBasic => ch == '\'',
199 CommentStyle::None => false,
200 };
201
202 if is_comment {
203 let rest = &line[chars[i].0..];
204 tokens.push(make_token(
205 comment_kind(in_ignore),
206 rest,
207 line_num,
208 col,
209 offset!(),
210 ));
211 break;
212 }
213
214 if ch == '"' || ch == '\'' {
216 let quote = ch;
217 let start_col = col;
218 let start_off = offset!();
219 let mut j = chars[i].0; let str_start = j;
221 col += 1;
222 i += 1;
223 j += 1;
224 while i < n && chars[i].1 != quote {
225 if chars[i].1 == '\\' && i + 1 < n {
226 col += chars[i].1.len_utf8() as u32 + chars[i + 1].1.len_utf8() as u32;
227 i += 2;
228 } else {
229 col += chars[i].1.len_utf8() as u32;
230 i += 1;
231 }
232 }
233 if i < n {
234 col += 1;
235 i += 1;
236 }
237 let str_end = if i < n {
238 chars[i - 1].0 + chars[i - 1].1.len_utf8()
239 } else {
240 line.len()
241 };
242 let _ = (j, str_start); let s = &line[str_start..str_end];
244 let kind = if in_ignore {
245 TokenKind::Ignore
246 } else {
247 TokenKind::Literal
248 };
249 tokens.push(make_token(kind, s, line_num, start_col, start_off));
250 continue;
251 }
252
253 if ch.is_whitespace() {
255 let start_col = col;
256 let start_off = offset!();
257 let byte_start = chars[i].0;
258 while i < n && chars[i].1.is_whitespace() {
259 col += chars[i].1.len_utf8() as u32;
260 i += 1;
261 }
262 let byte_end = if i < n { chars[i].0 } else { line.len() };
263 let kind = if in_ignore {
264 TokenKind::Ignore
265 } else {
266 TokenKind::Whitespace
267 };
268 tokens.push(make_token(
269 kind,
270 &line[byte_start..byte_end],
271 line_num,
272 start_col,
273 start_off,
274 ));
275 continue;
276 }
277
278 if ch.is_ascii_digit() {
280 let start_col = col;
281 let start_off = offset!();
282 let byte_start = chars[i].0;
283 while i < n && (chars[i].1.is_ascii_digit() || chars[i].1 == '.') {
284 col += 1;
285 i += 1;
286 }
287 let byte_end = if i < n { chars[i].0 } else { line.len() };
288 let kind = if in_ignore {
289 TokenKind::Ignore
290 } else {
291 TokenKind::Literal
292 };
293 tokens.push(make_token(
294 kind,
295 &line[byte_start..byte_end],
296 line_num,
297 start_col,
298 start_off,
299 ));
300 continue;
301 }
302
303 if ch.is_alphabetic() || ch == '_' {
305 let start_col = col;
306 let start_off = offset!();
307 let byte_start = chars[i].0;
308 while i < n && (chars[i].1.is_alphanumeric() || chars[i].1 == '_') {
309 col += chars[i].1.len_utf8() as u32;
310 i += 1;
311 }
312 let byte_end = if i < n { chars[i].0 } else { line.len() };
313 let s = &line[byte_start..byte_end];
314 let kind = if in_ignore {
315 TokenKind::Ignore
316 } else {
317 classify_word(s)
318 };
319 tokens.push(make_token(kind, s, line_num, start_col, start_off));
320 continue;
321 }
322
323 let start_col = col;
325 let start_off = offset!();
326 let byte_start = chars[i].0;
327 col += ch.len_utf8() as u32;
328 i += 1;
329 let byte_end = if i < n { chars[i].0 } else { line.len() };
330 let kind = if in_ignore {
331 TokenKind::Ignore
332 } else {
333 TokenKind::Punctuation
334 };
335 tokens.push(make_token(
336 kind,
337 &line[byte_start..byte_end],
338 line_num,
339 start_col,
340 start_off,
341 ));
342 }
343
344 tokens
345}
346
347pub fn tokenize_generic(source: &str, format: &str) -> Vec<Token> {
349 if source.is_empty() {
350 return Vec::new();
351 }
352
353 let style = comment_style(format);
354 let mut tokens = Vec::new();
355 let mut in_ignore = false;
356 let mut in_block_comment = false;
357 let mut offset = 0u32;
358
359 for (line_idx, line) in source.lines().enumerate() {
360 let line_num = line_idx as u32 + 1;
361 let trimmed = line.trim();
362
363 if is_ignore_start(trimmed) {
364 in_ignore = true;
365 }
366 if is_ignore_end(trimmed) {
367 in_ignore = false;
368 offset += line.len() as u32 + 1;
370 continue;
371 }
372
373 let line_tokens = tokenize_line_content(
374 line,
375 line_num,
376 offset,
377 style,
378 in_ignore,
379 &mut in_block_comment,
380 );
381 tokens.extend(line_tokens);
382 offset += line.len() as u32 + 1;
383 }
384
385 tokens
386}
387
388#[cfg(test)]
389mod tests {
390 use super::*;
391
392 #[test]
393 fn python_produces_tokens() {
394 let tokens = tokenize_generic("def hello():\n return 42\n", "python");
395 assert!(!tokens.is_empty());
396 }
397
398 #[test]
399 fn python_hash_comment_marked_as_comment() {
400 let tokens = tokenize_generic("# this is a comment\nx = 1\n", "python");
401 let has_comment = tokens.iter().any(|t| t.kind == TokenKind::Comment);
402 assert!(has_comment, "Python # comments must be Comment kind");
403 }
404
405 #[test]
406 fn lisp_family_semicolon_comment_marked_as_comment() {
407 for format in ["lisp", "clojure", "scheme", "racket"] {
408 let tokens = tokenize_generic(";; a comment\n(def x 1)\n", format);
409 let has_comment = tokens.iter().any(|t| t.kind == TokenKind::Comment);
410 assert!(has_comment, "{format} ; comments must be Comment kind");
411 }
412 }
413
414 #[test]
415 fn go_c_style_comment_recognized() {
416 let tokens = tokenize_generic("// hello\nfunc main() {}\n", "go");
417 let has_comment = tokens.iter().any(|t| t.kind == TokenKind::Comment);
418 assert!(has_comment);
419 }
420
421 #[test]
422 fn empty_input_returns_empty() {
423 let tokens = tokenize_generic("", "python");
424 assert!(
425 tokens.is_empty(),
426 "empty input must return empty vec, not panic"
427 );
428 }
429
430 #[test]
431 fn unknown_format_does_not_panic() {
432 let result =
433 std::panic::catch_unwind(|| tokenize_generic("hello world", "unknown_format_xyz"));
434 assert!(result.is_ok());
435 }
436
437 #[test]
438 fn ignore_region_tokens_marked_as_ignore() {
439 let source = "x = 1\n# jscpd:ignore-start\ny = 2\n# jscpd:ignore-end\nz = 3\n";
440 let tokens = tokenize_generic(source, "python");
441 let has_ignore = tokens.iter().any(|t| t.kind == TokenKind::Ignore);
442 assert!(has_ignore, "tokens in ignore region must be Ignore kind");
443 }
444
445 #[test]
446 fn sql_double_dash_comment_recognized() {
447 let tokens = tokenize_generic("-- a comment\nSELECT * FROM foo;\n", "sql");
448 let has_comment = tokens.iter().any(|t| t.kind == TokenKind::Comment);
449 assert!(has_comment);
450 }
451
452 #[test]
453 fn c_block_comment_recognized() {
454 let tokens = tokenize_generic("/* block */\nint x = 1;\n", "c");
455 let has_comment = tokens.iter().any(|t| t.kind == TokenKind::Comment);
456 assert!(has_comment);
457 }
458
459 #[test]
460 fn location_line_numbers_are_1_based() {
461 let tokens = tokenize_generic("x = 1\ny = 2\n", "python");
462 let first = tokens.first().expect("at least one token");
463 assert_eq!(first.start.line, 1);
464 }
465}