1use std::panic::{AssertUnwindSafe, catch_unwind};
5use std::path::Path;
6
7use oxc_allocator::Allocator;
8use oxc_parser::{Kind, Parser, config::TokensParserConfig};
9use oxc_span::SourceType;
10
11use cpd_core::models::{Token, TokenKind};
12
13use crate::line_index::LineIndex;
14
15mod fallback {
18 use super::{LineIndex, find_ignore_ranges, in_ignore};
19 use cpd_core::models::{Token, TokenKind};
20
21 pub fn tokenize(source: &str, _format: &str) -> Vec<Token> {
23 let ignore_ranges = find_ignore_ranges(source);
24 let bytes = source.as_bytes();
25 let line_index = LineIndex::new(bytes);
26 let mut tokens = Vec::new();
27 let mut i = 0;
28 while i < bytes.len() {
29 let ch = match source[i..].chars().next() {
30 Some(c) => c,
31 None => break,
32 };
33 if ch.is_whitespace() {
34 i += ch.len_utf8();
35 continue;
36 }
37 if ch.is_alphanumeric() || ch == '_' || ch == '$' {
38 let start = i;
39 while i < bytes.len() {
40 let c = source[i..].chars().next().unwrap_or('\0');
41 if c.is_alphanumeric() || c == '_' || c == '$' {
42 i += c.len_utf8();
43 } else {
44 break;
45 }
46 }
47 let kind = if in_ignore(start, i, &ignore_ranges) {
48 TokenKind::Ignore
49 } else {
50 TokenKind::Other
51 };
52 tokens.push(Token {
53 kind,
54 value: source[start..i].to_string(),
55 start: line_index.location(start),
56 end: line_index.location(i),
57 });
58 } else {
59 let start = i;
60 i += ch.len_utf8();
61 let kind = if in_ignore(start, i, &ignore_ranges) {
62 TokenKind::Ignore
63 } else {
64 TokenKind::Other
65 };
66 tokens.push(Token {
67 kind,
68 value: ch.to_string(),
69 start: line_index.location(start),
70 end: line_index.location(i),
71 });
72 }
73 }
74 tokens
75 }
76}
77
78fn find_ignore_ranges(source: &str) -> Vec<[usize; 2]> {
81 let mut ranges = Vec::new();
82 let mut start: Option<usize> = None;
83 let bytes = source.as_bytes();
84 let mut i = 0;
85 while i < bytes.len() {
86 if i + 1 < bytes.len() && bytes[i] == b'/' {
87 let end = if bytes[i + 1] == b'/' {
88 bytes[i..]
89 .iter()
90 .position(|&b| b == b'\n')
91 .map(|p| i + p)
92 .unwrap_or(bytes.len())
93 } else if bytes[i + 1] == b'*' {
94 bytes[i..]
95 .windows(2)
96 .position(|w| w == b"*/")
97 .map(|p| i + p + 2)
98 .unwrap_or(bytes.len())
99 } else {
100 i += 1;
101 continue;
102 };
103 let comment_text = &source[i..end];
104 if comment_text.contains("jscpd:ignore-start") {
105 start = Some(end);
106 } else if comment_text.contains("jscpd:ignore-end")
107 && let Some(s) = start.take()
108 {
109 ranges.push([s, i]);
110 }
111 i = end;
112 continue;
113 }
114 i += 1;
115 }
116 ranges
117}
118
119fn in_ignore(offset: usize, end: usize, ranges: &[[usize; 2]]) -> bool {
120 ranges.iter().any(|[rs, re]| offset < *re && end > *rs)
121}
122
123const fn map_kind(kind: Kind) -> TokenKind {
124 if matches!(kind, Kind::Ident) {
125 return TokenKind::Identifier;
126 }
127 if kind.is_any_keyword() {
128 return TokenKind::Keyword;
129 }
130 if kind.is_literal() {
131 return TokenKind::Literal;
132 }
133 if kind.is_assignment_operator() {
134 return TokenKind::Operator;
135 }
136 if kind.is_binary_operator()
137 || kind.is_logical_operator()
138 || kind.is_unary_operator()
139 || kind.is_update_operator()
140 {
141 return TokenKind::Operator;
142 }
143 match kind {
144 Kind::Arrow => TokenKind::Operator,
145 Kind::Semicolon
146 | Kind::Comma
147 | Kind::Dot
148 | Kind::Dot3
149 | Kind::Colon
150 | Kind::LParen
151 | Kind::RParen
152 | Kind::LCurly
153 | Kind::RCurly
154 | Kind::LBrack
155 | Kind::RBrack
156 | Kind::At => TokenKind::Punctuation,
157 Kind::QuestionDot => TokenKind::Punctuation,
158 _ => TokenKind::Other,
159 }
160}
161
162pub(crate) fn source_type_for_format(format: &str) -> SourceType {
163 let filename = match format {
164 "typescript" => "input.ts",
165 "tsx" => "input.tsx",
166 _ => "input.jsx", };
168 SourceType::from_path(Path::new(filename)).unwrap_or_default()
169}
170
171pub fn tokenize_js(source: &str, format: &str) -> Vec<Token> {
175 tokenize_js_impl(source, format, false)
176}
177
178pub fn tokenize_js_stripped(source: &str, format: &str) -> Vec<Token> {
187 tokenize_js_impl(source, format, true)
188}
189
190fn tokenize_js_impl(source: &str, format: &str, strip_types: bool) -> Vec<Token> {
191 if source.is_empty() {
192 return Vec::new();
193 }
194
195 match catch_unwind(AssertUnwindSafe(|| {
196 parse_with_oxc(source, format, strip_types)
197 })) {
198 Ok(Some(tokens)) => tokens,
199 Ok(None) => {
200 log::debug!("cpd-tokenizer: OXC parse errors in {format} source, using fallback");
201 fallback::tokenize(source, format)
202 }
203 Err(_) => {
204 log::debug!("cpd-tokenizer: OXC panicked on {format} source, using fallback");
205 fallback::tokenize(source, format)
206 }
207 }
208}
209
210fn parse_with_oxc(source: &str, format: &str, strip_types: bool) -> Option<Vec<Token>> {
211 let allocator = Allocator::new();
212 let source_type = source_type_for_format(format);
213
214 let parser_return = Parser::new(&allocator, source, source_type)
215 .with_config(TokensParserConfig)
216 .parse();
217
218 if parser_return.fatal_error || parser_return.tokens.is_empty() {
225 return None;
226 }
227
228 let erasable = if strip_types {
229 crate::ts_strip::collect_erasable_spans(&parser_return.program, source)
230 } else {
231 crate::ts_strip::ErasableSpans::default()
232 };
233
234 let ignore_ranges = find_ignore_ranges(source);
235 let bytes = source.as_bytes();
236 let line_index = LineIndex::new(bytes);
238
239 let tokens = parser_return
240 .tokens
241 .into_iter()
242 .filter_map(|token| {
243 let start = (token.start() as usize).min(source.len());
244 let end = (token.end() as usize).min(source.len());
245 if start >= end {
246 return None;
247 }
248 let kind = token.kind();
249 if matches!(kind, Kind::Eof | Kind::Undetermined | Kind::Skip) {
250 return None;
251 }
252 let value = &source[start..end];
253 if !erasable.is_empty()
254 && (erasable.intersects_range(start as u32, end as u32)
255 || erasable.is_modifier_token(start as u32, end as u32, value))
256 {
257 return None;
258 }
259 let token_kind = if in_ignore(start, end, &ignore_ranges) {
260 TokenKind::Ignore
261 } else {
262 map_kind(kind)
263 };
264
265 Some(Token {
266 kind: token_kind,
267 value: value.to_string(),
268 start: line_index.location(start),
269 end: line_index.location(end),
270 })
271 })
272 .collect::<Vec<Token>>();
273
274 Some(tokens)
275}
276
277#[cfg(test)]
279mod tests {
280 use super::*;
281
282 #[test]
286 fn parse_diagnostics_keep_the_oxc_token_stream() {
287 let clean = "export function save(row, db) {\n db.put(row.id, row);\n return row;\n}\n";
288 let twice = format!("{clean}\n{clean}");
289 let a = tokenize_js(clean, "javascript");
290 let b = tokenize_js(&twice, "javascript");
291 assert!(a.len() > 10);
292 assert_eq!(b.len(), a.len() * 2, "both copies lexed by oxc");
293 assert_eq!(
294 a.iter().map(|t| (&t.kind, &t.value)).collect::<Vec<_>>(),
295 b[..a.len()]
296 .iter()
297 .map(|t| (&t.kind, &t.value))
298 .collect::<Vec<_>>(),
299 "the first copy must tokenize exactly like the clean file"
300 );
301 assert_eq!(
302 b[0].kind,
303 TokenKind::Keyword,
304 "`export` classified by oxc, not word-split"
305 );
306 }
307
308 #[test]
309 fn empty_and_unlexable_sources_still_fall_back_safely() {
310 assert!(tokenize_js("", "javascript").is_empty());
311 assert!(!tokenize_js("function (", "javascript").is_empty());
313 }
314
315 #[test]
316 fn valid_js_produces_tokens() {
317 let tokens = tokenize_js("function hello() { return 42; }", "javascript");
318 assert!(!tokens.is_empty(), "valid JS must produce tokens");
319 }
320
321 #[test]
322 fn typescript_produces_tokens() {
323 let tokens = tokenize_js("const x: number = 5;", "typescript");
324 assert!(!tokens.is_empty());
325 }
326
327 #[test]
328 fn malformed_js_does_not_panic() {
329 let result = std::panic::catch_unwind(|| tokenize_js("let x = {{{", "javascript"));
330 assert!(result.is_ok(), "malformed JS must not panic");
331 }
332
333 #[test]
334 fn empty_source_returns_empty() {
335 let tokens = tokenize_js("", "javascript");
336 drop(tokens);
337 }
338
339 #[test]
340 fn ignore_region_tokens_marked_as_ignore() {
341 let source = r#"
342const a = 1;
343// jscpd:ignore-start
344const b = 2;
345// jscpd:ignore-end
346const c = 3;
347"#;
348 let tokens = tokenize_js(source, "javascript");
349 let has_ignore = tokens
350 .iter()
351 .any(|t| t.kind == cpd_core::models::TokenKind::Ignore);
352 assert!(has_ignore, "tokens in ignore region must be marked Ignore");
353 }
354
355 #[test]
356 fn jsx_produces_tokens() {
357 let tokens = tokenize_js("const el = <div>hello</div>;", "jsx");
358 assert!(!tokens.is_empty());
359 }
360
361 #[test]
362 fn tsx_with_type_annotation() {
363 let tokens = tokenize_js("const fn = (x: React.FC): void => {};", "tsx");
364 assert!(!tokens.is_empty());
365 }
366
367 #[test]
368 fn multiline_location_uses_binary_search() {
369 let source = "const a = 1;\nconst b = 2;\nconst c = 3;";
370 let tokens = tokenize_js(source, "javascript");
371 let b_token = tokens.iter().find(|t| t.value == "b");
373 assert!(b_token.is_some(), "must find token b");
374 assert_eq!(b_token.unwrap().start.line, 2, "b must be on line 2");
375 }
376}