cpd_tokenizer/
javascript.rs1use std::panic::{AssertUnwindSafe, catch_unwind};
5use std::path::Path;
6
7use oxc_allocator::Allocator;
8use oxc_parser::{Kind, Parser, config::TokensParserConfig};
9use oxc_span::SourceType;
10
11use cpd_core::models::{Token, TokenKind};
12
13use crate::line_index::LineIndex;
14
15mod fallback {
18 use super::{LineIndex, find_ignore_ranges, in_ignore};
19 use cpd_core::models::{Token, TokenKind};
20
21 pub fn tokenize(source: &str, _format: &str) -> Vec<Token> {
23 let ignore_ranges = find_ignore_ranges(source);
24 let bytes = source.as_bytes();
25 let line_index = LineIndex::new(bytes);
26 let mut tokens = Vec::new();
27 let mut i = 0;
28 while i < bytes.len() {
29 let ch = match source[i..].chars().next() {
30 Some(c) => c,
31 None => break,
32 };
33 if ch.is_whitespace() {
34 i += ch.len_utf8();
35 continue;
36 }
37 if ch.is_alphanumeric() || ch == '_' || ch == '$' {
38 let start = i;
39 while i < bytes.len() {
40 let c = source[i..].chars().next().unwrap_or('\0');
41 if c.is_alphanumeric() || c == '_' || c == '$' {
42 i += c.len_utf8();
43 } else {
44 break;
45 }
46 }
47 let kind = if in_ignore(start, i, &ignore_ranges) {
48 TokenKind::Ignore
49 } else {
50 TokenKind::Other
51 };
52 tokens.push(Token {
53 kind,
54 value: source[start..i].to_string(),
55 start: line_index.location(start),
56 end: line_index.location(i),
57 });
58 } else {
59 let start = i;
60 i += ch.len_utf8();
61 let kind = if in_ignore(start, i, &ignore_ranges) {
62 TokenKind::Ignore
63 } else {
64 TokenKind::Other
65 };
66 tokens.push(Token {
67 kind,
68 value: ch.to_string(),
69 start: line_index.location(start),
70 end: line_index.location(i),
71 });
72 }
73 }
74 tokens
75 }
76}
77
78fn find_ignore_ranges(source: &str) -> Vec<[usize; 2]> {
81 let mut ranges = Vec::new();
82 let mut start: Option<usize> = None;
83 let bytes = source.as_bytes();
84 let mut i = 0;
85 while i < bytes.len() {
86 if i + 1 < bytes.len() && bytes[i] == b'/' {
87 let end = if bytes[i + 1] == b'/' {
88 bytes[i..]
89 .iter()
90 .position(|&b| b == b'\n')
91 .map(|p| i + p)
92 .unwrap_or(bytes.len())
93 } else if bytes[i + 1] == b'*' {
94 bytes[i..]
95 .windows(2)
96 .position(|w| w == b"*/")
97 .map(|p| i + p + 2)
98 .unwrap_or(bytes.len())
99 } else {
100 i += 1;
101 continue;
102 };
103 let comment_text = &source[i..end];
104 if comment_text.contains("jscpd:ignore-start") {
105 start = Some(end);
106 } else if comment_text.contains("jscpd:ignore-end") {
107 if let Some(s) = start.take() {
108 ranges.push([s, i]);
109 }
110 }
111 i = end;
112 continue;
113 }
114 i += 1;
115 }
116 ranges
117}
118
119fn in_ignore(offset: usize, end: usize, ranges: &[[usize; 2]]) -> bool {
120 ranges.iter().any(|[rs, re]| offset < *re && end > *rs)
121}
122
123const fn map_kind(kind: Kind) -> TokenKind {
124 if matches!(kind, Kind::Ident) {
125 return TokenKind::Identifier;
126 }
127 if kind.is_any_keyword() {
128 return TokenKind::Keyword;
129 }
130 if kind.is_literal() {
131 return TokenKind::Literal;
132 }
133 if kind.is_assignment_operator() {
134 return TokenKind::Operator;
135 }
136 if kind.is_binary_operator()
137 || kind.is_logical_operator()
138 || kind.is_unary_operator()
139 || kind.is_update_operator()
140 {
141 return TokenKind::Operator;
142 }
143 match kind {
144 Kind::Arrow => TokenKind::Operator,
145 Kind::Semicolon
146 | Kind::Comma
147 | Kind::Dot
148 | Kind::Dot3
149 | Kind::Colon
150 | Kind::LParen
151 | Kind::RParen
152 | Kind::LCurly
153 | Kind::RCurly
154 | Kind::LBrack
155 | Kind::RBrack
156 | Kind::At => TokenKind::Punctuation,
157 Kind::QuestionDot => TokenKind::Punctuation,
158 _ => TokenKind::Other,
159 }
160}
161
162fn source_type_for_format(format: &str) -> SourceType {
163 let filename = match format {
164 "typescript" => "input.ts",
165 "tsx" => "input.tsx",
166 _ => "input.jsx", };
168 SourceType::from_path(Path::new(filename)).unwrap_or_default()
169}
170
171pub fn tokenize_js(source: &str, format: &str) -> Vec<Token> {
175 if source.is_empty() {
176 return Vec::new();
177 }
178
179 match catch_unwind(AssertUnwindSafe(|| parse_with_oxc(source, format))) {
180 Ok(Some(tokens)) => tokens,
181 Ok(None) => {
182 log::debug!("cpd-tokenizer: OXC parse errors in {format} source, using fallback");
183 fallback::tokenize(source, format)
184 }
185 Err(_) => {
186 log::debug!("cpd-tokenizer: OXC panicked on {format} source, using fallback");
187 fallback::tokenize(source, format)
188 }
189 }
190}
191
192fn parse_with_oxc(source: &str, format: &str) -> Option<Vec<Token>> {
193 let allocator = Allocator::new();
194 let source_type = source_type_for_format(format);
195
196 let parser_return = Parser::new(&allocator, source, source_type)
197 .with_config(TokensParserConfig)
198 .parse();
199
200 if !parser_return.errors.is_empty() {
201 return None;
202 }
203
204 let ignore_ranges = find_ignore_ranges(source);
205 let bytes = source.as_bytes();
206 let line_index = LineIndex::new(bytes);
208
209 let tokens = parser_return
210 .tokens
211 .into_iter()
212 .filter_map(|token| {
213 let start = (token.start() as usize).min(source.len());
214 let end = (token.end() as usize).min(source.len());
215 if start >= end {
216 return None;
217 }
218 let kind = token.kind();
219 if matches!(kind, Kind::Eof | Kind::Undetermined | Kind::Skip) {
220 return None;
221 }
222 let value = &source[start..end];
223 let token_kind = if in_ignore(start, end, &ignore_ranges) {
224 TokenKind::Ignore
225 } else {
226 map_kind(kind)
227 };
228
229 Some(Token {
230 kind: token_kind,
231 value: value.to_string(),
232 start: line_index.location(start),
233 end: line_index.location(end),
234 })
235 })
236 .collect::<Vec<Token>>();
237
238 Some(tokens)
239}
240
241#[cfg(test)]
243mod tests {
244 use super::*;
245
246 #[test]
247 fn valid_js_produces_tokens() {
248 let tokens = tokenize_js("function hello() { return 42; }", "javascript");
249 assert!(!tokens.is_empty(), "valid JS must produce tokens");
250 }
251
252 #[test]
253 fn typescript_produces_tokens() {
254 let tokens = tokenize_js("const x: number = 5;", "typescript");
255 assert!(!tokens.is_empty());
256 }
257
258 #[test]
259 fn malformed_js_does_not_panic() {
260 let result = std::panic::catch_unwind(|| tokenize_js("let x = {{{", "javascript"));
261 assert!(result.is_ok(), "malformed JS must not panic");
262 }
263
264 #[test]
265 fn empty_source_returns_empty() {
266 let tokens = tokenize_js("", "javascript");
267 drop(tokens);
268 }
269
270 #[test]
271 fn ignore_region_tokens_marked_as_ignore() {
272 let source = r#"
273const a = 1;
274// jscpd:ignore-start
275const b = 2;
276// jscpd:ignore-end
277const c = 3;
278"#;
279 let tokens = tokenize_js(source, "javascript");
280 let has_ignore = tokens
281 .iter()
282 .any(|t| t.kind == cpd_core::models::TokenKind::Ignore);
283 assert!(has_ignore, "tokens in ignore region must be marked Ignore");
284 }
285
286 #[test]
287 fn jsx_produces_tokens() {
288 let tokens = tokenize_js("const el = <div>hello</div>;", "jsx");
289 assert!(!tokens.is_empty());
290 }
291
292 #[test]
293 fn tsx_with_type_annotation() {
294 let tokens = tokenize_js("const fn = (x: React.FC): void => {};", "tsx");
295 assert!(!tokens.is_empty());
296 }
297
298 #[test]
299 fn multiline_location_uses_binary_search() {
300 let source = "const a = 1;\nconst b = 2;\nconst c = 3;";
301 let tokens = tokenize_js(source, "javascript");
302 let b_token = tokens.iter().find(|t| t.value == "b");
304 assert!(b_token.is_some(), "must find token b");
305 assert_eq!(b_token.unwrap().start.line, 2, "b must be on line 2");
306 }
307}