Skip to main content

cpd_tokenizer/
javascript.rs

1// cpd-tokenizer: OXC-based tokenizer for JavaScript/TypeScript/JSX/TSX.
2// Dispatched for formats: javascript, typescript, jsx, tsx.
3
4use std::panic::{AssertUnwindSafe, catch_unwind};
5use std::path::Path;
6
7use oxc_allocator::Allocator;
8use oxc_parser::{Kind, Parser, config::TokensParserConfig};
9use oxc_span::SourceType;
10
11use cpd_core::models::{Token, TokenKind};
12
13use crate::line_index::LineIndex;
14
15// ── fallback tokenizer ────────────────────────────────────────────────────────
16
17mod fallback {
18    use super::{LineIndex, find_ignore_ranges, in_ignore};
19    use cpd_core::models::{Token, TokenKind};
20
21    /// Simple word-split fallback tokenizer. Never panics.
22    pub fn tokenize(source: &str, _format: &str) -> Vec<Token> {
23        let ignore_ranges = find_ignore_ranges(source);
24        let bytes = source.as_bytes();
25        let line_index = LineIndex::new(bytes);
26        let mut tokens = Vec::new();
27        let mut i = 0;
28        while i < bytes.len() {
29            let ch = match source[i..].chars().next() {
30                Some(c) => c,
31                None => break,
32            };
33            if ch.is_whitespace() {
34                i += ch.len_utf8();
35                continue;
36            }
37            if ch.is_alphanumeric() || ch == '_' || ch == '$' {
38                let start = i;
39                while i < bytes.len() {
40                    let c = source[i..].chars().next().unwrap_or('\0');
41                    if c.is_alphanumeric() || c == '_' || c == '$' {
42                        i += c.len_utf8();
43                    } else {
44                        break;
45                    }
46                }
47                let kind = if in_ignore(start, i, &ignore_ranges) {
48                    TokenKind::Ignore
49                } else {
50                    TokenKind::Other
51                };
52                tokens.push(Token {
53                    kind,
54                    value: source[start..i].to_string(),
55                    start: line_index.location(start),
56                    end: line_index.location(i),
57                });
58            } else {
59                let start = i;
60                i += ch.len_utf8();
61                let kind = if in_ignore(start, i, &ignore_ranges) {
62                    TokenKind::Ignore
63                } else {
64                    TokenKind::Other
65                };
66                tokens.push(Token {
67                    kind,
68                    value: ch.to_string(),
69                    start: line_index.location(start),
70                    end: line_index.location(i),
71                });
72            }
73        }
74        tokens
75    }
76}
77
78// ── helpers ──────────────────────────────────────────────────────────────────
79
80fn find_ignore_ranges(source: &str) -> Vec<[usize; 2]> {
81    let mut ranges = Vec::new();
82    let mut start: Option<usize> = None;
83    let bytes = source.as_bytes();
84    let mut i = 0;
85    while i < bytes.len() {
86        if i + 1 < bytes.len() && bytes[i] == b'/' {
87            let end = if bytes[i + 1] == b'/' {
88                bytes[i..]
89                    .iter()
90                    .position(|&b| b == b'\n')
91                    .map(|p| i + p)
92                    .unwrap_or(bytes.len())
93            } else if bytes[i + 1] == b'*' {
94                bytes[i..]
95                    .windows(2)
96                    .position(|w| w == b"*/")
97                    .map(|p| i + p + 2)
98                    .unwrap_or(bytes.len())
99            } else {
100                i += 1;
101                continue;
102            };
103            let comment_text = &source[i..end];
104            if comment_text.contains("jscpd:ignore-start") {
105                start = Some(end);
106            } else if comment_text.contains("jscpd:ignore-end")
107                && let Some(s) = start.take()
108            {
109                ranges.push([s, i]);
110            }
111            i = end;
112            continue;
113        }
114        i += 1;
115    }
116    ranges
117}
118
119fn in_ignore(offset: usize, end: usize, ranges: &[[usize; 2]]) -> bool {
120    ranges.iter().any(|[rs, re]| offset < *re && end > *rs)
121}
122
123const fn map_kind(kind: Kind) -> TokenKind {
124    if matches!(kind, Kind::Ident) {
125        return TokenKind::Identifier;
126    }
127    if kind.is_any_keyword() {
128        return TokenKind::Keyword;
129    }
130    if kind.is_literal() {
131        return TokenKind::Literal;
132    }
133    if kind.is_assignment_operator() {
134        return TokenKind::Operator;
135    }
136    if kind.is_binary_operator()
137        || kind.is_logical_operator()
138        || kind.is_unary_operator()
139        || kind.is_update_operator()
140    {
141        return TokenKind::Operator;
142    }
143    match kind {
144        Kind::Arrow => TokenKind::Operator,
145        Kind::Semicolon
146        | Kind::Comma
147        | Kind::Dot
148        | Kind::Dot3
149        | Kind::Colon
150        | Kind::LParen
151        | Kind::RParen
152        | Kind::LCurly
153        | Kind::RCurly
154        | Kind::LBrack
155        | Kind::RBrack
156        | Kind::At => TokenKind::Punctuation,
157        Kind::QuestionDot => TokenKind::Punctuation,
158        _ => TokenKind::Other,
159    }
160}
161
162pub(crate) fn source_type_for_format(format: &str) -> SourceType {
163    let filename = match format {
164        "typescript" => "input.ts",
165        "tsx" => "input.tsx",
166        _ => "input.jsx", // javascript + jsx both use jsx
167    };
168    SourceType::from_path(Path::new(filename)).unwrap_or_default()
169}
170
171// ── public API ───────────────────────────────────────────────────────────────
172
173/// Tokenize JS/TS/JSX/TSX source. Never panics.
174pub fn tokenize_js(source: &str, format: &str) -> Vec<Token> {
175    tokenize_js_impl(source, format, false)
176}
177
178/// Tokenize TS/TSX source with erasable TypeScript-only syntax stripped from
179/// the token stream (cross-format detection). Token locations still reference
180/// the original source. Never panics.
181///
182/// Sources the parser gives up on fall back to the word-split tokenizer
183/// WITHOUT stripping — they simply won't cross-match JavaScript files.
184/// Recoverable errors keep the oxc token stream (and best-effort stripping
185/// from the partial AST).
186pub fn tokenize_js_stripped(source: &str, format: &str) -> Vec<Token> {
187    tokenize_js_impl(source, format, true)
188}
189
190fn tokenize_js_impl(source: &str, format: &str, strip_types: bool) -> Vec<Token> {
191    if source.is_empty() {
192        return Vec::new();
193    }
194
195    match catch_unwind(AssertUnwindSafe(|| {
196        parse_with_oxc(source, format, strip_types)
197    })) {
198        Ok(Some(tokens)) => tokens,
199        Ok(None) => {
200            log::debug!("cpd-tokenizer: OXC parse errors in {format} source, using fallback");
201            fallback::tokenize(source, format)
202        }
203        Err(_) => {
204            log::debug!("cpd-tokenizer: OXC panicked on {format} source, using fallback");
205            fallback::tokenize(source, format)
206        }
207    }
208}
209
210fn parse_with_oxc(source: &str, format: &str, strip_types: bool) -> Option<Vec<Token>> {
211    let allocator = Allocator::new();
212    let source_type = source_type_for_format(format);
213
214    let parser_return = Parser::new(&allocator, source, source_type)
215        .with_config(TokensParserConfig)
216        .parse();
217
218    // Parse diagnostics (redeclarations, recoverable syntax errors) do not
219    // touch the token stream: tokens come from the lexer. Only a parser that
220    // gave up, or one that produced no tokens, sends the file to the
221    // word-split fallback. Anything else would tokenize a file with one
222    // error differently from every other file, so it could never match
223    // them (issue #1023).
224    if parser_return.panicked || parser_return.tokens.is_empty() {
225        return None;
226    }
227
228    let erasable = if strip_types {
229        crate::ts_strip::collect_erasable_spans(&parser_return.program, source)
230    } else {
231        crate::ts_strip::ErasableSpans::default()
232    };
233
234    let ignore_ranges = find_ignore_ranges(source);
235    let bytes = source.as_bytes();
236    // Build LineIndex once — O(n) — then all location calls are O(log n).
237    let line_index = LineIndex::new(bytes);
238
239    let tokens = parser_return
240        .tokens
241        .into_iter()
242        .filter_map(|token| {
243            let start = (token.start() as usize).min(source.len());
244            let end = (token.end() as usize).min(source.len());
245            if start >= end {
246                return None;
247            }
248            let kind = token.kind();
249            if matches!(kind, Kind::Eof | Kind::Undetermined | Kind::Skip) {
250                return None;
251            }
252            let value = &source[start..end];
253            if !erasable.is_empty()
254                && (erasable.intersects_range(start as u32, end as u32)
255                    || erasable.is_modifier_token(start as u32, end as u32, value))
256            {
257                return None;
258            }
259            let token_kind = if in_ignore(start, end, &ignore_ranges) {
260                TokenKind::Ignore
261            } else {
262                map_kind(kind)
263            };
264
265            Some(Token {
266                kind: token_kind,
267                value: value.to_string(),
268                start: line_index.location(start),
269                end: line_index.location(end),
270            })
271        })
272        .collect::<Vec<Token>>();
273
274    Some(tokens)
275}
276
277// ── tests ─────────────────────────────────────────────────────────────────────
278#[cfg(test)]
279mod tests {
280    use super::*;
281
282    /// Issue #1023: a redeclaration is a parse diagnostic, not a lexing
283    /// failure, so the file must keep the oxc token stream and stay
284    /// comparable with files that parse cleanly.
285    #[test]
286    fn parse_diagnostics_keep_the_oxc_token_stream() {
287        let clean = "export function save(row, db) {\n  db.put(row.id, row);\n  return row;\n}\n";
288        let twice = format!("{clean}\n{clean}");
289        let a = tokenize_js(clean, "javascript");
290        let b = tokenize_js(&twice, "javascript");
291        assert!(a.len() > 10);
292        assert_eq!(b.len(), a.len() * 2, "both copies lexed by oxc");
293        assert_eq!(
294            a.iter().map(|t| (&t.kind, &t.value)).collect::<Vec<_>>(),
295            b[..a.len()]
296                .iter()
297                .map(|t| (&t.kind, &t.value))
298                .collect::<Vec<_>>(),
299            "the first copy must tokenize exactly like the clean file"
300        );
301        assert_eq!(
302            b[0].kind,
303            TokenKind::Keyword,
304            "`export` classified by oxc, not word-split"
305        );
306    }
307
308    #[test]
309    fn empty_and_unlexable_sources_still_fall_back_safely() {
310        assert!(tokenize_js("", "javascript").is_empty());
311        // A stray token soup parses with errors but lexes fine: still tokens.
312        assert!(!tokenize_js("function (", "javascript").is_empty());
313    }
314
315    #[test]
316    fn valid_js_produces_tokens() {
317        let tokens = tokenize_js("function hello() { return 42; }", "javascript");
318        assert!(!tokens.is_empty(), "valid JS must produce tokens");
319    }
320
321    #[test]
322    fn typescript_produces_tokens() {
323        let tokens = tokenize_js("const x: number = 5;", "typescript");
324        assert!(!tokens.is_empty());
325    }
326
327    #[test]
328    fn malformed_js_does_not_panic() {
329        let result = std::panic::catch_unwind(|| tokenize_js("let x = {{{", "javascript"));
330        assert!(result.is_ok(), "malformed JS must not panic");
331    }
332
333    #[test]
334    fn empty_source_returns_empty() {
335        let tokens = tokenize_js("", "javascript");
336        drop(tokens);
337    }
338
339    #[test]
340    fn ignore_region_tokens_marked_as_ignore() {
341        let source = r#"
342const a = 1;
343// jscpd:ignore-start
344const b = 2;
345// jscpd:ignore-end
346const c = 3;
347"#;
348        let tokens = tokenize_js(source, "javascript");
349        let has_ignore = tokens
350            .iter()
351            .any(|t| t.kind == cpd_core::models::TokenKind::Ignore);
352        assert!(has_ignore, "tokens in ignore region must be marked Ignore");
353    }
354
355    #[test]
356    fn jsx_produces_tokens() {
357        let tokens = tokenize_js("const el = <div>hello</div>;", "jsx");
358        assert!(!tokens.is_empty());
359    }
360
361    #[test]
362    fn tsx_with_type_annotation() {
363        let tokens = tokenize_js("const fn = (x: React.FC): void => {};", "tsx");
364        assert!(!tokens.is_empty());
365    }
366
367    #[test]
368    fn multiline_location_uses_binary_search() {
369        let source = "const a = 1;\nconst b = 2;\nconst c = 3;";
370        let tokens = tokenize_js(source, "javascript");
371        // "b" is on line 2
372        let b_token = tokens.iter().find(|t| t.value == "b");
373        assert!(b_token.is_some(), "must find token b");
374        assert_eq!(b_token.unwrap().start.line, 2, "b must be on line 2");
375    }
376}