Skip to main content

cpd_tokenizer/
tokenizer.rs

1use std::str::FromStr;
2
3use cpd_core::hash::hash_token;
4use cpd_core::models::{DetectionToken, Token, TokenKind};
5
6use crate::markdown::tokens_to_detection;
7
8/// A sub-format detection map produced by multi-format tokenizers.
9///
10/// For single-format files, `tokenize_to_detection_maps()` returns exactly one
11/// TokenMap with the same format as the file.
12///
13/// For multi-format files (markdown, SFC), one TokenMap is returned per
14/// detected sub-language, each carrying tokens that should enter that
15/// format's detection pool.
16#[derive(Debug, Clone)]
17pub struct TokenMap {
18    pub format: String,
19    pub tokens: Vec<DetectionToken>,
20}
21
22#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
23pub enum Mode {
24    #[default]
25    Mild,
26    Weak,
27    Strict,
28}
29
30impl FromStr for Mode {
31    type Err = ();
32
33    fn from_str(s: &str) -> Result<Self, Self::Err> {
34        match s {
35            "weak" => Ok(Self::Weak),
36            "strict" => Ok(Self::Strict),
37            _ => Ok(Self::Mild),
38        }
39    }
40}
41
42/// Options for the detection-path tokenizer.
43///
44/// Carries mode, case-folding flag, pre-parsed ignore-region byte ranges,
45/// and pre-compiled code-level regex patterns that skip matching tokens during detection.
46///
47/// Code-level ignore patterns (v4 `ignorePattern`) work by matching regex patterns
48/// against source text, collecting byte ranges of matches, and then filtering
49/// any token whose byte range overlaps a match — identical in effect to v4's
50/// `setupIgnorePatterns` which injected Prism grammar tokens.
51#[derive(Debug, Clone)]
52pub struct TokenizeOptions {
53    pub mode: Mode,
54    /// When true, token values are lowercased before hashing.
55    pub ignore_case: bool,
56    /// Ignored byte ranges from `jscpd:ignore-start` / `jscpd:ignore-end`
57    /// and code-level regex matches from `ignorePattern`.
58    /// Each entry is `[start_byte, end_byte)`.
59    pub ignore_ranges: Vec<[usize; 2]>,
60    /// Pre-compiled code-level regex patterns inherited from v4 `ignorePattern`.
61    /// Before tokenization, these are matched against the source text and
62    /// overlapping byte ranges are added to `ignore_ranges`.
63    pub code_ignore_regexes: Vec<regex::Regex>,
64}
65
66impl TokenizeOptions {
67    pub fn new(mode: Mode) -> Self {
68        Self {
69            mode,
70            ignore_case: false,
71            ignore_ranges: Vec::new(),
72            code_ignore_regexes: Vec::new(),
73        }
74    }
75
76    /// Build TokenizeOptions with pre-compiled regex patterns from string patterns.
77    /// Invalid regex patterns are silently skipped.
78    pub fn with_code_ignore_patterns(mode: Mode, patterns: &[String]) -> Self {
79        let code_ignore_regexes: Vec<regex::Regex> = patterns
80            .iter()
81            .filter_map(|p| regex::Regex::new(p).ok())
82            .collect();
83        Self {
84            mode,
85            ignore_case: false,
86            ignore_ranges: Vec::new(),
87            code_ignore_regexes,
88        }
89    }
90}
91
92/// Tokenize a single-format source snippet into detection tokens.
93///
94/// Used by markdown and SFC tokenizers to dispatch embedded code blocks to the
95/// appropriate language tokenizer.
96pub fn tokenize_format_to_detection(
97    format: &str,
98    source: &str,
99    options: &TokenizeOptions,
100) -> Vec<DetectionToken> {
101    let raw = match format {
102        "javascript" | "typescript" | "jsx" | "tsx" => {
103            crate::javascript::tokenize_js(source, format)
104        }
105        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc(source, format, options.mode),
106        "markdown" | "md" => crate::generic::tokenize_generic(source, format),
107        _ => crate::generic::tokenize_generic(source, format),
108    };
109    tokens_to_detection(raw, options)
110}
111
112/// Compute byte ranges of all regex matches against source text.
113/// Used to populate `ignore_ranges` from `ignorePattern` regexes before
114/// tokenization, matching v4 semantics where regex patterns match against
115/// source text regions (not individual token values).
116pub fn code_ignore_ranges(source: &str, regexes: &[regex::Regex]) -> Vec<[usize; 2]> {
117    let mut ranges = Vec::new();
118    for re in regexes {
119        for m in re.find_iter(source) {
120            ranges.push([m.start(), m.end()]);
121        }
122    }
123    ranges
124}
125
126/// Push a token into the detection output if it passes all filters.
127///
128/// Filtering happens here — at tokenize time — so the resulting
129/// `Vec<DetectionToken>` passed to detection is already minimal.
130/// Token values are not stored; only the pre-computed hash is kept.
131///
132/// The argument count is intentional: this function is a hot-path helper
133/// called from every tokenizer branch; grouping parameters into a struct
134/// would add an extra dereference per call.
135#[allow(clippy::too_many_arguments)]
136#[inline]
137pub fn push_token(
138    tokens: &mut Vec<DetectionToken>,
139    kind: TokenKind,
140    value: &str,
141    byte_start: usize,
142    byte_end: usize,
143    start: cpd_core::models::Location,
144    end: cpd_core::models::Location,
145    options: &TokenizeOptions,
146) {
147    // Drop Ignore-marked tokens in all modes.
148    if kind == TokenKind::Ignore {
149        return;
150    }
151    // Drop tokens in Ignore byte ranges.
152    // This covers both jscpd:ignore-start/end markers and code-level ignorePattern
153    // regex ranges (which are computed from source text before tokenization).
154    if options
155        .ignore_ranges
156        .iter()
157        .any(|[rs, re]| byte_start < *re && byte_end > *rs)
158    {
159        return;
160    }
161    // Mode-based filtering:
162    match options.mode {
163        Mode::Mild => {
164            if kind == TokenKind::Whitespace {
165                return;
166            }
167        }
168        Mode::Weak => {
169            if matches!(
170                kind,
171                TokenKind::Whitespace | TokenKind::Comment | TokenKind::BlockComment
172            ) {
173                return;
174            }
175        }
176        Mode::Strict => {} // keep everything (except Ignore, handled above)
177    }
178    tokens.push(DetectionToken {
179        hash: hash_token(kind.discriminant(), value, options.ignore_case),
180        start,
181        end,
182        range: [byte_start, byte_end],
183    });
184}
185
186/// Tokenize source code in the given format with the given mode.
187/// Returns a Vec<Token>. Never panics on empty input — returns empty Vec.
188///
189/// This is the display/reporter path. For the detection path, use
190/// `tokenize_to_detection`.
191pub fn tokenize(format: &str, source: &str, mode: Mode) -> Vec<Token> {
192    let raw = dispatch_tokenizer(format, source, mode);
193    // Apply mode filter inline — keeps Ignore tokens removed, drops Whitespace in
194    // Mild, drops Whitespace+Comment+BlockComment in Weak, keeps all in Strict.
195    raw.into_iter().filter(|t| keep_token(t, mode)).collect()
196}
197
198fn keep_token(token: &Token, mode: Mode) -> bool {
199    if token.kind == TokenKind::Ignore {
200        return false;
201    }
202    match mode {
203        Mode::Mild => !matches!(token.kind, TokenKind::Whitespace),
204        Mode::Weak => !matches!(
205            token.kind,
206            TokenKind::Whitespace | TokenKind::Comment | TokenKind::BlockComment
207        ),
208        Mode::Strict => true,
209    }
210}
211
212/// Tokenize source code for the detection hot path.
213///
214/// Returns `Vec<DetectionToken>` — tokens filtered and hashed inline at
215/// tokenize time. No per-token heap allocation survives in the output:
216/// the value string is consumed; only the hash, locations, and byte range
217/// are stored.
218///
219/// This replaces the `tokenize` → `apply_mode` → convert-to-hashes pipeline
220/// that existed in `detect.rs`.
221pub fn tokenize_to_detection(
222    format: &str,
223    source: &str,
224    options: &TokenizeOptions,
225) -> Vec<DetectionToken> {
226    // Produce the display tokens first (reuse existing tokenizer code),
227    // then convert to DetectionToken in one pass applying options filters.
228    //
229    // This approach is conservative: it reuses all existing tokenizer logic
230    // without risk of introducing per-tokenizer bugs. The conversion is O(n)
231    // and eliminates the separate filter pass and hash computation that
232    // previously happened inside detect.rs.
233    let raw = dispatch_tokenizer(format, source, options.mode);
234    let mut detection = Vec::with_capacity(raw.len());
235    for t in raw {
236        let byte_start = t.start.offset as usize;
237        let byte_end = t.end.offset as usize;
238        push_token(
239            &mut detection,
240            t.kind,
241            &t.value,
242            byte_start,
243            byte_end,
244            t.start,
245            t.end,
246            options,
247        );
248    }
249    detection
250}
251
252fn dispatch_tokenizer(format: &str, source: &str, mode: Mode) -> Vec<Token> {
253    match format {
254        "javascript" | "typescript" | "jsx" | "tsx" => {
255            crate::javascript::tokenize_js(source, format)
256        }
257        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc(source, format, mode),
258        "markdown" | "md" => crate::markdown::tokenize_markdown(source, mode),
259        _ => crate::generic::tokenize_generic(source, format),
260    }
261}
262
263/// Tokenize source code into one or more format-specific detection maps.
264///
265/// For single-format files, returns exactly one `TokenMap` with the same format.
266/// For multi-format files (markdown, SFCs), returns one `TokenMap` per detected
267/// sub-language — e.g. markdown prose + embedded JavaScript + embedded Python.
268///
269/// Each map's tokens carry byte offsets relative to the original source, so
270/// they can be used directly for clone detection within their format group.
271pub fn tokenize_to_detection_maps(
272    format: &str,
273    source: &str,
274    options: &TokenizeOptions,
275) -> Vec<TokenMap> {
276    match format {
277        "markdown" | "md" => crate::markdown::tokenize_markdown_maps(source, options),
278        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc_maps(source, format, options),
279        _ => {
280            let tokens = tokenize_to_detection(format, source, options);
281            vec![TokenMap {
282                format: format.to_string(),
283                tokens,
284            }]
285        }
286    }
287}
288
289#[cfg(test)]
290mod tests {
291    use super::*;
292
293    #[test]
294    fn mode_from_str_defaults_to_mild() {
295        assert_eq!("unknown".parse::<Mode>().unwrap(), Mode::Mild);
296        assert_eq!("mild".parse::<Mode>().unwrap(), Mode::Mild);
297    }
298
299    #[test]
300    fn mode_from_str_weak() {
301        assert_eq!("weak".parse::<Mode>().unwrap(), Mode::Weak);
302    }
303
304    #[test]
305    fn mode_from_str_strict() {
306        assert_eq!("strict".parse::<Mode>().unwrap(), Mode::Strict);
307    }
308
309    #[test]
310    fn tokenize_to_detection_returns_detection_tokens() {
311        let opts = TokenizeOptions::new(Mode::Mild);
312        let tokens = tokenize_to_detection("javascript", "function hello() { return 42; }", &opts);
313        assert!(
314            !tokens.is_empty(),
315            "must produce DetectionTokens for valid JS"
316        );
317    }
318
319    #[test]
320    fn tokenize_to_detection_mild_excludes_whitespace() {
321        let opts = TokenizeOptions::new(Mode::Mild);
322        // The raw tokenizer produces whitespace tokens; mild mode drops them.
323        // We verify by counting: detection output should have fewer tokens than
324        // a strict-mode tokenize which keeps whitespace.
325        let mild = tokenize_to_detection("javascript", "a b c", &opts);
326        let strict =
327            tokenize_to_detection("javascript", "a b c", &TokenizeOptions::new(Mode::Strict));
328        // Mild must not exceed strict count (whitespace removed).
329        // Note: JS tokenizer doesn't produce Whitespace kind for OXC tokens,
330        // but the contract is that push_token correctly drops them if present.
331        let _ = (mild, strict);
332    }
333
334    #[test]
335    fn push_token_drops_ignore_kind() {
336        let mut tokens = Vec::new();
337        let loc = cpd_core::models::Location {
338            line: 1,
339            column: 0,
340            offset: 0,
341        };
342        let opts = TokenizeOptions::new(Mode::Mild);
343        push_token(
344            &mut tokens,
345            TokenKind::Ignore,
346            "secret",
347            0,
348            6,
349            loc.clone(),
350            loc,
351            &opts,
352        );
353        assert!(tokens.is_empty(), "Ignore-kind tokens must be dropped");
354    }
355
356    #[test]
357    fn push_token_drops_whitespace_in_mild_mode() {
358        let mut tokens = Vec::new();
359        let loc = cpd_core::models::Location {
360            line: 1,
361            column: 0,
362            offset: 0,
363        };
364        let opts = TokenizeOptions::new(Mode::Mild);
365        push_token(
366            &mut tokens,
367            TokenKind::Whitespace,
368            " ",
369            0,
370            1,
371            loc.clone(),
372            loc,
373            &opts,
374        );
375        assert!(tokens.is_empty(), "Whitespace must be dropped in Mild mode");
376    }
377
378    #[test]
379    fn push_token_keeps_whitespace_in_strict_mode() {
380        let mut tokens = Vec::new();
381        let loc = cpd_core::models::Location {
382            line: 1,
383            column: 0,
384            offset: 0,
385        };
386        let opts = TokenizeOptions::new(Mode::Strict);
387        push_token(
388            &mut tokens,
389            TokenKind::Whitespace,
390            " ",
391            0,
392            1,
393            loc.clone(),
394            loc,
395            &opts,
396        );
397        assert_eq!(tokens.len(), 1, "Whitespace must be kept in Strict mode");
398    }
399
400    #[test]
401    fn push_token_drops_comment_in_weak_mode() {
402        let mut tokens = Vec::new();
403        let loc = cpd_core::models::Location {
404            line: 1,
405            column: 0,
406            offset: 0,
407        };
408        let opts = TokenizeOptions::new(Mode::Weak);
409        push_token(
410            &mut tokens,
411            TokenKind::Comment,
412            "// note",
413            0,
414            7,
415            loc.clone(),
416            loc,
417            &opts,
418        );
419        assert!(tokens.is_empty(), "Comment must be dropped in Weak mode");
420    }
421
422    #[test]
423    fn push_token_ignore_case_folds_hash() {
424        let mut t1 = Vec::new();
425        let mut t2 = Vec::new();
426        let loc = cpd_core::models::Location {
427            line: 1,
428            column: 0,
429            offset: 0,
430        };
431        let mut opts = TokenizeOptions::new(Mode::Mild);
432        opts.ignore_case = true;
433        push_token(
434            &mut t1,
435            TokenKind::Identifier,
436            "Hello",
437            0,
438            5,
439            loc.clone(),
440            loc.clone(),
441            &opts,
442        );
443        push_token(
444            &mut t2,
445            TokenKind::Identifier,
446            "hello",
447            0,
448            5,
449            loc.clone(),
450            loc,
451            &opts,
452        );
453        assert_eq!(t1[0].hash, t2[0].hash, "ignore_case must fold case in hash");
454    }
455
456    #[test]
457    fn push_token_code_ignore_range_skips_overlapping_token() {
458        // Simulate: source = "foo// cpd-disable"
459        // regex "//\\s*cpd-disable" matches bytes 3..18
460        // Token "foo" is at 0..3 (no overlap -> kept)
461        // Token "// cpd-disable" is at 3..18 (overlaps -> skipped)
462        let mut tokens = Vec::new();
463        let loc = cpd_core::models::Location {
464            line: 1,
465            column: 0,
466            offset: 0,
467        };
468        let mut opts = TokenizeOptions::new(Mode::Mild);
469        // Pre-computed byte ranges from regex match on source text
470        opts.ignore_ranges = vec![[3, 18]];
471        push_token(
472            &mut tokens,
473            TokenKind::Identifier,
474            "foo",
475            0,
476            3,
477            loc.clone(),
478            loc.clone(),
479            &opts,
480        );
481        push_token(
482            &mut tokens,
483            TokenKind::Comment,
484            "// cpd-disable",
485            3,
486            18,
487            loc.clone(),
488            loc,
489            &opts,
490        );
491        assert_eq!(tokens.len(), 1, "only the non-matching token should remain");
492        assert_eq!(tokens[0].range, [0, 3]);
493    }
494
495    #[test]
496    fn push_token_code_ignore_range_no_overlap_keeps_all() {
497        // regex match at bytes 100..120 doesn't overlap tokens at 0..3, 3..6
498        let mut tokens = Vec::new();
499        let loc = cpd_core::models::Location {
500            line: 1,
501            column: 0,
502            offset: 0,
503        };
504        let mut opts = TokenizeOptions::new(Mode::Mild);
505        opts.ignore_ranges = vec![[100, 120]];
506        push_token(
507            &mut tokens,
508            TokenKind::Identifier,
509            "foo",
510            0,
511            3,
512            loc.clone(),
513            loc.clone(),
514            &opts,
515        );
516        push_token(
517            &mut tokens,
518            TokenKind::Identifier,
519            "bar",
520            3,
521            6,
522            loc.clone(),
523            loc,
524            &opts,
525        );
526        assert_eq!(
527            tokens.len(),
528            2,
529            "both tokens should remain when range doesn't overlap"
530        );
531    }
532
533    #[test]
534    fn code_ignore_ranges_computes_from_source_text() {
535        let source = "import foo from 'bar';\nconst x = 1;";
536        let re = regex::Regex::new(r"import\s+\w+\s+from").unwrap();
537        let ranges = code_ignore_ranges(source, &[re]);
538        assert_eq!(ranges.len(), 1, "should find one regex match");
539        // "import foo from" starts at byte 0, ends at byte 15
540        assert_eq!(ranges[0], [0, 15]);
541    }
542
543    #[test]
544    fn code_ignore_ranges_multiple_patterns() {
545        let source = "// MIT License\nfunction foo() {}\n// Copyright";
546        let re1 = regex::Regex::new(r"//\s*MIT\s+License").unwrap();
547        let re2 = regex::Regex::new(r"//\s*Copyright").unwrap();
548        let ranges = code_ignore_ranges(source, &[re1, re2]);
549        assert_eq!(ranges.len(), 2, "should find two regex matches");
550    }
551
552    #[test]
553    fn code_ignore_ranges_empty_regexes() {
554        let source = "function foo() {}";
555        let ranges = code_ignore_ranges(source, &[]);
556        assert!(ranges.is_empty(), "no regexes means no ranges");
557    }
558
559    #[test]
560    fn with_code_ignore_patterns_builds_regexes() {
561        let opts = TokenizeOptions::with_code_ignore_patterns(
562            Mode::Mild,
563            &["function".to_string(), r"//\s*cpd-disable".to_string()],
564        );
565        assert_eq!(opts.code_ignore_regexes.len(), 2);
566        assert!(opts.code_ignore_regexes[0].is_match("function"));
567        assert!(opts.code_ignore_regexes[1].is_match("// cpd-disable"));
568        assert!(!opts.code_ignore_regexes[1].is_match("function"));
569    }
570
571    #[test]
572    fn tokenize_to_detection_with_code_ignore_ranges_skips_imports() {
573        let source = "import * from 'lodash';\nconst x = 1;";
574        let regexes = vec![regex::Regex::new(r"import\s+\*\s+from").unwrap()];
575        let ranges = code_ignore_ranges(source, &regexes);
576        assert!(!ranges.is_empty(), "should find regex match in source");
577
578        let mut opts = TokenizeOptions::new(Mode::Mild);
579        opts.ignore_ranges = ranges;
580        let tokens = tokenize_to_detection("javascript", source, &opts);
581
582        // Tokens whose byte ranges overlap the import match should be skipped.
583        // "import" (0-6), "*" (7-8), "from" (9-13) should all be in range,
584        // but "const" (24-29) and "x" (30-31) etc should remain.
585        let has_const = tokens.iter().any(|t| {
586            // Check that tokens after the import line are still present
587            t.range[0] >= 24
588        });
589        assert!(
590            has_const,
591            "tokens after the import line should still be present"
592        );
593    }
594
595    #[test]
596    fn code_ignore_ranges_multi_token_match() {
597        // The key test: regex "import.*from" matches multi-token source text
598        // like "import * from 'module-name'" — not just a single token value.
599        let source = "import * from 'lodash';\nconst result = 42;";
600        let re = regex::Regex::new(r"import\s+.*?\s+from").unwrap();
601        let ranges = code_ignore_ranges(source, &[re]);
602        assert_eq!(
603            ranges.len(),
604            1,
605            "should find one regex match spanning import statement"
606        );
607        assert!(ranges[0][0] == 0, "match should start at beginning");
608        assert!(ranges[0][1] > 0, "match should have non-zero end");
609    }
610}