Skip to main content

cpd_tokenizer/
tokenizer.rs

1use std::str::FromStr;
2
3use cpd_core::hash::hash_token;
4use cpd_core::models::{DetectionToken, Token, TokenKind};
5
6use crate::markdown::tokens_to_detection;
7
8/// A sub-format detection map produced by multi-format tokenizers.
9///
10/// For single-format files, `tokenize_to_detection_maps()` returns exactly one
11/// TokenMap with the same format as the file.
12///
13/// For multi-format files (markdown, SFC), one TokenMap is returned per
14/// detected sub-language, each carrying tokens that should enter that
15/// format's detection pool.
16#[derive(Debug, Clone)]
17pub struct TokenMap {
18    pub format: String,
19    pub tokens: Vec<DetectionToken>,
20}
21
22#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
23pub enum Mode {
24    #[default]
25    Mild,
26    Weak,
27    Strict,
28}
29
30impl FromStr for Mode {
31    type Err = ();
32
33    fn from_str(s: &str) -> Result<Self, Self::Err> {
34        match s {
35            "weak" => Ok(Self::Weak),
36            "strict" => Ok(Self::Strict),
37            _ => Ok(Self::Mild),
38        }
39    }
40}
41
42/// Options for the detection-path tokenizer.
43///
44/// Carries mode, case-folding flag, pre-parsed ignore-region byte ranges,
45/// and pre-compiled code-level regex patterns that skip matching tokens during detection.
46///
47/// Code-level ignore patterns (v4 `ignorePattern`) work by matching regex patterns
48/// against source text, collecting byte ranges of matches, and then filtering
49/// any token whose byte range overlaps a match — identical in effect to v4's
50/// `setupIgnorePatterns` which injected Prism grammar tokens.
51#[derive(Debug, Clone)]
52pub struct TokenizeOptions {
53    pub mode: Mode,
54    /// When true, token values are lowercased before hashing.
55    pub ignore_case: bool,
56    /// Ignored byte ranges from `jscpd:ignore-start` / `jscpd:ignore-end`
57    /// and code-level regex matches from `ignorePattern`.
58    /// Each entry is `[start_byte, end_byte)`.
59    pub ignore_ranges: Vec<[usize; 2]>,
60    /// Pre-compiled code-level regex patterns inherited from v4 `ignorePattern`.
61    /// Before tokenization, these are matched against the source text and
62    /// overlapping byte ranges are added to `ignore_ranges`.
63    pub code_ignore_regexes: Vec<regex::Regex>,
64    /// Formats whose TypeScript-only syntax is stripped from the detection
65    /// token stream (`--cross-formats` groups mixing TS with JS). Only
66    /// `typescript` and `tsx` are meaningful here; empty by default so the
67    /// standard detection path is untouched.
68    pub strip_types_formats: std::collections::HashSet<String>,
69}
70
71impl TokenizeOptions {
72    pub fn new(mode: Mode) -> Self {
73        Self {
74            mode,
75            ignore_case: false,
76            ignore_ranges: Vec::new(),
77            code_ignore_regexes: Vec::new(),
78            strip_types_formats: std::collections::HashSet::new(),
79        }
80    }
81
82    /// Build TokenizeOptions with pre-compiled regex patterns from string patterns.
83    /// Invalid regex patterns are silently skipped.
84    pub fn with_code_ignore_patterns(mode: Mode, patterns: &[String]) -> Self {
85        let code_ignore_regexes: Vec<regex::Regex> = patterns
86            .iter()
87            .filter_map(|p| regex::Regex::new(p).ok())
88            .collect();
89        Self {
90            mode,
91            ignore_case: false,
92            ignore_ranges: Vec::new(),
93            code_ignore_regexes,
94            strip_types_formats: std::collections::HashSet::new(),
95        }
96    }
97}
98
99/// Tokenize a single-format source snippet into detection tokens.
100///
101/// Used by markdown and SFC tokenizers to dispatch embedded code blocks to the
102/// appropriate language tokenizer.
103pub fn tokenize_format_to_detection(
104    format: &str,
105    source: &str,
106    options: &TokenizeOptions,
107) -> Vec<DetectionToken> {
108    let raw = match format {
109        "javascript" | "typescript" | "jsx" | "tsx" => {
110            if should_strip_types(format, options) {
111                crate::javascript::tokenize_js_stripped(source, format)
112            } else {
113                crate::javascript::tokenize_js(source, format)
114            }
115        }
116        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc(source, format, options.mode),
117        "markdown" | "md" => crate::generic::tokenize_generic(source, format),
118        _ => crate::generic::tokenize_generic(source, format),
119    };
120    tokens_to_detection(raw, options)
121}
122
123/// True when this format's TypeScript-only syntax must be stripped for
124/// cross-format detection (see `TokenizeOptions::strip_types_formats`).
125fn should_strip_types(format: &str, options: &TokenizeOptions) -> bool {
126    matches!(format, "typescript" | "tsx") && options.strip_types_formats.contains(format)
127}
128
129/// Compute byte ranges of all regex matches against source text.
130/// Used to populate `ignore_ranges` from `ignorePattern` regexes before
131/// tokenization, matching v4 semantics where regex patterns match against
132/// source text regions (not individual token values).
133pub fn code_ignore_ranges(source: &str, regexes: &[regex::Regex]) -> Vec<[usize; 2]> {
134    let mut ranges = Vec::new();
135    for re in regexes {
136        for m in re.find_iter(source) {
137            ranges.push([m.start(), m.end()]);
138        }
139    }
140    ranges
141}
142
143/// Push a token into the detection output if it passes all filters.
144///
145/// Filtering happens here — at tokenize time — so the resulting
146/// `Vec<DetectionToken>` passed to detection is already minimal.
147/// Token values are not stored; only the pre-computed hash is kept.
148///
149/// The argument count is intentional: this function is a hot-path helper
150/// called from every tokenizer branch; grouping parameters into a struct
151/// would add an extra dereference per call.
152#[allow(clippy::too_many_arguments)]
153#[inline]
154pub fn push_token(
155    tokens: &mut Vec<DetectionToken>,
156    kind: TokenKind,
157    value: &str,
158    byte_start: usize,
159    byte_end: usize,
160    start: cpd_core::models::Location,
161    end: cpd_core::models::Location,
162    options: &TokenizeOptions,
163) {
164    // Drop Ignore-marked tokens in all modes.
165    if kind == TokenKind::Ignore {
166        return;
167    }
168    // Drop tokens in Ignore byte ranges.
169    // This covers both jscpd:ignore-start/end markers and code-level ignorePattern
170    // regex ranges (which are computed from source text before tokenization).
171    if options
172        .ignore_ranges
173        .iter()
174        .any(|[rs, re]| byte_start < *re && byte_end > *rs)
175    {
176        return;
177    }
178    // Mode-based filtering:
179    match options.mode {
180        Mode::Mild => {
181            if kind == TokenKind::Whitespace {
182                return;
183            }
184        }
185        Mode::Weak => {
186            if matches!(
187                kind,
188                TokenKind::Whitespace | TokenKind::Comment | TokenKind::BlockComment
189            ) {
190                return;
191            }
192        }
193        Mode::Strict => {} // keep everything (except Ignore, handled above)
194    }
195    tokens.push(DetectionToken {
196        hash: hash_token(kind.discriminant(), value, options.ignore_case),
197        start,
198        end,
199        range: [byte_start, byte_end],
200    });
201}
202
203/// Tokenize source code in the given format with the given mode.
204/// Returns a Vec<Token>. Never panics on empty input — returns empty Vec.
205///
206/// This is the display/reporter path. For the detection path, use
207/// `tokenize_to_detection`.
208pub fn tokenize(format: &str, source: &str, mode: Mode) -> Vec<Token> {
209    let raw = dispatch_tokenizer(format, source, mode);
210    // Apply mode filter inline — keeps Ignore tokens removed, drops Whitespace in
211    // Mild, drops Whitespace+Comment+BlockComment in Weak, keeps all in Strict.
212    raw.into_iter().filter(|t| keep_token(t, mode)).collect()
213}
214
215fn keep_token(token: &Token, mode: Mode) -> bool {
216    if token.kind == TokenKind::Ignore {
217        return false;
218    }
219    match mode {
220        Mode::Mild => !matches!(token.kind, TokenKind::Whitespace),
221        Mode::Weak => !matches!(
222            token.kind,
223            TokenKind::Whitespace | TokenKind::Comment | TokenKind::BlockComment
224        ),
225        Mode::Strict => true,
226    }
227}
228
229/// Tokenize source code for the detection hot path.
230///
231/// Returns `Vec<DetectionToken>` — tokens filtered and hashed inline at
232/// tokenize time. No per-token heap allocation survives in the output:
233/// the value string is consumed; only the hash, locations, and byte range
234/// are stored.
235///
236/// This replaces the `tokenize` → `apply_mode` → convert-to-hashes pipeline
237/// that existed in `detect.rs`.
238pub fn tokenize_to_detection(
239    format: &str,
240    source: &str,
241    options: &TokenizeOptions,
242) -> Vec<DetectionToken> {
243    // Produce the display tokens first (reuse existing tokenizer code),
244    // then convert to DetectionToken in one pass applying options filters.
245    //
246    // This approach is conservative: it reuses all existing tokenizer logic
247    // without risk of introducing per-tokenizer bugs. The conversion is O(n)
248    // and eliminates the separate filter pass and hash computation that
249    // previously happened inside detect.rs.
250    let raw = if should_strip_types(format, options) {
251        crate::javascript::tokenize_js_stripped(source, format)
252    } else {
253        dispatch_tokenizer(format, source, options.mode)
254    };
255    let mut detection = Vec::with_capacity(raw.len());
256    for t in raw {
257        let byte_start = t.start.offset as usize;
258        let byte_end = t.end.offset as usize;
259        push_token(
260            &mut detection,
261            t.kind,
262            &t.value,
263            byte_start,
264            byte_end,
265            t.start,
266            t.end,
267            options,
268        );
269    }
270    detection
271}
272
273fn dispatch_tokenizer(format: &str, source: &str, mode: Mode) -> Vec<Token> {
274    match format {
275        "javascript" | "typescript" | "jsx" | "tsx" => {
276            crate::javascript::tokenize_js(source, format)
277        }
278        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc(source, format, mode),
279        "razor" => crate::razor::tokenize_razor(source, mode),
280        "markdown" | "md" => crate::markdown::tokenize_markdown(source, mode),
281        _ => crate::generic::tokenize_generic(source, format),
282    }
283}
284
285/// Tokenize source code into one or more format-specific detection maps.
286///
287/// For single-format files, returns exactly one `TokenMap` with the same format.
288/// For multi-format files (markdown, SFCs), returns one `TokenMap` per detected
289/// sub-language — e.g. markdown prose + embedded JavaScript + embedded Python.
290///
291/// Each map's tokens carry byte offsets relative to the original source, so
292/// they can be used directly for clone detection within their format group.
293pub fn tokenize_to_detection_maps(
294    format: &str,
295    source: &str,
296    options: &TokenizeOptions,
297) -> Vec<TokenMap> {
298    match format {
299        "markdown" | "md" => crate::markdown::tokenize_markdown_maps(source, options),
300        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc_maps(source, format, options),
301        "razor" => crate::razor::tokenize_razor_maps(source, options),
302        _ => {
303            let tokens = tokenize_to_detection(format, source, options);
304            vec![TokenMap {
305                format: format.to_string(),
306                tokens,
307            }]
308        }
309    }
310}
311
312#[cfg(test)]
313mod tests {
314    use super::*;
315
316    #[test]
317    fn mode_from_str_defaults_to_mild() {
318        assert_eq!("unknown".parse::<Mode>().unwrap(), Mode::Mild);
319        assert_eq!("mild".parse::<Mode>().unwrap(), Mode::Mild);
320    }
321
322    #[test]
323    fn mode_from_str_weak() {
324        assert_eq!("weak".parse::<Mode>().unwrap(), Mode::Weak);
325    }
326
327    #[test]
328    fn mode_from_str_strict() {
329        assert_eq!("strict".parse::<Mode>().unwrap(), Mode::Strict);
330    }
331
332    #[test]
333    fn tokenize_to_detection_returns_detection_tokens() {
334        let opts = TokenizeOptions::new(Mode::Mild);
335        let tokens = tokenize_to_detection("javascript", "function hello() { return 42; }", &opts);
336        assert!(
337            !tokens.is_empty(),
338            "must produce DetectionTokens for valid JS"
339        );
340    }
341
342    #[test]
343    fn tokenize_to_detection_mild_excludes_whitespace() {
344        let opts = TokenizeOptions::new(Mode::Mild);
345        // The raw tokenizer produces whitespace tokens; mild mode drops them.
346        // We verify by counting: detection output should have fewer tokens than
347        // a strict-mode tokenize which keeps whitespace.
348        let mild = tokenize_to_detection("javascript", "a b c", &opts);
349        let strict =
350            tokenize_to_detection("javascript", "a b c", &TokenizeOptions::new(Mode::Strict));
351        // Mild must not exceed strict count (whitespace removed).
352        // Note: JS tokenizer doesn't produce Whitespace kind for OXC tokens,
353        // but the contract is that push_token correctly drops them if present.
354        let _ = (mild, strict);
355    }
356
357    #[test]
358    fn push_token_drops_ignore_kind() {
359        let mut tokens = Vec::new();
360        let loc = cpd_core::models::Location {
361            line: 1,
362            column: 0,
363            offset: 0,
364        };
365        let opts = TokenizeOptions::new(Mode::Mild);
366        push_token(
367            &mut tokens,
368            TokenKind::Ignore,
369            "secret",
370            0,
371            6,
372            loc.clone(),
373            loc,
374            &opts,
375        );
376        assert!(tokens.is_empty(), "Ignore-kind tokens must be dropped");
377    }
378
379    #[test]
380    fn push_token_drops_whitespace_in_mild_mode() {
381        let mut tokens = Vec::new();
382        let loc = cpd_core::models::Location {
383            line: 1,
384            column: 0,
385            offset: 0,
386        };
387        let opts = TokenizeOptions::new(Mode::Mild);
388        push_token(
389            &mut tokens,
390            TokenKind::Whitespace,
391            " ",
392            0,
393            1,
394            loc.clone(),
395            loc,
396            &opts,
397        );
398        assert!(tokens.is_empty(), "Whitespace must be dropped in Mild mode");
399    }
400
401    #[test]
402    fn push_token_keeps_whitespace_in_strict_mode() {
403        let mut tokens = Vec::new();
404        let loc = cpd_core::models::Location {
405            line: 1,
406            column: 0,
407            offset: 0,
408        };
409        let opts = TokenizeOptions::new(Mode::Strict);
410        push_token(
411            &mut tokens,
412            TokenKind::Whitespace,
413            " ",
414            0,
415            1,
416            loc.clone(),
417            loc,
418            &opts,
419        );
420        assert_eq!(tokens.len(), 1, "Whitespace must be kept in Strict mode");
421    }
422
423    #[test]
424    fn push_token_drops_comment_in_weak_mode() {
425        let mut tokens = Vec::new();
426        let loc = cpd_core::models::Location {
427            line: 1,
428            column: 0,
429            offset: 0,
430        };
431        let opts = TokenizeOptions::new(Mode::Weak);
432        push_token(
433            &mut tokens,
434            TokenKind::Comment,
435            "// note",
436            0,
437            7,
438            loc.clone(),
439            loc,
440            &opts,
441        );
442        assert!(tokens.is_empty(), "Comment must be dropped in Weak mode");
443    }
444
445    #[test]
446    fn push_token_ignore_case_folds_hash() {
447        let mut t1 = Vec::new();
448        let mut t2 = Vec::new();
449        let loc = cpd_core::models::Location {
450            line: 1,
451            column: 0,
452            offset: 0,
453        };
454        let mut opts = TokenizeOptions::new(Mode::Mild);
455        opts.ignore_case = true;
456        push_token(
457            &mut t1,
458            TokenKind::Identifier,
459            "Hello",
460            0,
461            5,
462            loc.clone(),
463            loc.clone(),
464            &opts,
465        );
466        push_token(
467            &mut t2,
468            TokenKind::Identifier,
469            "hello",
470            0,
471            5,
472            loc.clone(),
473            loc,
474            &opts,
475        );
476        assert_eq!(t1[0].hash, t2[0].hash, "ignore_case must fold case in hash");
477    }
478
479    #[test]
480    fn push_token_code_ignore_range_skips_overlapping_token() {
481        // Simulate: source = "foo// cpd-disable"
482        // regex "//\\s*cpd-disable" matches bytes 3..18
483        // Token "foo" is at 0..3 (no overlap -> kept)
484        // Token "// cpd-disable" is at 3..18 (overlaps -> skipped)
485        let mut tokens = Vec::new();
486        let loc = cpd_core::models::Location {
487            line: 1,
488            column: 0,
489            offset: 0,
490        };
491        let mut opts = TokenizeOptions::new(Mode::Mild);
492        // Pre-computed byte ranges from regex match on source text
493        opts.ignore_ranges = vec![[3, 18]];
494        push_token(
495            &mut tokens,
496            TokenKind::Identifier,
497            "foo",
498            0,
499            3,
500            loc.clone(),
501            loc.clone(),
502            &opts,
503        );
504        push_token(
505            &mut tokens,
506            TokenKind::Comment,
507            "// cpd-disable",
508            3,
509            18,
510            loc.clone(),
511            loc,
512            &opts,
513        );
514        assert_eq!(tokens.len(), 1, "only the non-matching token should remain");
515        assert_eq!(tokens[0].range, [0, 3]);
516    }
517
518    #[test]
519    fn push_token_code_ignore_range_no_overlap_keeps_all() {
520        // regex match at bytes 100..120 doesn't overlap tokens at 0..3, 3..6
521        let mut tokens = Vec::new();
522        let loc = cpd_core::models::Location {
523            line: 1,
524            column: 0,
525            offset: 0,
526        };
527        let mut opts = TokenizeOptions::new(Mode::Mild);
528        opts.ignore_ranges = vec![[100, 120]];
529        push_token(
530            &mut tokens,
531            TokenKind::Identifier,
532            "foo",
533            0,
534            3,
535            loc.clone(),
536            loc.clone(),
537            &opts,
538        );
539        push_token(
540            &mut tokens,
541            TokenKind::Identifier,
542            "bar",
543            3,
544            6,
545            loc.clone(),
546            loc,
547            &opts,
548        );
549        assert_eq!(
550            tokens.len(),
551            2,
552            "both tokens should remain when range doesn't overlap"
553        );
554    }
555
556    #[test]
557    fn code_ignore_ranges_computes_from_source_text() {
558        let source = "import foo from 'bar';\nconst x = 1;";
559        let re = regex::Regex::new(r"import\s+\w+\s+from").unwrap();
560        let ranges = code_ignore_ranges(source, &[re]);
561        assert_eq!(ranges.len(), 1, "should find one regex match");
562        // "import foo from" starts at byte 0, ends at byte 15
563        assert_eq!(ranges[0], [0, 15]);
564    }
565
566    #[test]
567    fn code_ignore_ranges_multiple_patterns() {
568        let source = "// MIT License\nfunction foo() {}\n// Copyright";
569        let re1 = regex::Regex::new(r"//\s*MIT\s+License").unwrap();
570        let re2 = regex::Regex::new(r"//\s*Copyright").unwrap();
571        let ranges = code_ignore_ranges(source, &[re1, re2]);
572        assert_eq!(ranges.len(), 2, "should find two regex matches");
573    }
574
575    #[test]
576    fn code_ignore_ranges_empty_regexes() {
577        let source = "function foo() {}";
578        let ranges = code_ignore_ranges(source, &[]);
579        assert!(ranges.is_empty(), "no regexes means no ranges");
580    }
581
582    #[test]
583    fn with_code_ignore_patterns_builds_regexes() {
584        let opts = TokenizeOptions::with_code_ignore_patterns(
585            Mode::Mild,
586            &["function".to_string(), r"//\s*cpd-disable".to_string()],
587        );
588        assert_eq!(opts.code_ignore_regexes.len(), 2);
589        assert!(opts.code_ignore_regexes[0].is_match("function"));
590        assert!(opts.code_ignore_regexes[1].is_match("// cpd-disable"));
591        assert!(!opts.code_ignore_regexes[1].is_match("function"));
592    }
593
594    #[test]
595    fn tokenize_to_detection_with_code_ignore_ranges_skips_imports() {
596        let source = "import * from 'lodash';\nconst x = 1;";
597        let regexes = vec![regex::Regex::new(r"import\s+\*\s+from").unwrap()];
598        let ranges = code_ignore_ranges(source, &regexes);
599        assert!(!ranges.is_empty(), "should find regex match in source");
600
601        let mut opts = TokenizeOptions::new(Mode::Mild);
602        opts.ignore_ranges = ranges;
603        let tokens = tokenize_to_detection("javascript", source, &opts);
604
605        // Tokens whose byte ranges overlap the import match should be skipped.
606        // "import" (0-6), "*" (7-8), "from" (9-13) should all be in range,
607        // but "const" (24-29) and "x" (30-31) etc should remain.
608        let has_const = tokens.iter().any(|t| {
609            // Check that tokens after the import line are still present
610            t.range[0] >= 24
611        });
612        assert!(
613            has_const,
614            "tokens after the import line should still be present"
615        );
616    }
617
618    #[test]
619    fn code_ignore_ranges_multi_token_match() {
620        // The key test: regex "import.*from" matches multi-token source text
621        // like "import * from 'module-name'" — not just a single token value.
622        let source = "import * from 'lodash';\nconst result = 42;";
623        let re = regex::Regex::new(r"import\s+.*?\s+from").unwrap();
624        let ranges = code_ignore_ranges(source, &[re]);
625        assert_eq!(
626            ranges.len(),
627            1,
628            "should find one regex match spanning import statement"
629        );
630        assert!(ranges[0][0] == 0, "match should start at beginning");
631        assert!(ranges[0][1] > 0, "match should have non-zero end");
632    }
633}