Skip to main content

cpd_tokenizer/
tokenizer.rs

1use std::str::FromStr;
2
3use cpd_core::hash::hash_token;
4use cpd_core::models::{DetectionToken, Token, TokenKind};
5
6use crate::markdown::tokens_to_detection;
7
8/// A sub-format detection map produced by multi-format tokenizers.
9///
10/// For single-format files, `tokenize_to_detection_maps()` returns exactly one
11/// TokenMap with the same format as the file.
12///
13/// For multi-format files (markdown, SFC), one TokenMap is returned per
14/// detected sub-language, each carrying tokens that should enter that
15/// format's detection pool.
16#[derive(Debug, Clone)]
17pub struct TokenMap {
18    pub format: String,
19    pub tokens: Vec<DetectionToken>,
20}
21
22#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
23pub enum Mode {
24    #[default]
25    Mild,
26    Weak,
27    Strict,
28}
29
30impl FromStr for Mode {
31    type Err = ();
32
33    fn from_str(s: &str) -> Result<Self, Self::Err> {
34        match s {
35            "weak" => Ok(Self::Weak),
36            "strict" => Ok(Self::Strict),
37            _ => Ok(Self::Mild),
38        }
39    }
40}
41
42/// Options for the detection-path tokenizer.
43///
44/// Carries mode, case-folding flag, pre-parsed ignore-region byte ranges,
45/// and pre-compiled code-level regex patterns that skip matching tokens during detection.
46///
47/// Code-level ignore patterns (v4 `ignorePattern`) work by matching regex patterns
48/// against source text, collecting byte ranges of matches, and then filtering
49/// any token whose byte range overlaps a match — identical in effect to v4's
50/// `setupIgnorePatterns` which injected Prism grammar tokens.
51#[derive(Debug, Clone)]
52pub struct TokenizeOptions {
53    pub mode: Mode,
54    /// When true, token values are lowercased before hashing.
55    pub ignore_case: bool,
56    /// Ignored byte ranges from `jscpd:ignore-start` / `jscpd:ignore-end`
57    /// and code-level regex matches from `ignorePattern`.
58    /// Each entry is `[start_byte, end_byte)`.
59    pub ignore_ranges: Vec<[usize; 2]>,
60    /// Pre-compiled code-level regex patterns inherited from v4 `ignorePattern`.
61    /// Before tokenization, these are matched against the source text and
62    /// overlapping byte ranges are added to `ignore_ranges`.
63    pub code_ignore_regexes: Vec<regex::Regex>,
64}
65
66impl TokenizeOptions {
67    pub fn new(mode: Mode) -> Self {
68        Self {
69            mode,
70            ignore_case: false,
71            ignore_ranges: Vec::new(),
72            code_ignore_regexes: Vec::new(),
73        }
74    }
75
76    /// Build TokenizeOptions with pre-compiled regex patterns from string patterns.
77    /// Invalid regex patterns are silently skipped.
78    pub fn with_code_ignore_patterns(mode: Mode, patterns: &[String]) -> Self {
79        let code_ignore_regexes: Vec<regex::Regex> = patterns
80            .iter()
81            .filter_map(|p| regex::Regex::new(p).ok())
82            .collect();
83        Self {
84            mode,
85            ignore_case: false,
86            ignore_ranges: Vec::new(),
87            code_ignore_regexes,
88        }
89    }
90}
91
92/// Tokenize a single-format source snippet into detection tokens.
93///
94/// Used by markdown and SFC tokenizers to dispatch embedded code blocks to the
95/// appropriate language tokenizer.
96pub fn tokenize_format_to_detection(
97    format: &str,
98    source: &str,
99    options: &TokenizeOptions,
100) -> Vec<DetectionToken> {
101    let raw = match format {
102        "javascript" | "typescript" | "jsx" | "tsx" => {
103            crate::javascript::tokenize_js(source, format)
104        }
105        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc(source, format, options.mode),
106        "markdown" | "md" => crate::generic::tokenize_generic(source, format),
107        _ => crate::generic::tokenize_generic(source, format),
108    };
109    tokens_to_detection(raw, options)
110}
111
112/// Compute byte ranges of all regex matches against source text.
113/// Used to populate `ignore_ranges` from `ignorePattern` regexes before
114/// tokenization, matching v4 semantics where regex patterns match against
115/// source text regions (not individual token values).
116pub fn code_ignore_ranges(source: &str, regexes: &[regex::Regex]) -> Vec<[usize; 2]> {
117    let mut ranges = Vec::new();
118    for re in regexes {
119        for m in re.find_iter(source) {
120            ranges.push([m.start(), m.end()]);
121        }
122    }
123    ranges
124}
125
126/// Push a token into the detection output if it passes all filters.
127///
128/// Filtering happens here — at tokenize time — so the resulting
129/// `Vec<DetectionToken>` passed to detection is already minimal.
130/// Token values are not stored; only the pre-computed hash is kept.
131///
132/// The argument count is intentional: this function is a hot-path helper
133/// called from every tokenizer branch; grouping parameters into a struct
134/// would add an extra dereference per call.
135#[allow(clippy::too_many_arguments)]
136#[inline]
137pub fn push_token(
138    tokens: &mut Vec<DetectionToken>,
139    kind: TokenKind,
140    value: &str,
141    byte_start: usize,
142    byte_end: usize,
143    start: cpd_core::models::Location,
144    end: cpd_core::models::Location,
145    options: &TokenizeOptions,
146) {
147    // Drop Ignore-marked tokens in all modes.
148    if kind == TokenKind::Ignore {
149        return;
150    }
151    // Drop tokens in Ignore byte ranges.
152    // This covers both jscpd:ignore-start/end markers and code-level ignorePattern
153    // regex ranges (which are computed from source text before tokenization).
154    if options
155        .ignore_ranges
156        .iter()
157        .any(|[rs, re]| byte_start < *re && byte_end > *rs)
158    {
159        return;
160    }
161    // Mode-based filtering:
162    match options.mode {
163        Mode::Mild => {
164            if kind == TokenKind::Whitespace {
165                return;
166            }
167        }
168        Mode::Weak => {
169            if matches!(
170                kind,
171                TokenKind::Whitespace | TokenKind::Comment | TokenKind::BlockComment
172            ) {
173                return;
174            }
175        }
176        Mode::Strict => {} // keep everything (except Ignore, handled above)
177    }
178    tokens.push(DetectionToken {
179        hash: hash_token(kind.discriminant(), value, options.ignore_case),
180        start,
181        end,
182        range: [byte_start, byte_end],
183    });
184}
185
186/// Tokenize source code in the given format with the given mode.
187/// Returns a Vec<Token>. Never panics on empty input — returns empty Vec.
188///
189/// This is the display/reporter path. For the detection path, use
190/// `tokenize_to_detection`.
191pub fn tokenize(format: &str, source: &str, mode: Mode) -> Vec<Token> {
192    let raw = dispatch_tokenizer(format, source, mode);
193    // Apply mode filter inline — keeps Ignore tokens removed, drops Whitespace in
194    // Mild, drops Whitespace+Comment+BlockComment in Weak, keeps all in Strict.
195    raw.into_iter().filter(|t| keep_token(t, mode)).collect()
196}
197
198fn keep_token(token: &Token, mode: Mode) -> bool {
199    if token.kind == TokenKind::Ignore {
200        return false;
201    }
202    match mode {
203        Mode::Mild => !matches!(token.kind, TokenKind::Whitespace),
204        Mode::Weak => !matches!(
205            token.kind,
206            TokenKind::Whitespace | TokenKind::Comment | TokenKind::BlockComment
207        ),
208        Mode::Strict => true,
209    }
210}
211
212/// Tokenize source code for the detection hot path.
213///
214/// Returns `Vec<DetectionToken>` — tokens filtered and hashed inline at
215/// tokenize time. No per-token heap allocation survives in the output:
216/// the value string is consumed; only the hash, locations, and byte range
217/// are stored.
218///
219/// This replaces the `tokenize` → `apply_mode` → convert-to-hashes pipeline
220/// that existed in `detect.rs`.
221pub fn tokenize_to_detection(
222    format: &str,
223    source: &str,
224    options: &TokenizeOptions,
225) -> Vec<DetectionToken> {
226    // Produce the display tokens first (reuse existing tokenizer code),
227    // then convert to DetectionToken in one pass applying options filters.
228    //
229    // This approach is conservative: it reuses all existing tokenizer logic
230    // without risk of introducing per-tokenizer bugs. The conversion is O(n)
231    // and eliminates the separate filter pass and hash computation that
232    // previously happened inside detect.rs.
233    let raw = dispatch_tokenizer(format, source, options.mode);
234    let mut detection = Vec::with_capacity(raw.len());
235    for t in raw {
236        let byte_start = t.start.offset as usize;
237        let byte_end = t.end.offset as usize;
238        push_token(
239            &mut detection,
240            t.kind,
241            &t.value,
242            byte_start,
243            byte_end,
244            t.start,
245            t.end,
246            options,
247        );
248    }
249    detection
250}
251
252fn dispatch_tokenizer(format: &str, source: &str, mode: Mode) -> Vec<Token> {
253    match format {
254        "javascript" | "typescript" | "jsx" | "tsx" => {
255            crate::javascript::tokenize_js(source, format)
256        }
257        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc(source, format, mode),
258        "razor" => crate::razor::tokenize_razor(source, mode),
259        "markdown" | "md" => crate::markdown::tokenize_markdown(source, mode),
260        _ => crate::generic::tokenize_generic(source, format),
261    }
262}
263
264/// Tokenize source code into one or more format-specific detection maps.
265///
266/// For single-format files, returns exactly one `TokenMap` with the same format.
267/// For multi-format files (markdown, SFCs), returns one `TokenMap` per detected
268/// sub-language — e.g. markdown prose + embedded JavaScript + embedded Python.
269///
270/// Each map's tokens carry byte offsets relative to the original source, so
271/// they can be used directly for clone detection within their format group.
272pub fn tokenize_to_detection_maps(
273    format: &str,
274    source: &str,
275    options: &TokenizeOptions,
276) -> Vec<TokenMap> {
277    match format {
278        "markdown" | "md" => crate::markdown::tokenize_markdown_maps(source, options),
279        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc_maps(source, format, options),
280        "razor" => crate::razor::tokenize_razor_maps(source, options),
281        _ => {
282            let tokens = tokenize_to_detection(format, source, options);
283            vec![TokenMap {
284                format: format.to_string(),
285                tokens,
286            }]
287        }
288    }
289}
290
291#[cfg(test)]
292mod tests {
293    use super::*;
294
295    #[test]
296    fn mode_from_str_defaults_to_mild() {
297        assert_eq!("unknown".parse::<Mode>().unwrap(), Mode::Mild);
298        assert_eq!("mild".parse::<Mode>().unwrap(), Mode::Mild);
299    }
300
301    #[test]
302    fn mode_from_str_weak() {
303        assert_eq!("weak".parse::<Mode>().unwrap(), Mode::Weak);
304    }
305
306    #[test]
307    fn mode_from_str_strict() {
308        assert_eq!("strict".parse::<Mode>().unwrap(), Mode::Strict);
309    }
310
311    #[test]
312    fn tokenize_to_detection_returns_detection_tokens() {
313        let opts = TokenizeOptions::new(Mode::Mild);
314        let tokens = tokenize_to_detection("javascript", "function hello() { return 42; }", &opts);
315        assert!(
316            !tokens.is_empty(),
317            "must produce DetectionTokens for valid JS"
318        );
319    }
320
321    #[test]
322    fn tokenize_to_detection_mild_excludes_whitespace() {
323        let opts = TokenizeOptions::new(Mode::Mild);
324        // The raw tokenizer produces whitespace tokens; mild mode drops them.
325        // We verify by counting: detection output should have fewer tokens than
326        // a strict-mode tokenize which keeps whitespace.
327        let mild = tokenize_to_detection("javascript", "a b c", &opts);
328        let strict =
329            tokenize_to_detection("javascript", "a b c", &TokenizeOptions::new(Mode::Strict));
330        // Mild must not exceed strict count (whitespace removed).
331        // Note: JS tokenizer doesn't produce Whitespace kind for OXC tokens,
332        // but the contract is that push_token correctly drops them if present.
333        let _ = (mild, strict);
334    }
335
336    #[test]
337    fn push_token_drops_ignore_kind() {
338        let mut tokens = Vec::new();
339        let loc = cpd_core::models::Location {
340            line: 1,
341            column: 0,
342            offset: 0,
343        };
344        let opts = TokenizeOptions::new(Mode::Mild);
345        push_token(
346            &mut tokens,
347            TokenKind::Ignore,
348            "secret",
349            0,
350            6,
351            loc.clone(),
352            loc,
353            &opts,
354        );
355        assert!(tokens.is_empty(), "Ignore-kind tokens must be dropped");
356    }
357
358    #[test]
359    fn push_token_drops_whitespace_in_mild_mode() {
360        let mut tokens = Vec::new();
361        let loc = cpd_core::models::Location {
362            line: 1,
363            column: 0,
364            offset: 0,
365        };
366        let opts = TokenizeOptions::new(Mode::Mild);
367        push_token(
368            &mut tokens,
369            TokenKind::Whitespace,
370            " ",
371            0,
372            1,
373            loc.clone(),
374            loc,
375            &opts,
376        );
377        assert!(tokens.is_empty(), "Whitespace must be dropped in Mild mode");
378    }
379
380    #[test]
381    fn push_token_keeps_whitespace_in_strict_mode() {
382        let mut tokens = Vec::new();
383        let loc = cpd_core::models::Location {
384            line: 1,
385            column: 0,
386            offset: 0,
387        };
388        let opts = TokenizeOptions::new(Mode::Strict);
389        push_token(
390            &mut tokens,
391            TokenKind::Whitespace,
392            " ",
393            0,
394            1,
395            loc.clone(),
396            loc,
397            &opts,
398        );
399        assert_eq!(tokens.len(), 1, "Whitespace must be kept in Strict mode");
400    }
401
402    #[test]
403    fn push_token_drops_comment_in_weak_mode() {
404        let mut tokens = Vec::new();
405        let loc = cpd_core::models::Location {
406            line: 1,
407            column: 0,
408            offset: 0,
409        };
410        let opts = TokenizeOptions::new(Mode::Weak);
411        push_token(
412            &mut tokens,
413            TokenKind::Comment,
414            "// note",
415            0,
416            7,
417            loc.clone(),
418            loc,
419            &opts,
420        );
421        assert!(tokens.is_empty(), "Comment must be dropped in Weak mode");
422    }
423
424    #[test]
425    fn push_token_ignore_case_folds_hash() {
426        let mut t1 = Vec::new();
427        let mut t2 = Vec::new();
428        let loc = cpd_core::models::Location {
429            line: 1,
430            column: 0,
431            offset: 0,
432        };
433        let mut opts = TokenizeOptions::new(Mode::Mild);
434        opts.ignore_case = true;
435        push_token(
436            &mut t1,
437            TokenKind::Identifier,
438            "Hello",
439            0,
440            5,
441            loc.clone(),
442            loc.clone(),
443            &opts,
444        );
445        push_token(
446            &mut t2,
447            TokenKind::Identifier,
448            "hello",
449            0,
450            5,
451            loc.clone(),
452            loc,
453            &opts,
454        );
455        assert_eq!(t1[0].hash, t2[0].hash, "ignore_case must fold case in hash");
456    }
457
458    #[test]
459    fn push_token_code_ignore_range_skips_overlapping_token() {
460        // Simulate: source = "foo// cpd-disable"
461        // regex "//\\s*cpd-disable" matches bytes 3..18
462        // Token "foo" is at 0..3 (no overlap -> kept)
463        // Token "// cpd-disable" is at 3..18 (overlaps -> skipped)
464        let mut tokens = Vec::new();
465        let loc = cpd_core::models::Location {
466            line: 1,
467            column: 0,
468            offset: 0,
469        };
470        let mut opts = TokenizeOptions::new(Mode::Mild);
471        // Pre-computed byte ranges from regex match on source text
472        opts.ignore_ranges = vec![[3, 18]];
473        push_token(
474            &mut tokens,
475            TokenKind::Identifier,
476            "foo",
477            0,
478            3,
479            loc.clone(),
480            loc.clone(),
481            &opts,
482        );
483        push_token(
484            &mut tokens,
485            TokenKind::Comment,
486            "// cpd-disable",
487            3,
488            18,
489            loc.clone(),
490            loc,
491            &opts,
492        );
493        assert_eq!(tokens.len(), 1, "only the non-matching token should remain");
494        assert_eq!(tokens[0].range, [0, 3]);
495    }
496
497    #[test]
498    fn push_token_code_ignore_range_no_overlap_keeps_all() {
499        // regex match at bytes 100..120 doesn't overlap tokens at 0..3, 3..6
500        let mut tokens = Vec::new();
501        let loc = cpd_core::models::Location {
502            line: 1,
503            column: 0,
504            offset: 0,
505        };
506        let mut opts = TokenizeOptions::new(Mode::Mild);
507        opts.ignore_ranges = vec![[100, 120]];
508        push_token(
509            &mut tokens,
510            TokenKind::Identifier,
511            "foo",
512            0,
513            3,
514            loc.clone(),
515            loc.clone(),
516            &opts,
517        );
518        push_token(
519            &mut tokens,
520            TokenKind::Identifier,
521            "bar",
522            3,
523            6,
524            loc.clone(),
525            loc,
526            &opts,
527        );
528        assert_eq!(
529            tokens.len(),
530            2,
531            "both tokens should remain when range doesn't overlap"
532        );
533    }
534
535    #[test]
536    fn code_ignore_ranges_computes_from_source_text() {
537        let source = "import foo from 'bar';\nconst x = 1;";
538        let re = regex::Regex::new(r"import\s+\w+\s+from").unwrap();
539        let ranges = code_ignore_ranges(source, &[re]);
540        assert_eq!(ranges.len(), 1, "should find one regex match");
541        // "import foo from" starts at byte 0, ends at byte 15
542        assert_eq!(ranges[0], [0, 15]);
543    }
544
545    #[test]
546    fn code_ignore_ranges_multiple_patterns() {
547        let source = "// MIT License\nfunction foo() {}\n// Copyright";
548        let re1 = regex::Regex::new(r"//\s*MIT\s+License").unwrap();
549        let re2 = regex::Regex::new(r"//\s*Copyright").unwrap();
550        let ranges = code_ignore_ranges(source, &[re1, re2]);
551        assert_eq!(ranges.len(), 2, "should find two regex matches");
552    }
553
554    #[test]
555    fn code_ignore_ranges_empty_regexes() {
556        let source = "function foo() {}";
557        let ranges = code_ignore_ranges(source, &[]);
558        assert!(ranges.is_empty(), "no regexes means no ranges");
559    }
560
561    #[test]
562    fn with_code_ignore_patterns_builds_regexes() {
563        let opts = TokenizeOptions::with_code_ignore_patterns(
564            Mode::Mild,
565            &["function".to_string(), r"//\s*cpd-disable".to_string()],
566        );
567        assert_eq!(opts.code_ignore_regexes.len(), 2);
568        assert!(opts.code_ignore_regexes[0].is_match("function"));
569        assert!(opts.code_ignore_regexes[1].is_match("// cpd-disable"));
570        assert!(!opts.code_ignore_regexes[1].is_match("function"));
571    }
572
573    #[test]
574    fn tokenize_to_detection_with_code_ignore_ranges_skips_imports() {
575        let source = "import * from 'lodash';\nconst x = 1;";
576        let regexes = vec![regex::Regex::new(r"import\s+\*\s+from").unwrap()];
577        let ranges = code_ignore_ranges(source, &regexes);
578        assert!(!ranges.is_empty(), "should find regex match in source");
579
580        let mut opts = TokenizeOptions::new(Mode::Mild);
581        opts.ignore_ranges = ranges;
582        let tokens = tokenize_to_detection("javascript", source, &opts);
583
584        // Tokens whose byte ranges overlap the import match should be skipped.
585        // "import" (0-6), "*" (7-8), "from" (9-13) should all be in range,
586        // but "const" (24-29) and "x" (30-31) etc should remain.
587        let has_const = tokens.iter().any(|t| {
588            // Check that tokens after the import line are still present
589            t.range[0] >= 24
590        });
591        assert!(
592            has_const,
593            "tokens after the import line should still be present"
594        );
595    }
596
597    #[test]
598    fn code_ignore_ranges_multi_token_match() {
599        // The key test: regex "import.*from" matches multi-token source text
600        // like "import * from 'module-name'" — not just a single token value.
601        let source = "import * from 'lodash';\nconst result = 42;";
602        let re = regex::Regex::new(r"import\s+.*?\s+from").unwrap();
603        let ranges = code_ignore_ranges(source, &[re]);
604        assert_eq!(
605            ranges.len(),
606            1,
607            "should find one regex match spanning import statement"
608        );
609        assert!(ranges[0][0] == 0, "match should start at beginning");
610        assert!(ranges[0][1] > 0, "match should have non-zero end");
611    }
612}