Skip to main content

cpd_tokenizer/
tokenizer.rs

1use std::str::FromStr;
2
3use cpd_core::hash::hash_token;
4use cpd_core::models::{DetectionToken, Token, TokenKind};
5
6use crate::markdown::tokens_to_detection;
7
8/// A sub-format detection map produced by multi-format tokenizers.
9///
10/// For single-format files, `tokenize_to_detection_maps()` returns exactly one
11/// TokenMap with the same format as the file.
12///
13/// For multi-format files (markdown, SFC), one TokenMap is returned per
14/// detected sub-language, each carrying tokens that should enter that
15/// format's detection pool.
16#[derive(Debug, Clone)]
17pub struct TokenMap {
18    pub format: String,
19    pub tokens: Vec<DetectionToken>,
20}
21
22#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
23pub enum Mode {
24    #[default]
25    Mild,
26    Weak,
27    Strict,
28}
29
30impl FromStr for Mode {
31    type Err = ();
32
33    fn from_str(s: &str) -> Result<Self, Self::Err> {
34        match s {
35            "weak" => Ok(Self::Weak),
36            "strict" => Ok(Self::Strict),
37            _ => Ok(Self::Mild),
38        }
39    }
40}
41
42/// Options for the detection-path tokenizer.
43///
44/// Carries mode, case-folding flag, pre-parsed ignore-region byte ranges,
45/// and pre-compiled code-level regex patterns that skip matching tokens during detection.
46///
47/// Code-level ignore patterns (v4 `ignorePattern`) work by matching regex patterns
48/// against source text, collecting byte ranges of matches, and then filtering
49/// any token whose byte range overlaps a match — identical in effect to v4's
50/// `setupIgnorePatterns` which injected Prism grammar tokens.
51#[derive(Debug, Clone)]
52pub struct TokenizeOptions {
53    pub mode: Mode,
54    /// When true, token values are lowercased before hashing.
55    pub ignore_case: bool,
56    /// Ignored byte ranges from `jscpd:ignore-start` / `jscpd:ignore-end`
57    /// and code-level regex matches from `ignorePattern`.
58    /// Each entry is `[start_byte, end_byte)`.
59    pub ignore_ranges: Vec<[usize; 2]>,
60    /// Hash every identifier as `$id` so clones that differ only in variable,
61    /// function or type names match (issue #998). Keywords are kept: the
62    /// JavaScript tokenizer classifies them, other languages fall back to
63    /// [`is_common_keyword`].
64    pub ignore_identifiers: bool,
65    /// Hash string literals as `$str` and numeric literals as `$num`.
66    pub ignore_literals: bool,
67    /// Drop `@Name`, `@a.b.Name` and `@Name(...)` annotation/decorator
68    /// sequences before hashing, in formats listed by [`strips_annotations`].
69    pub ignore_annotations: bool,
70    /// Pre-compiled code-level regex patterns inherited from v4 `ignorePattern`.
71    /// Before tokenization, these are matched against the source text and
72    /// overlapping byte ranges are added to `ignore_ranges`.
73    pub code_ignore_regexes: Vec<regex::Regex>,
74    /// Formats whose TypeScript-only syntax is stripped from the detection
75    /// token stream (`--cross-formats` groups mixing TS with JS). Only
76    /// `typescript` and `tsx` are meaningful here; empty by default so the
77    /// standard detection path is untouched.
78    pub strip_types_formats: std::collections::HashSet<String>,
79}
80
81impl TokenizeOptions {
82    pub fn new(mode: Mode) -> Self {
83        Self {
84            mode,
85            ignore_case: false,
86            ignore_identifiers: false,
87            ignore_literals: false,
88            ignore_annotations: false,
89            ignore_ranges: Vec::new(),
90            code_ignore_regexes: Vec::new(),
91            strip_types_formats: std::collections::HashSet::new(),
92        }
93    }
94}
95
96/// Tokenize a single-format source snippet into detection tokens.
97///
98/// Used by markdown and SFC tokenizers to dispatch embedded code blocks to the
99/// appropriate language tokenizer.
100pub fn tokenize_format_to_detection(
101    format: &str,
102    source: &str,
103    options: &TokenizeOptions,
104) -> Vec<DetectionToken> {
105    let raw = match format {
106        "javascript" | "typescript" | "jsx" | "tsx" => {
107            if should_strip_types(format, options) {
108                crate::javascript::tokenize_js_stripped(source, format)
109            } else {
110                crate::javascript::tokenize_js(source, format)
111            }
112        }
113        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc(source, format, options.mode),
114        "markdown" | "md" => crate::generic::tokenize_generic(source, format),
115        _ => crate::generic::tokenize_generic(source, format),
116    };
117    tokens_to_detection(raw, options)
118}
119
120/// True when this format's TypeScript-only syntax must be stripped for
121/// cross-format detection (see `TokenizeOptions::strip_types_formats`).
122fn should_strip_types(format: &str, options: &TokenizeOptions) -> bool {
123    matches!(format, "typescript" | "tsx") && options.strip_types_formats.contains(format)
124}
125
126/// Compute byte ranges of all regex matches against source text.
127/// Used to populate `ignore_ranges` from `ignorePattern` regexes before
128/// tokenization, matching v4 semantics where regex patterns match against
129/// source text regions (not individual token values).
130pub fn code_ignore_ranges(source: &str, regexes: &[regex::Regex]) -> Vec<[usize; 2]> {
131    let mut ranges = Vec::new();
132    for re in regexes {
133        for m in re.find_iter(source) {
134            ranges.push([m.start(), m.end()]);
135        }
136    }
137    ranges
138}
139
140/// Push a token into the detection output if it passes all filters.
141///
142/// Filtering happens here — at tokenize time — so the resulting
143/// `Vec<DetectionToken>` passed to detection is already minimal.
144/// Token values are not stored; only the pre-computed hash is kept.
145///
146/// The argument count is intentional: this function is a hot-path helper
147/// called from every tokenizer branch; grouping parameters into a struct
148/// would add an extra dereference per call.
149#[allow(clippy::too_many_arguments)]
150#[inline]
151pub fn push_token(
152    tokens: &mut Vec<DetectionToken>,
153    kind: TokenKind,
154    value: &str,
155    byte_start: usize,
156    byte_end: usize,
157    start: cpd_core::models::Location,
158    end: cpd_core::models::Location,
159    options: &TokenizeOptions,
160) {
161    // Drop Ignore-marked tokens in all modes.
162    if kind == TokenKind::Ignore {
163        return;
164    }
165    // Drop tokens in Ignore byte ranges.
166    // This covers both jscpd:ignore-start/end markers and code-level ignorePattern
167    // regex ranges (which are computed from source text before tokenization).
168    if options
169        .ignore_ranges
170        .iter()
171        .any(|[rs, re]| byte_start < *re && byte_end > *rs)
172    {
173        return;
174    }
175    // Mode-based filtering:
176    match options.mode {
177        Mode::Mild => {
178            if kind == TokenKind::Whitespace {
179                return;
180            }
181        }
182        Mode::Weak => {
183            if matches!(
184                kind,
185                TokenKind::Whitespace | TokenKind::Comment | TokenKind::BlockComment
186            ) {
187                return;
188            }
189        }
190        Mode::Strict => {} // keep everything (except Ignore, handled above)
191    }
192    let raw_hash = hash_token(kind.discriminant(), value, options.ignore_case);
193    let hash = match normalized_value(&kind, value, options) {
194        Some(placeholder) => hash_token(kind.discriminant(), placeholder, false),
195        None => raw_hash,
196    };
197    tokens.push(DetectionToken {
198        hash,
199        raw_hash,
200        start,
201        end,
202        range: [byte_start, byte_end],
203    });
204}
205
206/// Placeholder that replaces `value` under the active normalization options,
207/// or `None` when the token hashes as-is (issue #998).
208#[inline]
209fn normalized_value(
210    kind: &TokenKind,
211    value: &str,
212    options: &TokenizeOptions,
213) -> Option<&'static str> {
214    match kind {
215        TokenKind::Identifier if options.ignore_identifiers => {
216            if is_common_keyword(value) {
217                None
218            } else {
219                Some("$id")
220            }
221        }
222        TokenKind::Literal if options.ignore_literals => literal_placeholder(value),
223        _ => None,
224    }
225}
226
227/// `$str` for quoted literals (with an optional short alphabetic prefix such
228/// as Python's `r"..."` or C#'s `@"..."`), `$num` for numbers, `None` for
229/// anything else (`true`, `null`, regex literals) which keeps its own hash.
230fn literal_placeholder(value: &str) -> Option<&'static str> {
231    let bytes = value.as_bytes();
232    let first = *bytes.first()?;
233    if matches!(first, b'"' | b'\'' | b'`') {
234        return Some("$str");
235    }
236    if first.is_ascii_digit() {
237        return Some("$num");
238    }
239    if first == b'.' && bytes.get(1).is_some_and(u8::is_ascii_digit) {
240        return Some("$num");
241    }
242    if first.is_ascii_alphabetic() || first == b'@' {
243        let quote_at = bytes
244            .iter()
245            .take(4)
246            .position(|b| matches!(b, b'"' | b'\'' | b'`'));
247        if quote_at.is_some() {
248            return Some("$str");
249        }
250    }
251    None
252}
253
254/// Keywords the generic tokenizer reports as identifiers. Kept verbatim under
255/// `--ignore-identifiers` so control flow still has to match; the list is the
256/// union of common keywords across C-like, Python-like and ML-like languages.
257/// Must stay sorted: looked up by binary search.
258static COMMON_KEYWORDS: &[&str] = &[
259    "abstract",
260    "and",
261    "as",
262    "assert",
263    "async",
264    "await",
265    "begin",
266    "break",
267    "case",
268    "catch",
269    "class",
270    "const",
271    "continue",
272    "def",
273    "default",
274    "defer",
275    "del",
276    "do",
277    "elif",
278    "else",
279    "elsif",
280    "end",
281    "enum",
282    "except",
283    "export",
284    "extends",
285    "extern",
286    "false",
287    "final",
288    "finally",
289    "fn",
290    "for",
291    "foreach",
292    "from",
293    "func",
294    "function",
295    "global",
296    "go",
297    "goto",
298    "if",
299    "impl",
300    "implements",
301    "import",
302    "in",
303    "inline",
304    "instanceof",
305    "interface",
306    "is",
307    "lambda",
308    "let",
309    "loop",
310    "match",
311    "mod",
312    "module",
313    "mut",
314    "namespace",
315    "new",
316    "nil",
317    "none",
318    "not",
319    "null",
320    "or",
321    "override",
322    "package",
323    "pass",
324    "private",
325    "protected",
326    "pub",
327    "public",
328    "raise",
329    "record",
330    "ref",
331    "require",
332    "rescue",
333    "return",
334    "sealed",
335    "select",
336    "self",
337    "sizeof",
338    "static",
339    "struct",
340    "super",
341    "switch",
342    "then",
343    "this",
344    "throw",
345    "throws",
346    "trait",
347    "true",
348    "try",
349    "type",
350    "typedef",
351    "typeof",
352    "undefined",
353    "union",
354    "unless",
355    "unsafe",
356    "until",
357    "use",
358    "using",
359    "var",
360    "virtual",
361    "void",
362    "volatile",
363    "when",
364    "where",
365    "while",
366    "with",
367    "yield",
368];
369
370/// True for words in [`COMMON_KEYWORDS`] (case-sensitive).
371pub fn is_common_keyword(word: &str) -> bool {
372    COMMON_KEYWORDS.binary_search(&word).is_ok()
373}
374
375/// Formats where `@Name` / `@Name(...)` means an annotation or decorator.
376/// Elsewhere (`Ruby`, `Perl`, `T-SQL`, `Razor`, `CSS`) `@` prefixes variables
377/// or directives and must not be dropped.
378pub fn strips_annotations(format: &str) -> bool {
379    matches!(
380        format,
381        "javascript"
382            | "typescript"
383            | "jsx"
384            | "tsx"
385            | "java"
386            | "kotlin"
387            | "scala"
388            | "groovy"
389            | "python"
390            | "dart"
391            | "swift"
392    )
393}
394
395/// Flag `@Name`, `@a.b.Name` and `@Name(...)` token runs so the detection
396/// path drops them (issue #998). Whitespace tokens inside a run are tolerated
397/// only between the name and its argument list. Returns one flag per token,
398/// or an empty vector when nothing matched.
399fn mark_annotations(tokens: &[Token]) -> Vec<bool> {
400    let n = tokens.len();
401    let mut i = 0;
402    let mut flags: Vec<bool> = Vec::new();
403    while i < n {
404        // `@` followed by an identifier — but never by a keyword: Java's
405        // `@interface Name { … }` declares an annotation type rather than
406        // applying one, and the generic tokenizer reports `interface` as an
407        // identifier.
408        let starts_annotation = tokens[i].value == "@"
409            && tokens[i].kind != TokenKind::Ignore
410            && tokens
411                .get(i + 1)
412                .is_some_and(|t| t.kind == TokenKind::Identifier && !is_common_keyword(&t.value));
413        if !starts_annotation {
414            i += 1;
415            continue;
416        }
417        let start = i;
418        let mut j = i + 2;
419        while j + 1 < n && tokens[j].value == "." && tokens[j + 1].kind == TokenKind::Identifier {
420            j += 2;
421        }
422        let mut k = j;
423        while k < n && tokens[k].kind == TokenKind::Whitespace {
424            k += 1;
425        }
426        if k < n && tokens[k].value == "(" {
427            let mut depth = 0usize;
428            j = k;
429            while j < n {
430                match tokens[j].value.as_str() {
431                    "(" => depth += 1,
432                    ")" => {
433                        depth -= 1;
434                        if depth == 0 {
435                            j += 1;
436                            break;
437                        }
438                    }
439                    _ => {}
440                }
441                j += 1;
442            }
443        }
444        if flags.is_empty() {
445            flags = vec![false; n];
446        }
447        for flag in &mut flags[start..j] {
448            *flag = true;
449        }
450        i = j;
451    }
452    flags
453}
454
455/// Tokenize source code in the given format with the given mode.
456/// Returns a Vec<Token>. Never panics on empty input — returns empty Vec.
457///
458/// This is the display/reporter path. For the detection path, use
459/// `tokenize_to_detection`.
460pub fn tokenize(format: &str, source: &str, mode: Mode) -> Vec<Token> {
461    let raw = dispatch_tokenizer(format, source, mode);
462    // Apply mode filter inline — keeps Ignore tokens removed, drops Whitespace in
463    // Mild, drops Whitespace+Comment+BlockComment in Weak, keeps all in Strict.
464    raw.into_iter().filter(|t| keep_token(t, mode)).collect()
465}
466
467fn keep_token(token: &Token, mode: Mode) -> bool {
468    if token.kind == TokenKind::Ignore {
469        return false;
470    }
471    match mode {
472        Mode::Mild => !matches!(token.kind, TokenKind::Whitespace),
473        Mode::Weak => !matches!(
474            token.kind,
475            TokenKind::Whitespace | TokenKind::Comment | TokenKind::BlockComment
476        ),
477        Mode::Strict => true,
478    }
479}
480
481/// Tokenize source code for the detection hot path.
482///
483/// Returns `Vec<DetectionToken>` — tokens filtered and hashed inline at
484/// tokenize time. No per-token heap allocation survives in the output:
485/// the value string is consumed; only the hash, locations, and byte range
486/// are stored.
487///
488/// This replaces the `tokenize` → `apply_mode` → convert-to-hashes pipeline
489/// that existed in `detect.rs`.
490pub fn tokenize_to_detection(
491    format: &str,
492    source: &str,
493    options: &TokenizeOptions,
494) -> Vec<DetectionToken> {
495    // Produce the display tokens first (reuse existing tokenizer code),
496    // then convert to DetectionToken in one pass applying options filters.
497    //
498    // This approach is conservative: it reuses all existing tokenizer logic
499    // without risk of introducing per-tokenizer bugs. The conversion is O(n)
500    // and eliminates the separate filter pass and hash computation that
501    // previously happened inside detect.rs.
502    let raw = if should_strip_types(format, options) {
503        crate::javascript::tokenize_js_stripped(source, format)
504    } else {
505        dispatch_tokenizer(format, source, options.mode)
506    };
507    let annotation = if options.ignore_annotations && strips_annotations(format) {
508        mark_annotations(&raw)
509    } else {
510        Vec::new()
511    };
512    let mut detection: Vec<DetectionToken> = Vec::with_capacity(raw.len());
513    for (i, t) in raw.into_iter().enumerate() {
514        if annotation.get(i).copied().unwrap_or(false) {
515            // Dropped annotation: fold its text into the raw hash of the token
516            // that precedes it. A clone whose matched run contains that token
517            // then classifies as `renamed` when the annotations differ, while
518            // a run that starts after the annotation is unaffected and stays
519            // `exact` — the reported fragment text really is identical there.
520            if let Some(prev) = detection.last_mut() {
521                let h = hash_token(t.kind.discriminant(), &t.value, false);
522                prev.raw_hash = prev.raw_hash.rotate_left(7) ^ h;
523            }
524            continue;
525        }
526        let byte_start = t.start.offset as usize;
527        let byte_end = t.end.offset as usize;
528        push_token(
529            &mut detection,
530            t.kind,
531            &t.value,
532            byte_start,
533            byte_end,
534            t.start,
535            t.end,
536            options,
537        );
538    }
539    detection
540}
541
542fn dispatch_tokenizer(format: &str, source: &str, mode: Mode) -> Vec<Token> {
543    match format {
544        "javascript" | "typescript" | "jsx" | "tsx" => {
545            crate::javascript::tokenize_js(source, format)
546        }
547        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc(source, format, mode),
548        "razor" => crate::razor::tokenize_razor(source, mode),
549        "markdown" | "md" => crate::markdown::tokenize_markdown(source, mode),
550        _ => crate::generic::tokenize_generic(source, format),
551    }
552}
553
554/// Tokenize source code into one or more format-specific detection maps.
555///
556/// For single-format files, returns exactly one `TokenMap` with the same format.
557/// For multi-format files (markdown, SFCs), returns one `TokenMap` per detected
558/// sub-language — e.g. markdown prose + embedded JavaScript + embedded Python.
559///
560/// Each map's tokens carry byte offsets relative to the original source, so
561/// they can be used directly for clone detection within their format group.
562pub fn tokenize_to_detection_maps(
563    format: &str,
564    source: &str,
565    options: &TokenizeOptions,
566) -> Vec<TokenMap> {
567    match format {
568        "markdown" | "md" => crate::markdown::tokenize_markdown_maps(source, options),
569        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc_maps(source, format, options),
570        "razor" => crate::razor::tokenize_razor_maps(source, options),
571        _ => {
572            let tokens = tokenize_to_detection(format, source, options);
573            vec![TokenMap {
574                format: format.to_string(),
575                tokens,
576            }]
577        }
578    }
579}
580
581#[cfg(test)]
582mod tests {
583    use super::*;
584
585    #[test]
586    fn mode_from_str_defaults_to_mild() {
587        assert_eq!("unknown".parse::<Mode>().unwrap(), Mode::Mild);
588        assert_eq!("mild".parse::<Mode>().unwrap(), Mode::Mild);
589    }
590
591    #[test]
592    fn mode_from_str_weak() {
593        assert_eq!("weak".parse::<Mode>().unwrap(), Mode::Weak);
594    }
595
596    #[test]
597    fn mode_from_str_strict() {
598        assert_eq!("strict".parse::<Mode>().unwrap(), Mode::Strict);
599    }
600
601    #[test]
602    fn tokenize_to_detection_returns_detection_tokens() {
603        let opts = TokenizeOptions::new(Mode::Mild);
604        let tokens = tokenize_to_detection("javascript", "function hello() { return 42; }", &opts);
605        assert!(
606            !tokens.is_empty(),
607            "must produce DetectionTokens for valid JS"
608        );
609    }
610
611    #[test]
612    fn tokenize_to_detection_mild_excludes_whitespace() {
613        let opts = TokenizeOptions::new(Mode::Mild);
614        // The raw tokenizer produces whitespace tokens; mild mode drops them.
615        // We verify by counting: detection output should have fewer tokens than
616        // a strict-mode tokenize which keeps whitespace.
617        let mild = tokenize_to_detection("javascript", "a b c", &opts);
618        let strict =
619            tokenize_to_detection("javascript", "a b c", &TokenizeOptions::new(Mode::Strict));
620        // Mild must not exceed strict count (whitespace removed).
621        // Note: JS tokenizer doesn't produce Whitespace kind for OXC tokens,
622        // but the contract is that push_token correctly drops them if present.
623        let _ = (mild, strict);
624    }
625
626    #[test]
627    fn push_token_drops_ignore_kind() {
628        let mut tokens = Vec::new();
629        let loc = cpd_core::models::Location {
630            line: 1,
631            column: 0,
632            offset: 0,
633        };
634        let opts = TokenizeOptions::new(Mode::Mild);
635        push_token(
636            &mut tokens,
637            TokenKind::Ignore,
638            "secret",
639            0,
640            6,
641            loc.clone(),
642            loc,
643            &opts,
644        );
645        assert!(tokens.is_empty(), "Ignore-kind tokens must be dropped");
646    }
647
648    #[test]
649    fn push_token_drops_whitespace_in_mild_mode() {
650        let mut tokens = Vec::new();
651        let loc = cpd_core::models::Location {
652            line: 1,
653            column: 0,
654            offset: 0,
655        };
656        let opts = TokenizeOptions::new(Mode::Mild);
657        push_token(
658            &mut tokens,
659            TokenKind::Whitespace,
660            " ",
661            0,
662            1,
663            loc.clone(),
664            loc,
665            &opts,
666        );
667        assert!(tokens.is_empty(), "Whitespace must be dropped in Mild mode");
668    }
669
670    #[test]
671    fn push_token_keeps_whitespace_in_strict_mode() {
672        let mut tokens = Vec::new();
673        let loc = cpd_core::models::Location {
674            line: 1,
675            column: 0,
676            offset: 0,
677        };
678        let opts = TokenizeOptions::new(Mode::Strict);
679        push_token(
680            &mut tokens,
681            TokenKind::Whitespace,
682            " ",
683            0,
684            1,
685            loc.clone(),
686            loc,
687            &opts,
688        );
689        assert_eq!(tokens.len(), 1, "Whitespace must be kept in Strict mode");
690    }
691
692    #[test]
693    fn push_token_drops_comment_in_weak_mode() {
694        let mut tokens = Vec::new();
695        let loc = cpd_core::models::Location {
696            line: 1,
697            column: 0,
698            offset: 0,
699        };
700        let opts = TokenizeOptions::new(Mode::Weak);
701        push_token(
702            &mut tokens,
703            TokenKind::Comment,
704            "// note",
705            0,
706            7,
707            loc.clone(),
708            loc,
709            &opts,
710        );
711        assert!(tokens.is_empty(), "Comment must be dropped in Weak mode");
712    }
713
714    fn det(source: &str, format: &str, opts: &TokenizeOptions) -> Vec<DetectionToken> {
715        tokenize_to_detection(format, source, opts)
716    }
717
718    fn hashes(tokens: &[DetectionToken]) -> Vec<u64> {
719        tokens.iter().map(|t| t.hash).collect()
720    }
721
722    #[test]
723    fn default_options_keep_raw_hash_equal_to_hash() {
724        let opts = TokenizeOptions::new(Mode::Mild);
725        let tokens = det("function a(x) { return x + 1; }", "javascript", &opts);
726        assert!(!tokens.is_empty());
727        assert!(tokens.iter().all(|t| t.raw_hash == t.hash));
728    }
729
730    #[test]
731    fn ignore_identifiers_matches_renamed_code_and_keeps_keywords() {
732        let mut opts = TokenizeOptions::new(Mode::Mild);
733        opts.ignore_identifiers = true;
734        let a = det("function a(x) { return x + 1; }", "javascript", &opts);
735        let b = det("function b(y) { return y + 1; }", "javascript", &opts);
736        assert_eq!(
737            hashes(&a),
738            hashes(&b),
739            "renamed identifiers must hash alike"
740        );
741        // raw hashes still differ where the names differ
742        assert_ne!(
743            a.iter().map(|t| t.raw_hash).collect::<Vec<_>>(),
744            b.iter().map(|t| t.raw_hash).collect::<Vec<_>>()
745        );
746        // keywords are not folded: `return` vs `throw` must stay distinct
747        let c = det("function a(x) { throw x + 1; }", "javascript", &opts);
748        assert_ne!(hashes(&a), hashes(&c));
749    }
750
751    #[test]
752    fn ignore_identifiers_keeps_common_keywords_in_generic_languages() {
753        let mut opts = TokenizeOptions::new(Mode::Mild);
754        opts.ignore_identifiers = true;
755        let a = det("if x:\n    return y\n", "python", &opts);
756        let b = det("if p:\n    return q\n", "python", &opts);
757        let c = det("while x:\n    return y\n", "python", &opts);
758        assert_eq!(hashes(&a), hashes(&b));
759        assert_ne!(
760            hashes(&a),
761            hashes(&c),
762            "`if` and `while` are keywords, not identifiers"
763        );
764        assert!(is_common_keyword("return"));
765        assert!(!is_common_keyword("total"));
766    }
767
768    #[test]
769    fn ignore_literals_folds_strings_and_numbers_separately() {
770        let mut opts = TokenizeOptions::new(Mode::Mild);
771        opts.ignore_literals = true;
772        let a = det("const a = 10; const b = 'x';", "javascript", &opts);
773        let b = det("const a = 25; const b = \"yy\";", "javascript", &opts);
774        assert_eq!(hashes(&a), hashes(&b));
775        let c = det("const a = 'ten'; const b = 'x';", "javascript", &opts);
776        assert_ne!(hashes(&a), hashes(&c), "a string is not a number");
777        assert_eq!(literal_placeholder("42"), Some("$num"));
778        assert_eq!(literal_placeholder(".5"), Some("$num"));
779        assert_eq!(literal_placeholder("\"s\""), Some("$str"));
780        assert_eq!(literal_placeholder("r'raw'"), Some("$str"));
781        assert_eq!(literal_placeholder("true"), None);
782    }
783
784    #[test]
785    fn ignore_annotations_drops_decorators_in_listed_formats_only() {
786        let mut opts = TokenizeOptions::new(Mode::Mild);
787        opts.ignore_annotations = true;
788        let plain = det("class A { m() { return 1; } }", "typescript", &opts);
789        let decorated = det(
790            "@Component({ selector: 'a' })\nclass A { @Input() m() { return 1; } }",
791            "typescript",
792            &opts,
793        );
794        assert_eq!(hashes(&plain), hashes(&decorated));
795        // the token before an inner annotation carries a salted raw hash
796        assert!(decorated.iter().any(|t| t.raw_hash != t.hash));
797
798        let java_a = det(
799            "class A {\n  @Override\n  int f() { return 1; }\n}",
800            "java",
801            &opts,
802        );
803        let java_b = det(
804            "class A {\n  @Deprecated\n  int f() { return 1; }\n}",
805            "java",
806            &opts,
807        );
808        assert_eq!(hashes(&java_a), hashes(&java_b));
809        assert_ne!(
810            java_a.iter().map(|t| t.raw_hash).collect::<Vec<_>>(),
811            java_b.iter().map(|t| t.raw_hash).collect::<Vec<_>>(),
812            "different annotations must leave different raw hashes"
813        );
814
815        // `@interface` declares an annotation type; it is not an annotation use
816        let decl = "public @interface Marker {\n  String value() default \"\";\n}";
817        let stripped = det(decl, "java", &opts);
818        let untouched = det(decl, "java", &TokenizeOptions::new(Mode::Mild));
819        assert_eq!(hashes(&stripped), hashes(&untouched));
820
821        // Ruby instance variables use `@` and must survive
822        let ruby = det("@count = 1", "ruby", &opts);
823        assert!(strips_annotations("kotlin"));
824        assert!(!strips_annotations("ruby"));
825        assert_eq!(
826            ruby.len(),
827            det("@count = 1", "ruby", &TokenizeOptions::new(Mode::Mild)).len()
828        );
829    }
830
831    #[test]
832    fn push_token_ignore_case_folds_hash() {
833        let mut t1 = Vec::new();
834        let mut t2 = Vec::new();
835        let loc = cpd_core::models::Location {
836            line: 1,
837            column: 0,
838            offset: 0,
839        };
840        let mut opts = TokenizeOptions::new(Mode::Mild);
841        opts.ignore_case = true;
842        push_token(
843            &mut t1,
844            TokenKind::Identifier,
845            "Hello",
846            0,
847            5,
848            loc.clone(),
849            loc.clone(),
850            &opts,
851        );
852        push_token(
853            &mut t2,
854            TokenKind::Identifier,
855            "hello",
856            0,
857            5,
858            loc.clone(),
859            loc,
860            &opts,
861        );
862        assert_eq!(t1[0].hash, t2[0].hash, "ignore_case must fold case in hash");
863    }
864
865    #[test]
866    fn push_token_code_ignore_range_skips_overlapping_token() {
867        // Simulate: source = "foo// cpd-disable"
868        // regex "//\\s*cpd-disable" matches bytes 3..18
869        // Token "foo" is at 0..3 (no overlap -> kept)
870        // Token "// cpd-disable" is at 3..18 (overlaps -> skipped)
871        let mut tokens = Vec::new();
872        let loc = cpd_core::models::Location {
873            line: 1,
874            column: 0,
875            offset: 0,
876        };
877        let mut opts = TokenizeOptions::new(Mode::Mild);
878        // Pre-computed byte ranges from regex match on source text
879        opts.ignore_ranges = vec![[3, 18]];
880        push_token(
881            &mut tokens,
882            TokenKind::Identifier,
883            "foo",
884            0,
885            3,
886            loc.clone(),
887            loc.clone(),
888            &opts,
889        );
890        push_token(
891            &mut tokens,
892            TokenKind::Comment,
893            "// cpd-disable",
894            3,
895            18,
896            loc.clone(),
897            loc,
898            &opts,
899        );
900        assert_eq!(tokens.len(), 1, "only the non-matching token should remain");
901        assert_eq!(tokens[0].range, [0, 3]);
902    }
903
904    #[test]
905    fn push_token_code_ignore_range_no_overlap_keeps_all() {
906        // regex match at bytes 100..120 doesn't overlap tokens at 0..3, 3..6
907        let mut tokens = Vec::new();
908        let loc = cpd_core::models::Location {
909            line: 1,
910            column: 0,
911            offset: 0,
912        };
913        let mut opts = TokenizeOptions::new(Mode::Mild);
914        opts.ignore_ranges = vec![[100, 120]];
915        push_token(
916            &mut tokens,
917            TokenKind::Identifier,
918            "foo",
919            0,
920            3,
921            loc.clone(),
922            loc.clone(),
923            &opts,
924        );
925        push_token(
926            &mut tokens,
927            TokenKind::Identifier,
928            "bar",
929            3,
930            6,
931            loc.clone(),
932            loc,
933            &opts,
934        );
935        assert_eq!(
936            tokens.len(),
937            2,
938            "both tokens should remain when range doesn't overlap"
939        );
940    }
941
942    #[test]
943    fn code_ignore_ranges_computes_from_source_text() {
944        let source = "import foo from 'bar';\nconst x = 1;";
945        let re = regex::Regex::new(r"import\s+\w+\s+from").unwrap();
946        let ranges = code_ignore_ranges(source, &[re]);
947        assert_eq!(ranges.len(), 1, "should find one regex match");
948        // "import foo from" starts at byte 0, ends at byte 15
949        assert_eq!(ranges[0], [0, 15]);
950    }
951
952    #[test]
953    fn code_ignore_ranges_multiple_patterns() {
954        let source = "// MIT License\nfunction foo() {}\n// Copyright";
955        let re1 = regex::Regex::new(r"//\s*MIT\s+License").unwrap();
956        let re2 = regex::Regex::new(r"//\s*Copyright").unwrap();
957        let ranges = code_ignore_ranges(source, &[re1, re2]);
958        assert_eq!(ranges.len(), 2, "should find two regex matches");
959    }
960
961    #[test]
962    fn code_ignore_ranges_empty_regexes() {
963        let source = "function foo() {}";
964        let ranges = code_ignore_ranges(source, &[]);
965        assert!(ranges.is_empty(), "no regexes means no ranges");
966    }
967
968    #[test]
969    fn tokenize_to_detection_with_code_ignore_ranges_skips_imports() {
970        let source = "import * from 'lodash';\nconst x = 1;";
971        let regexes = vec![regex::Regex::new(r"import\s+\*\s+from").unwrap()];
972        let ranges = code_ignore_ranges(source, &regexes);
973        assert!(!ranges.is_empty(), "should find regex match in source");
974
975        let mut opts = TokenizeOptions::new(Mode::Mild);
976        opts.ignore_ranges = ranges;
977        let tokens = tokenize_to_detection("javascript", source, &opts);
978
979        // Tokens whose byte ranges overlap the import match should be skipped.
980        // "import" (0-6), "*" (7-8), "from" (9-13) should all be in range,
981        // but "const" (24-29) and "x" (30-31) etc should remain.
982        let has_const = tokens.iter().any(|t| {
983            // Check that tokens after the import line are still present
984            t.range[0] >= 24
985        });
986        assert!(
987            has_const,
988            "tokens after the import line should still be present"
989        );
990    }
991
992    #[test]
993    fn code_ignore_ranges_multi_token_match() {
994        // The key test: regex "import.*from" matches multi-token source text
995        // like "import * from 'module-name'" — not just a single token value.
996        let source = "import * from 'lodash';\nconst result = 42;";
997        let re = regex::Regex::new(r"import\s+.*?\s+from").unwrap();
998        let ranges = code_ignore_ranges(source, &[re]);
999        assert_eq!(
1000            ranges.len(),
1001            1,
1002            "should find one regex match spanning import statement"
1003        );
1004        assert!(ranges[0][0] == 0, "match should start at beginning");
1005        assert!(ranges[0][1] > 0, "match should have non-zero end");
1006    }
1007}