Skip to main content

cpd_tokenizer/
tokenizer.rs

1use std::str::FromStr;
2
3use cpd_core::hash::hash_token;
4use cpd_core::models::{DetectionToken, Token, TokenKind};
5
6use crate::markdown::tokens_to_detection;
7
8/// A sub-format detection map produced by multi-format tokenizers.
9///
10/// For single-format files, `tokenize_to_detection_maps()` returns exactly one
11/// TokenMap with the same format as the file.
12///
13/// For multi-format files (markdown, SFC), one TokenMap is returned per
14/// detected sub-language, each carrying tokens that should enter that
15/// format's detection pool.
16#[derive(Debug, Clone)]
17pub struct TokenMap {
18    pub format: String,
19    pub tokens: Vec<DetectionToken>,
20}
21
22#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
23pub enum Mode {
24    #[default]
25    Mild,
26    Weak,
27    Strict,
28}
29
30impl FromStr for Mode {
31    type Err = ();
32
33    fn from_str(s: &str) -> Result<Self, Self::Err> {
34        match s {
35            "weak" => Ok(Self::Weak),
36            "strict" => Ok(Self::Strict),
37            _ => Ok(Self::Mild),
38        }
39    }
40}
41
42/// Options for the detection-path tokenizer.
43///
44/// Carries mode, case-folding flag, pre-parsed ignore-region byte ranges,
45/// and pre-compiled code-level regex patterns that skip matching tokens during detection.
46///
47/// Code-level ignore patterns (v4 `ignorePattern`) work by matching regex patterns
48/// against source text, collecting byte ranges of matches, and then filtering
49/// any token whose byte range overlaps a match — identical in effect to v4's
50/// `setupIgnorePatterns` which injected Prism grammar tokens.
51#[derive(Debug, Clone)]
52pub struct TokenizeOptions {
53    pub mode: Mode,
54    /// When true, token values are lowercased before hashing.
55    pub ignore_case: bool,
56    /// Ignored byte ranges from `jscpd:ignore-start` / `jscpd:ignore-end`
57    /// and code-level regex matches from `ignorePattern`.
58    /// Each entry is `[start_byte, end_byte)`.
59    pub ignore_ranges: Vec<[usize; 2]>,
60    /// Hash every identifier as `$id` so clones that differ only in variable,
61    /// function or type names match (issue #998). Keywords are kept: the
62    /// JavaScript tokenizer classifies them, other languages fall back to
63    /// [`is_common_keyword`].
64    pub ignore_identifiers: bool,
65    /// Hash string literals as `$str` and numeric literals as `$num`.
66    pub ignore_literals: bool,
67    /// Drop `@Name`, `@a.b.Name` and `@Name(...)` annotation/decorator
68    /// sequences before hashing, in formats listed by [`strips_annotations`].
69    pub ignore_annotations: bool,
70    /// Pre-compiled code-level regex patterns inherited from v4 `ignorePattern`.
71    /// Before tokenization, these are matched against the source text and
72    /// overlapping byte ranges are added to `ignore_ranges`.
73    pub code_ignore_regexes: Vec<regex::Regex>,
74    /// Formats whose TypeScript-only syntax is stripped from the detection
75    /// token stream (`--cross-formats` groups mixing TS with JS). Only
76    /// `typescript` and `tsx` are meaningful here; empty by default so the
77    /// standard detection path is untouched.
78    pub strip_types_formats: std::collections::HashSet<String>,
79}
80
81impl TokenizeOptions {
82    pub fn new(mode: Mode) -> Self {
83        Self {
84            mode,
85            ignore_case: false,
86            ignore_identifiers: false,
87            ignore_literals: false,
88            ignore_annotations: false,
89            ignore_ranges: Vec::new(),
90            code_ignore_regexes: Vec::new(),
91            strip_types_formats: std::collections::HashSet::new(),
92        }
93    }
94
95    /// True when any Type-2 normalization option is on.
96    pub fn normalizes(&self) -> bool {
97        self.ignore_identifiers || self.ignore_literals || self.ignore_annotations
98    }
99
100    /// Build TokenizeOptions with pre-compiled regex patterns from string patterns.
101    /// Invalid regex patterns are silently skipped.
102    pub fn with_code_ignore_patterns(mode: Mode, patterns: &[String]) -> Self {
103        let code_ignore_regexes: Vec<regex::Regex> = patterns
104            .iter()
105            .filter_map(|p| regex::Regex::new(p).ok())
106            .collect();
107        Self {
108            mode,
109            ignore_case: false,
110            ignore_identifiers: false,
111            ignore_literals: false,
112            ignore_annotations: false,
113            ignore_ranges: Vec::new(),
114            code_ignore_regexes,
115            strip_types_formats: std::collections::HashSet::new(),
116        }
117    }
118}
119
120/// Tokenize a single-format source snippet into detection tokens.
121///
122/// Used by markdown and SFC tokenizers to dispatch embedded code blocks to the
123/// appropriate language tokenizer.
124pub fn tokenize_format_to_detection(
125    format: &str,
126    source: &str,
127    options: &TokenizeOptions,
128) -> Vec<DetectionToken> {
129    let raw = match format {
130        "javascript" | "typescript" | "jsx" | "tsx" => {
131            if should_strip_types(format, options) {
132                crate::javascript::tokenize_js_stripped(source, format)
133            } else {
134                crate::javascript::tokenize_js(source, format)
135            }
136        }
137        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc(source, format, options.mode),
138        "markdown" | "md" => crate::generic::tokenize_generic(source, format),
139        _ => crate::generic::tokenize_generic(source, format),
140    };
141    tokens_to_detection(raw, options)
142}
143
144/// True when this format's TypeScript-only syntax must be stripped for
145/// cross-format detection (see `TokenizeOptions::strip_types_formats`).
146fn should_strip_types(format: &str, options: &TokenizeOptions) -> bool {
147    matches!(format, "typescript" | "tsx") && options.strip_types_formats.contains(format)
148}
149
150/// Compute byte ranges of all regex matches against source text.
151/// Used to populate `ignore_ranges` from `ignorePattern` regexes before
152/// tokenization, matching v4 semantics where regex patterns match against
153/// source text regions (not individual token values).
154pub fn code_ignore_ranges(source: &str, regexes: &[regex::Regex]) -> Vec<[usize; 2]> {
155    let mut ranges = Vec::new();
156    for re in regexes {
157        for m in re.find_iter(source) {
158            ranges.push([m.start(), m.end()]);
159        }
160    }
161    ranges
162}
163
164/// Push a token into the detection output if it passes all filters.
165///
166/// Filtering happens here — at tokenize time — so the resulting
167/// `Vec<DetectionToken>` passed to detection is already minimal.
168/// Token values are not stored; only the pre-computed hash is kept.
169///
170/// The argument count is intentional: this function is a hot-path helper
171/// called from every tokenizer branch; grouping parameters into a struct
172/// would add an extra dereference per call.
173#[allow(clippy::too_many_arguments)]
174#[inline]
175pub fn push_token(
176    tokens: &mut Vec<DetectionToken>,
177    kind: TokenKind,
178    value: &str,
179    byte_start: usize,
180    byte_end: usize,
181    start: cpd_core::models::Location,
182    end: cpd_core::models::Location,
183    options: &TokenizeOptions,
184) {
185    // Drop Ignore-marked tokens in all modes.
186    if kind == TokenKind::Ignore {
187        return;
188    }
189    // Drop tokens in Ignore byte ranges.
190    // This covers both jscpd:ignore-start/end markers and code-level ignorePattern
191    // regex ranges (which are computed from source text before tokenization).
192    if options
193        .ignore_ranges
194        .iter()
195        .any(|[rs, re]| byte_start < *re && byte_end > *rs)
196    {
197        return;
198    }
199    // Mode-based filtering:
200    match options.mode {
201        Mode::Mild => {
202            if kind == TokenKind::Whitespace {
203                return;
204            }
205        }
206        Mode::Weak => {
207            if matches!(
208                kind,
209                TokenKind::Whitespace | TokenKind::Comment | TokenKind::BlockComment
210            ) {
211                return;
212            }
213        }
214        Mode::Strict => {} // keep everything (except Ignore, handled above)
215    }
216    let raw_hash = hash_token(kind.discriminant(), value, options.ignore_case);
217    let hash = match normalized_value(&kind, value, options) {
218        Some(placeholder) => hash_token(kind.discriminant(), placeholder, false),
219        None => raw_hash,
220    };
221    tokens.push(DetectionToken {
222        hash,
223        raw_hash,
224        start,
225        end,
226        range: [byte_start, byte_end],
227    });
228}
229
230/// Placeholder that replaces `value` under the active normalization options,
231/// or `None` when the token hashes as-is (issue #998).
232#[inline]
233fn normalized_value(
234    kind: &TokenKind,
235    value: &str,
236    options: &TokenizeOptions,
237) -> Option<&'static str> {
238    match kind {
239        TokenKind::Identifier if options.ignore_identifiers => {
240            if is_common_keyword(value) {
241                None
242            } else {
243                Some("$id")
244            }
245        }
246        TokenKind::Literal if options.ignore_literals => literal_placeholder(value),
247        _ => None,
248    }
249}
250
251/// `$str` for quoted literals (with an optional short alphabetic prefix such
252/// as Python's `r"..."` or C#'s `@"..."`), `$num` for numbers, `None` for
253/// anything else (`true`, `null`, regex literals) which keeps its own hash.
254fn literal_placeholder(value: &str) -> Option<&'static str> {
255    let bytes = value.as_bytes();
256    let first = *bytes.first()?;
257    if matches!(first, b'"' | b'\'' | b'`') {
258        return Some("$str");
259    }
260    if first.is_ascii_digit() {
261        return Some("$num");
262    }
263    if first == b'.' && bytes.get(1).is_some_and(u8::is_ascii_digit) {
264        return Some("$num");
265    }
266    if first.is_ascii_alphabetic() || first == b'@' {
267        let quote_at = bytes
268            .iter()
269            .take(4)
270            .position(|b| matches!(b, b'"' | b'\'' | b'`'));
271        if quote_at.is_some() {
272            return Some("$str");
273        }
274    }
275    None
276}
277
278/// Keywords the generic tokenizer reports as identifiers. Kept verbatim under
279/// `--ignore-identifiers` so control flow still has to match; the list is the
280/// union of common keywords across C-like, Python-like and ML-like languages.
281/// Must stay sorted: looked up by binary search.
282static COMMON_KEYWORDS: &[&str] = &[
283    "abstract",
284    "and",
285    "as",
286    "assert",
287    "async",
288    "await",
289    "begin",
290    "break",
291    "case",
292    "catch",
293    "class",
294    "const",
295    "continue",
296    "def",
297    "default",
298    "defer",
299    "del",
300    "do",
301    "elif",
302    "else",
303    "elsif",
304    "end",
305    "enum",
306    "except",
307    "export",
308    "extends",
309    "extern",
310    "false",
311    "final",
312    "finally",
313    "fn",
314    "for",
315    "foreach",
316    "from",
317    "func",
318    "function",
319    "global",
320    "go",
321    "goto",
322    "if",
323    "impl",
324    "implements",
325    "import",
326    "in",
327    "inline",
328    "instanceof",
329    "interface",
330    "is",
331    "lambda",
332    "let",
333    "loop",
334    "match",
335    "mod",
336    "module",
337    "mut",
338    "namespace",
339    "new",
340    "nil",
341    "none",
342    "not",
343    "null",
344    "or",
345    "override",
346    "package",
347    "pass",
348    "private",
349    "protected",
350    "pub",
351    "public",
352    "raise",
353    "record",
354    "ref",
355    "require",
356    "rescue",
357    "return",
358    "sealed",
359    "select",
360    "self",
361    "sizeof",
362    "static",
363    "struct",
364    "super",
365    "switch",
366    "then",
367    "this",
368    "throw",
369    "throws",
370    "trait",
371    "true",
372    "try",
373    "type",
374    "typedef",
375    "typeof",
376    "undefined",
377    "union",
378    "unless",
379    "unsafe",
380    "until",
381    "use",
382    "using",
383    "var",
384    "virtual",
385    "void",
386    "volatile",
387    "when",
388    "where",
389    "while",
390    "with",
391    "yield",
392];
393
394/// True for words in [`COMMON_KEYWORDS`] (case-sensitive).
395pub fn is_common_keyword(word: &str) -> bool {
396    COMMON_KEYWORDS.binary_search(&word).is_ok()
397}
398
399/// Formats where `@Name` / `@Name(...)` means an annotation or decorator.
400/// Elsewhere (`Ruby`, `Perl`, `T-SQL`, `Razor`, `CSS`) `@` prefixes variables
401/// or directives and must not be dropped.
402pub fn strips_annotations(format: &str) -> bool {
403    matches!(
404        format,
405        "javascript"
406            | "typescript"
407            | "jsx"
408            | "tsx"
409            | "java"
410            | "kotlin"
411            | "scala"
412            | "groovy"
413            | "python"
414            | "dart"
415            | "swift"
416    )
417}
418
419/// Flag `@Name`, `@a.b.Name` and `@Name(...)` token runs so the detection
420/// path drops them (issue #998). Whitespace tokens inside a run are tolerated
421/// only between the name and its argument list. Returns one flag per token,
422/// or an empty vector when nothing matched.
423fn mark_annotations(tokens: &[Token]) -> Vec<bool> {
424    let n = tokens.len();
425    let mut i = 0;
426    let mut flags: Vec<bool> = Vec::new();
427    while i < n {
428        // `@` followed by an identifier — but never by a keyword: Java's
429        // `@interface Name { … }` declares an annotation type rather than
430        // applying one, and the generic tokenizer reports `interface` as an
431        // identifier.
432        let starts_annotation = tokens[i].value == "@"
433            && tokens[i].kind != TokenKind::Ignore
434            && tokens
435                .get(i + 1)
436                .is_some_and(|t| t.kind == TokenKind::Identifier && !is_common_keyword(&t.value));
437        if !starts_annotation {
438            i += 1;
439            continue;
440        }
441        let start = i;
442        let mut j = i + 2;
443        while j + 1 < n && tokens[j].value == "." && tokens[j + 1].kind == TokenKind::Identifier {
444            j += 2;
445        }
446        let mut k = j;
447        while k < n && tokens[k].kind == TokenKind::Whitespace {
448            k += 1;
449        }
450        if k < n && tokens[k].value == "(" {
451            let mut depth = 0usize;
452            j = k;
453            while j < n {
454                match tokens[j].value.as_str() {
455                    "(" => depth += 1,
456                    ")" => {
457                        depth -= 1;
458                        if depth == 0 {
459                            j += 1;
460                            break;
461                        }
462                    }
463                    _ => {}
464                }
465                j += 1;
466            }
467        }
468        if flags.is_empty() {
469            flags = vec![false; n];
470        }
471        for flag in &mut flags[start..j] {
472            *flag = true;
473        }
474        i = j;
475    }
476    flags
477}
478
479/// Tokenize source code in the given format with the given mode.
480/// Returns a Vec<Token>. Never panics on empty input — returns empty Vec.
481///
482/// This is the display/reporter path. For the detection path, use
483/// `tokenize_to_detection`.
484pub fn tokenize(format: &str, source: &str, mode: Mode) -> Vec<Token> {
485    let raw = dispatch_tokenizer(format, source, mode);
486    // Apply mode filter inline — keeps Ignore tokens removed, drops Whitespace in
487    // Mild, drops Whitespace+Comment+BlockComment in Weak, keeps all in Strict.
488    raw.into_iter().filter(|t| keep_token(t, mode)).collect()
489}
490
491fn keep_token(token: &Token, mode: Mode) -> bool {
492    if token.kind == TokenKind::Ignore {
493        return false;
494    }
495    match mode {
496        Mode::Mild => !matches!(token.kind, TokenKind::Whitespace),
497        Mode::Weak => !matches!(
498            token.kind,
499            TokenKind::Whitespace | TokenKind::Comment | TokenKind::BlockComment
500        ),
501        Mode::Strict => true,
502    }
503}
504
505/// Tokenize source code for the detection hot path.
506///
507/// Returns `Vec<DetectionToken>` — tokens filtered and hashed inline at
508/// tokenize time. No per-token heap allocation survives in the output:
509/// the value string is consumed; only the hash, locations, and byte range
510/// are stored.
511///
512/// This replaces the `tokenize` → `apply_mode` → convert-to-hashes pipeline
513/// that existed in `detect.rs`.
514pub fn tokenize_to_detection(
515    format: &str,
516    source: &str,
517    options: &TokenizeOptions,
518) -> Vec<DetectionToken> {
519    // Produce the display tokens first (reuse existing tokenizer code),
520    // then convert to DetectionToken in one pass applying options filters.
521    //
522    // This approach is conservative: it reuses all existing tokenizer logic
523    // without risk of introducing per-tokenizer bugs. The conversion is O(n)
524    // and eliminates the separate filter pass and hash computation that
525    // previously happened inside detect.rs.
526    let raw = if should_strip_types(format, options) {
527        crate::javascript::tokenize_js_stripped(source, format)
528    } else {
529        dispatch_tokenizer(format, source, options.mode)
530    };
531    let annotation = if options.ignore_annotations && strips_annotations(format) {
532        mark_annotations(&raw)
533    } else {
534        Vec::new()
535    };
536    let mut detection: Vec<DetectionToken> = Vec::with_capacity(raw.len());
537    for (i, t) in raw.into_iter().enumerate() {
538        if annotation.get(i).copied().unwrap_or(false) {
539            // Dropped annotation: fold its text into the raw hash of the token
540            // that precedes it. A clone whose matched run contains that token
541            // then classifies as `renamed` when the annotations differ, while
542            // a run that starts after the annotation is unaffected and stays
543            // `exact` — the reported fragment text really is identical there.
544            if let Some(prev) = detection.last_mut() {
545                let h = hash_token(t.kind.discriminant(), &t.value, false);
546                prev.raw_hash = prev.raw_hash.rotate_left(7) ^ h;
547            }
548            continue;
549        }
550        let byte_start = t.start.offset as usize;
551        let byte_end = t.end.offset as usize;
552        push_token(
553            &mut detection,
554            t.kind,
555            &t.value,
556            byte_start,
557            byte_end,
558            t.start,
559            t.end,
560            options,
561        );
562    }
563    detection
564}
565
566fn dispatch_tokenizer(format: &str, source: &str, mode: Mode) -> Vec<Token> {
567    match format {
568        "javascript" | "typescript" | "jsx" | "tsx" => {
569            crate::javascript::tokenize_js(source, format)
570        }
571        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc(source, format, mode),
572        "razor" => crate::razor::tokenize_razor(source, mode),
573        "markdown" | "md" => crate::markdown::tokenize_markdown(source, mode),
574        _ => crate::generic::tokenize_generic(source, format),
575    }
576}
577
578/// Tokenize source code into one or more format-specific detection maps.
579///
580/// For single-format files, returns exactly one `TokenMap` with the same format.
581/// For multi-format files (markdown, SFCs), returns one `TokenMap` per detected
582/// sub-language — e.g. markdown prose + embedded JavaScript + embedded Python.
583///
584/// Each map's tokens carry byte offsets relative to the original source, so
585/// they can be used directly for clone detection within their format group.
586pub fn tokenize_to_detection_maps(
587    format: &str,
588    source: &str,
589    options: &TokenizeOptions,
590) -> Vec<TokenMap> {
591    match format {
592        "markdown" | "md" => crate::markdown::tokenize_markdown_maps(source, options),
593        "vue" | "svelte" | "astro" => crate::sfc::tokenize_sfc_maps(source, format, options),
594        "razor" => crate::razor::tokenize_razor_maps(source, options),
595        _ => {
596            let tokens = tokenize_to_detection(format, source, options);
597            vec![TokenMap {
598                format: format.to_string(),
599                tokens,
600            }]
601        }
602    }
603}
604
605#[cfg(test)]
606mod tests {
607    use super::*;
608
609    #[test]
610    fn mode_from_str_defaults_to_mild() {
611        assert_eq!("unknown".parse::<Mode>().unwrap(), Mode::Mild);
612        assert_eq!("mild".parse::<Mode>().unwrap(), Mode::Mild);
613    }
614
615    #[test]
616    fn mode_from_str_weak() {
617        assert_eq!("weak".parse::<Mode>().unwrap(), Mode::Weak);
618    }
619
620    #[test]
621    fn mode_from_str_strict() {
622        assert_eq!("strict".parse::<Mode>().unwrap(), Mode::Strict);
623    }
624
625    #[test]
626    fn tokenize_to_detection_returns_detection_tokens() {
627        let opts = TokenizeOptions::new(Mode::Mild);
628        let tokens = tokenize_to_detection("javascript", "function hello() { return 42; }", &opts);
629        assert!(
630            !tokens.is_empty(),
631            "must produce DetectionTokens for valid JS"
632        );
633    }
634
635    #[test]
636    fn tokenize_to_detection_mild_excludes_whitespace() {
637        let opts = TokenizeOptions::new(Mode::Mild);
638        // The raw tokenizer produces whitespace tokens; mild mode drops them.
639        // We verify by counting: detection output should have fewer tokens than
640        // a strict-mode tokenize which keeps whitespace.
641        let mild = tokenize_to_detection("javascript", "a b c", &opts);
642        let strict =
643            tokenize_to_detection("javascript", "a b c", &TokenizeOptions::new(Mode::Strict));
644        // Mild must not exceed strict count (whitespace removed).
645        // Note: JS tokenizer doesn't produce Whitespace kind for OXC tokens,
646        // but the contract is that push_token correctly drops them if present.
647        let _ = (mild, strict);
648    }
649
650    #[test]
651    fn push_token_drops_ignore_kind() {
652        let mut tokens = Vec::new();
653        let loc = cpd_core::models::Location {
654            line: 1,
655            column: 0,
656            offset: 0,
657        };
658        let opts = TokenizeOptions::new(Mode::Mild);
659        push_token(
660            &mut tokens,
661            TokenKind::Ignore,
662            "secret",
663            0,
664            6,
665            loc.clone(),
666            loc,
667            &opts,
668        );
669        assert!(tokens.is_empty(), "Ignore-kind tokens must be dropped");
670    }
671
672    #[test]
673    fn push_token_drops_whitespace_in_mild_mode() {
674        let mut tokens = Vec::new();
675        let loc = cpd_core::models::Location {
676            line: 1,
677            column: 0,
678            offset: 0,
679        };
680        let opts = TokenizeOptions::new(Mode::Mild);
681        push_token(
682            &mut tokens,
683            TokenKind::Whitespace,
684            " ",
685            0,
686            1,
687            loc.clone(),
688            loc,
689            &opts,
690        );
691        assert!(tokens.is_empty(), "Whitespace must be dropped in Mild mode");
692    }
693
694    #[test]
695    fn push_token_keeps_whitespace_in_strict_mode() {
696        let mut tokens = Vec::new();
697        let loc = cpd_core::models::Location {
698            line: 1,
699            column: 0,
700            offset: 0,
701        };
702        let opts = TokenizeOptions::new(Mode::Strict);
703        push_token(
704            &mut tokens,
705            TokenKind::Whitespace,
706            " ",
707            0,
708            1,
709            loc.clone(),
710            loc,
711            &opts,
712        );
713        assert_eq!(tokens.len(), 1, "Whitespace must be kept in Strict mode");
714    }
715
716    #[test]
717    fn push_token_drops_comment_in_weak_mode() {
718        let mut tokens = Vec::new();
719        let loc = cpd_core::models::Location {
720            line: 1,
721            column: 0,
722            offset: 0,
723        };
724        let opts = TokenizeOptions::new(Mode::Weak);
725        push_token(
726            &mut tokens,
727            TokenKind::Comment,
728            "// note",
729            0,
730            7,
731            loc.clone(),
732            loc,
733            &opts,
734        );
735        assert!(tokens.is_empty(), "Comment must be dropped in Weak mode");
736    }
737
738    fn det(source: &str, format: &str, opts: &TokenizeOptions) -> Vec<DetectionToken> {
739        tokenize_to_detection(format, source, opts)
740    }
741
742    fn hashes(tokens: &[DetectionToken]) -> Vec<u64> {
743        tokens.iter().map(|t| t.hash).collect()
744    }
745
746    #[test]
747    fn default_options_keep_raw_hash_equal_to_hash() {
748        let opts = TokenizeOptions::new(Mode::Mild);
749        let tokens = det("function a(x) { return x + 1; }", "javascript", &opts);
750        assert!(!tokens.is_empty());
751        assert!(tokens.iter().all(|t| t.raw_hash == t.hash));
752        assert!(!opts.normalizes());
753    }
754
755    #[test]
756    fn ignore_identifiers_matches_renamed_code_and_keeps_keywords() {
757        let mut opts = TokenizeOptions::new(Mode::Mild);
758        opts.ignore_identifiers = true;
759        let a = det("function a(x) { return x + 1; }", "javascript", &opts);
760        let b = det("function b(y) { return y + 1; }", "javascript", &opts);
761        assert_eq!(
762            hashes(&a),
763            hashes(&b),
764            "renamed identifiers must hash alike"
765        );
766        // raw hashes still differ where the names differ
767        assert_ne!(
768            a.iter().map(|t| t.raw_hash).collect::<Vec<_>>(),
769            b.iter().map(|t| t.raw_hash).collect::<Vec<_>>()
770        );
771        // keywords are not folded: `return` vs `throw` must stay distinct
772        let c = det("function a(x) { throw x + 1; }", "javascript", &opts);
773        assert_ne!(hashes(&a), hashes(&c));
774    }
775
776    #[test]
777    fn ignore_identifiers_keeps_common_keywords_in_generic_languages() {
778        let mut opts = TokenizeOptions::new(Mode::Mild);
779        opts.ignore_identifiers = true;
780        let a = det("if x:\n    return y\n", "python", &opts);
781        let b = det("if p:\n    return q\n", "python", &opts);
782        let c = det("while x:\n    return y\n", "python", &opts);
783        assert_eq!(hashes(&a), hashes(&b));
784        assert_ne!(
785            hashes(&a),
786            hashes(&c),
787            "`if` and `while` are keywords, not identifiers"
788        );
789        assert!(is_common_keyword("return"));
790        assert!(!is_common_keyword("total"));
791    }
792
793    #[test]
794    fn ignore_literals_folds_strings_and_numbers_separately() {
795        let mut opts = TokenizeOptions::new(Mode::Mild);
796        opts.ignore_literals = true;
797        let a = det("const a = 10; const b = 'x';", "javascript", &opts);
798        let b = det("const a = 25; const b = \"yy\";", "javascript", &opts);
799        assert_eq!(hashes(&a), hashes(&b));
800        let c = det("const a = 'ten'; const b = 'x';", "javascript", &opts);
801        assert_ne!(hashes(&a), hashes(&c), "a string is not a number");
802        assert_eq!(literal_placeholder("42"), Some("$num"));
803        assert_eq!(literal_placeholder(".5"), Some("$num"));
804        assert_eq!(literal_placeholder("\"s\""), Some("$str"));
805        assert_eq!(literal_placeholder("r'raw'"), Some("$str"));
806        assert_eq!(literal_placeholder("true"), None);
807    }
808
809    #[test]
810    fn ignore_annotations_drops_decorators_in_listed_formats_only() {
811        let mut opts = TokenizeOptions::new(Mode::Mild);
812        opts.ignore_annotations = true;
813        let plain = det("class A { m() { return 1; } }", "typescript", &opts);
814        let decorated = det(
815            "@Component({ selector: 'a' })\nclass A { @Input() m() { return 1; } }",
816            "typescript",
817            &opts,
818        );
819        assert_eq!(hashes(&plain), hashes(&decorated));
820        // the token before an inner annotation carries a salted raw hash
821        assert!(decorated.iter().any(|t| t.raw_hash != t.hash));
822
823        let java_a = det(
824            "class A {\n  @Override\n  int f() { return 1; }\n}",
825            "java",
826            &opts,
827        );
828        let java_b = det(
829            "class A {\n  @Deprecated\n  int f() { return 1; }\n}",
830            "java",
831            &opts,
832        );
833        assert_eq!(hashes(&java_a), hashes(&java_b));
834        assert_ne!(
835            java_a.iter().map(|t| t.raw_hash).collect::<Vec<_>>(),
836            java_b.iter().map(|t| t.raw_hash).collect::<Vec<_>>(),
837            "different annotations must leave different raw hashes"
838        );
839
840        // `@interface` declares an annotation type; it is not an annotation use
841        let decl = "public @interface Marker {\n  String value() default \"\";\n}";
842        let stripped = det(decl, "java", &opts);
843        let untouched = det(decl, "java", &TokenizeOptions::new(Mode::Mild));
844        assert_eq!(hashes(&stripped), hashes(&untouched));
845
846        // Ruby instance variables use `@` and must survive
847        let ruby = det("@count = 1", "ruby", &opts);
848        assert!(strips_annotations("kotlin"));
849        assert!(!strips_annotations("ruby"));
850        assert_eq!(
851            ruby.len(),
852            det("@count = 1", "ruby", &TokenizeOptions::new(Mode::Mild)).len()
853        );
854    }
855
856    #[test]
857    fn push_token_ignore_case_folds_hash() {
858        let mut t1 = Vec::new();
859        let mut t2 = Vec::new();
860        let loc = cpd_core::models::Location {
861            line: 1,
862            column: 0,
863            offset: 0,
864        };
865        let mut opts = TokenizeOptions::new(Mode::Mild);
866        opts.ignore_case = true;
867        push_token(
868            &mut t1,
869            TokenKind::Identifier,
870            "Hello",
871            0,
872            5,
873            loc.clone(),
874            loc.clone(),
875            &opts,
876        );
877        push_token(
878            &mut t2,
879            TokenKind::Identifier,
880            "hello",
881            0,
882            5,
883            loc.clone(),
884            loc,
885            &opts,
886        );
887        assert_eq!(t1[0].hash, t2[0].hash, "ignore_case must fold case in hash");
888    }
889
890    #[test]
891    fn push_token_code_ignore_range_skips_overlapping_token() {
892        // Simulate: source = "foo// cpd-disable"
893        // regex "//\\s*cpd-disable" matches bytes 3..18
894        // Token "foo" is at 0..3 (no overlap -> kept)
895        // Token "// cpd-disable" is at 3..18 (overlaps -> skipped)
896        let mut tokens = Vec::new();
897        let loc = cpd_core::models::Location {
898            line: 1,
899            column: 0,
900            offset: 0,
901        };
902        let mut opts = TokenizeOptions::new(Mode::Mild);
903        // Pre-computed byte ranges from regex match on source text
904        opts.ignore_ranges = vec![[3, 18]];
905        push_token(
906            &mut tokens,
907            TokenKind::Identifier,
908            "foo",
909            0,
910            3,
911            loc.clone(),
912            loc.clone(),
913            &opts,
914        );
915        push_token(
916            &mut tokens,
917            TokenKind::Comment,
918            "// cpd-disable",
919            3,
920            18,
921            loc.clone(),
922            loc,
923            &opts,
924        );
925        assert_eq!(tokens.len(), 1, "only the non-matching token should remain");
926        assert_eq!(tokens[0].range, [0, 3]);
927    }
928
929    #[test]
930    fn push_token_code_ignore_range_no_overlap_keeps_all() {
931        // regex match at bytes 100..120 doesn't overlap tokens at 0..3, 3..6
932        let mut tokens = Vec::new();
933        let loc = cpd_core::models::Location {
934            line: 1,
935            column: 0,
936            offset: 0,
937        };
938        let mut opts = TokenizeOptions::new(Mode::Mild);
939        opts.ignore_ranges = vec![[100, 120]];
940        push_token(
941            &mut tokens,
942            TokenKind::Identifier,
943            "foo",
944            0,
945            3,
946            loc.clone(),
947            loc.clone(),
948            &opts,
949        );
950        push_token(
951            &mut tokens,
952            TokenKind::Identifier,
953            "bar",
954            3,
955            6,
956            loc.clone(),
957            loc,
958            &opts,
959        );
960        assert_eq!(
961            tokens.len(),
962            2,
963            "both tokens should remain when range doesn't overlap"
964        );
965    }
966
967    #[test]
968    fn code_ignore_ranges_computes_from_source_text() {
969        let source = "import foo from 'bar';\nconst x = 1;";
970        let re = regex::Regex::new(r"import\s+\w+\s+from").unwrap();
971        let ranges = code_ignore_ranges(source, &[re]);
972        assert_eq!(ranges.len(), 1, "should find one regex match");
973        // "import foo from" starts at byte 0, ends at byte 15
974        assert_eq!(ranges[0], [0, 15]);
975    }
976
977    #[test]
978    fn code_ignore_ranges_multiple_patterns() {
979        let source = "// MIT License\nfunction foo() {}\n// Copyright";
980        let re1 = regex::Regex::new(r"//\s*MIT\s+License").unwrap();
981        let re2 = regex::Regex::new(r"//\s*Copyright").unwrap();
982        let ranges = code_ignore_ranges(source, &[re1, re2]);
983        assert_eq!(ranges.len(), 2, "should find two regex matches");
984    }
985
986    #[test]
987    fn code_ignore_ranges_empty_regexes() {
988        let source = "function foo() {}";
989        let ranges = code_ignore_ranges(source, &[]);
990        assert!(ranges.is_empty(), "no regexes means no ranges");
991    }
992
993    #[test]
994    fn with_code_ignore_patterns_builds_regexes() {
995        let opts = TokenizeOptions::with_code_ignore_patterns(
996            Mode::Mild,
997            &["function".to_string(), r"//\s*cpd-disable".to_string()],
998        );
999        assert_eq!(opts.code_ignore_regexes.len(), 2);
1000        assert!(opts.code_ignore_regexes[0].is_match("function"));
1001        assert!(opts.code_ignore_regexes[1].is_match("// cpd-disable"));
1002        assert!(!opts.code_ignore_regexes[1].is_match("function"));
1003    }
1004
1005    #[test]
1006    fn tokenize_to_detection_with_code_ignore_ranges_skips_imports() {
1007        let source = "import * from 'lodash';\nconst x = 1;";
1008        let regexes = vec![regex::Regex::new(r"import\s+\*\s+from").unwrap()];
1009        let ranges = code_ignore_ranges(source, &regexes);
1010        assert!(!ranges.is_empty(), "should find regex match in source");
1011
1012        let mut opts = TokenizeOptions::new(Mode::Mild);
1013        opts.ignore_ranges = ranges;
1014        let tokens = tokenize_to_detection("javascript", source, &opts);
1015
1016        // Tokens whose byte ranges overlap the import match should be skipped.
1017        // "import" (0-6), "*" (7-8), "from" (9-13) should all be in range,
1018        // but "const" (24-29) and "x" (30-31) etc should remain.
1019        let has_const = tokens.iter().any(|t| {
1020            // Check that tokens after the import line are still present
1021            t.range[0] >= 24
1022        });
1023        assert!(
1024            has_const,
1025            "tokens after the import line should still be present"
1026        );
1027    }
1028
1029    #[test]
1030    fn code_ignore_ranges_multi_token_match() {
1031        // The key test: regex "import.*from" matches multi-token source text
1032        // like "import * from 'module-name'" — not just a single token value.
1033        let source = "import * from 'lodash';\nconst result = 42;";
1034        let re = regex::Regex::new(r"import\s+.*?\s+from").unwrap();
1035        let ranges = code_ignore_ranges(source, &[re]);
1036        assert_eq!(
1037            ranges.len(),
1038            1,
1039            "should find one regex match spanning import statement"
1040        );
1041        assert!(ranges[0][0] == 0, "match should start at beginning");
1042        assert!(ranges[0][1] > 0, "match should have non-zero end");
1043    }
1044}