Skip to main content

cpd_core/
models.rs

1use serde::{Deserialize, Serialize};
2use std::collections::HashMap;
3
4#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
5#[serde(rename_all = "snake_case")]
6pub enum TokenKind {
7    Keyword,
8    Identifier,
9    Literal,
10    Operator,
11    Punctuation,
12    Comment,
13    BlockComment,
14    Whitespace,
15    Ignore,
16    Other,
17}
18
19impl TokenKind {
20    /// Return a stable byte discriminant for use in token hashing.
21    pub fn discriminant(&self) -> u8 {
22        match self {
23            Self::Keyword => 1,
24            Self::Identifier => 2,
25            Self::Literal => 3,
26            Self::Operator => 4,
27            Self::Punctuation => 5,
28            Self::Comment => 6,
29            Self::BlockComment => 7,
30            Self::Whitespace => 8,
31            Self::Ignore => 9,
32            Self::Other => 10,
33        }
34    }
35}
36
37#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
38pub struct Location {
39    pub line: u32,
40    pub column: u32,
41    pub offset: u32,
42}
43
44#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
45pub struct Token {
46    pub kind: TokenKind,
47    pub value: String,
48    pub start: Location,
49    pub end: Location,
50}
51
52#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
53pub struct BlameEntry {
54    pub commit_sha: String,
55    pub author: String,
56    pub timestamp: i64,
57}
58
59/// How the two fragments of a clone relate at the token level (issue #998).
60#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
61#[serde(rename_all = "lowercase")]
62pub enum CloneKind {
63    /// The fragments are token-for-token identical.
64    #[default]
65    Exact,
66    /// The fragments match only after identifier, literal or annotation
67    /// normalization (`--ignore-identifiers`, `--ignore-literals`,
68    /// `--ignore-annotations`): a Type-2 clone.
69    Renamed,
70    /// A Type-3 near-miss clone: two or more matches of the same file pair
71    /// merged across a gap of unmatched lines (`--max-gap-lines`), or a pair
72    /// of structurally similar functions (`--similarity`). `similar` takes
73    /// precedence over `renamed`: a merge of renamed halves is `similar`.
74    Similar,
75    /// A Type-4 clone (`--semantic`, experimental): two functions that do the
76    /// same thing written differently, possibly in different languages, found
77    /// by comparing embeddings of their code. `similarity` is the cosine
78    /// similarity of the two embeddings.
79    Semantic,
80}
81
82impl CloneKind {
83    pub fn is_renamed(self) -> bool {
84        matches!(self, CloneKind::Renamed)
85    }
86
87    pub fn is_similar(self) -> bool {
88        matches!(self, CloneKind::Similar)
89    }
90
91    pub fn is_semantic(self) -> bool {
92        matches!(self, CloneKind::Semantic)
93    }
94
95    pub fn as_str(self) -> &'static str {
96        match self {
97            CloneKind::Exact => "exact",
98            CloneKind::Renamed => "renamed",
99            CloneKind::Similar => "similar",
100            CloneKind::Semantic => "semantic",
101        }
102    }
103}
104
105#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
106pub struct Fragment {
107    pub source_id: String,
108    #[serde(default, skip_serializing_if = "Option::is_none")]
109    pub source_root: Option<String>,
110    pub start: Location,
111    pub end: Location,
112    pub range: [u32; 2],
113    pub blame: Option<BlameEntry>,
114}
115
116/// How a `similar` clone was produced (issue #999).
117#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
118#[serde(rename_all = "lowercase")]
119pub enum SimilarityMethod {
120    /// Exact matches merged across a gap of unmatched lines
121    /// (`--max-gap-lines`); `similarity` is matched tokens over the span.
122    Gap,
123    /// Whole functions compared by syntax-tree structure (`--similarity`);
124    /// `similarity` is the weighted Jaccard index of node-type shingles.
125    Ast,
126}
127
128impl SimilarityMethod {
129    pub fn as_str(self) -> &'static str {
130        match self {
131            SimilarityMethod::Gap => "gap",
132            SimilarityMethod::Ast => "ast",
133        }
134    }
135}
136
137/// One `--kind` value: a clone kind, or one of the two mechanisms that find
138/// `similar` clones.
139#[derive(Debug, Clone, Copy, PartialEq, Eq)]
140pub enum KindFilter {
141    Exact,
142    Renamed,
143    /// Every `similar` clone, whichever mechanism found it.
144    Similar,
145    /// `similar` clones merged across a gap (`--max-gap-lines`).
146    Gap,
147    /// `similar` function pairs compared by syntax tree (`--similarity`).
148    Ast,
149    /// Type-4 function pairs found by comparing embeddings (`--semantic`).
150    Semantic,
151}
152
153impl KindFilter {
154    pub const NAMES: &'static str = "exact, renamed, similar, gap, ast, semantic";
155
156    pub fn as_str(self) -> &'static str {
157        match self {
158            KindFilter::Exact => "exact",
159            KindFilter::Renamed => "renamed",
160            KindFilter::Similar => "similar",
161            KindFilter::Gap => "gap",
162            KindFilter::Ast => "ast",
163            KindFilter::Semantic => "semantic",
164        }
165    }
166
167    pub fn matches(self, clone: &CpdClone) -> bool {
168        match self {
169            KindFilter::Exact => clone.kind == CloneKind::Exact,
170            KindFilter::Renamed => clone.kind == CloneKind::Renamed,
171            KindFilter::Similar => clone.kind == CloneKind::Similar,
172            KindFilter::Gap => clone.similarity_method == Some(SimilarityMethod::Gap),
173            KindFilter::Ast => clone.similarity_method == Some(SimilarityMethod::Ast),
174            KindFilter::Semantic => clone.kind == CloneKind::Semantic,
175        }
176    }
177}
178
179impl std::str::FromStr for KindFilter {
180    type Err = String;
181
182    fn from_str(s: &str) -> Result<Self, Self::Err> {
183        match s.trim().to_ascii_lowercase().as_str() {
184            "exact" => Ok(KindFilter::Exact),
185            "renamed" => Ok(KindFilter::Renamed),
186            "similar" => Ok(KindFilter::Similar),
187            "gap" => Ok(KindFilter::Gap),
188            "ast" => Ok(KindFilter::Ast),
189            "semantic" => Ok(KindFilter::Semantic),
190            other => Err(format!(
191                "unknown clone kind '{other}': must be one of: {}",
192                KindFilter::NAMES
193            )),
194        }
195    }
196}
197
198#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
199pub struct CpdClone {
200    pub format: String,
201    pub fragment_a: Fragment,
202    pub fragment_b: Fragment,
203    pub token_count: u32,
204    /// True when the clone is absent from the configured baseline (issue #944).
205    /// Always false when no baseline is in use.
206    #[serde(default)]
207    pub is_new: bool,
208    /// `exact` when the raw tokens of both fragments are identical, `renamed`
209    /// when they match only after normalization (issue #998). Always `exact`
210    /// when no normalization option is on.
211    #[serde(default)]
212    pub kind: CloneKind,
213    /// For `similar` clones: matched tokens divided by the tokens of the
214    /// longer merged span, in `(0, 1)`; for `semantic` clones: the cosine
215    /// similarity of the two functions' embeddings. `None` for exact and
216    /// renamed clones.
217    #[serde(default, skip_serializing_if = "Option::is_none")]
218    pub similarity: Option<f32>,
219    /// Which mechanism produced a `similar` clone; the two scores are not
220    /// on the same scale, so reporters show it next to the value.
221    #[serde(default, rename = "method", skip_serializing_if = "Option::is_none")]
222    pub similarity_method: Option<SimilarityMethod>,
223    /// Lines inside each fragment's span that are not duplicated code, for
224    /// `fragment_a` and `fragment_b` in that order. Two things land here: the
225    /// lines a `--max-gap-lines` merge left unmatched between its halves, and,
226    /// for an embedded block, the host language's lines lying between two
227    /// blocks of the same fragment (issue #1090). Statistics subtract them
228    /// from the fragment's span.
229    #[serde(skip)]
230    pub unmatched_lines: [u32; 2],
231}
232
233impl Location {
234    pub fn new(line: u32, column: u32, offset: u32) -> Self {
235        Self {
236            line,
237            column,
238            offset,
239        }
240    }
241}
242
243impl Fragment {
244    /// A fragment with no scan root and no blame data, the shape detection
245    /// produces before enrichment.
246    pub fn new(
247        source_id: impl Into<String>,
248        start: Location,
249        end: Location,
250        range: [u32; 2],
251    ) -> Self {
252        Self {
253            source_id: source_id.into(),
254            source_root: None,
255            start,
256            end,
257            range,
258            blame: None,
259        }
260    }
261
262    pub fn with_blame(mut self, blame: BlameEntry) -> Self {
263        self.blame = Some(blame);
264        self
265    }
266}
267
268impl CpdClone {
269    /// Duplicated lines this clone adds to the statistics: the matched lines
270    /// of its primary fragment.
271    pub fn matched_lines(&self) -> u64 {
272        self.fragment_lines(0)
273    }
274
275    /// Matched lines of one fragment — 0 is A, 1 is B. Per-file summaries need
276    /// both, and they must not drift apart.
277    ///
278    /// The span is inclusive, so a clone of lines 10 through 19 is ten lines,
279    /// the same count the reporters print next to it. Whatever the span covers
280    /// but does not duplicate — gap-merge lines, host-language lines between
281    /// two embedded blocks — is already in `unmatched_lines`.
282    pub fn fragment_lines(&self, index: usize) -> u64 {
283        let fragment = if index == 0 {
284            &self.fragment_a
285        } else {
286            &self.fragment_b
287        };
288        let span = fragment.end.line.saturating_sub(fragment.start.line) + 1;
289        span.saturating_sub(self.unmatched_lines[index]) as u64
290    }
291
292    /// An exact clone with no baseline, similarity or gap metadata.
293    pub fn exact(
294        format: impl Into<String>,
295        fragment_a: Fragment,
296        fragment_b: Fragment,
297        token_count: u32,
298    ) -> Self {
299        Self {
300            format: format.into(),
301            fragment_a,
302            fragment_b,
303            token_count,
304            is_new: false,
305            kind: CloneKind::default(),
306            similarity: None,
307            similarity_method: None,
308            unmatched_lines: [0, 0],
309        }
310    }
311    /// The order clone lists are reported in: by the first fragment's file
312    /// and line, then the second's.
313    pub fn position_key(&self) -> (&str, u32, &str, u32) {
314        (
315            &self.fragment_a.source_id,
316            self.fragment_a.start.line,
317            &self.fragment_b.source_id,
318            self.fragment_b.start.line,
319        )
320    }
321
322    /// `similarity` rounded to three decimals as an f64, the form reporters
323    /// print (an f32 widened to JSON would print as `0.8510638475418091`).
324    pub fn similarity_rounded(&self) -> Option<f64> {
325        self.similarity
326            .map(|s| (f64::from(s) * 1000.0).round() / 1000.0)
327    }
328}
329
330/// Internal detection unit — no heap allocation per token.
331///
332/// Produced by the tokenizer's detection path at tokenize time.
333/// `Token` is used for display, blame, and reporter output;
334/// `DetectionToken` is used only during the clone detection hot path.
335/// The token's value string is not stored — only its pre-computed hash.
336#[derive(Debug, Clone, PartialEq, Eq)]
337pub struct DetectionToken {
338    /// Pre-computed hash of (kind, value) — detection never re-hashes.
339    /// When a normalization option rewrote the value (`$id`, `$str`, `$num`)
340    /// this is the hash of the placeholder.
341    pub hash: u64,
342    /// Hash of the original (kind, value). Equal to `hash` unless a
343    /// normalization option applied; lets detection tell exact clones from
344    /// renamed ones without re-tokenizing.
345    pub raw_hash: u64,
346    pub start: Location,
347    pub end: Location,
348    /// Byte range in the source content: `[start_byte, end_byte]`.
349    pub range: [usize; 2],
350}
351
352/// A source file with pre-tokenized tokens, ready for clone detection.
353#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
354pub struct SourceFile {
355    pub id: String,
356    pub format: String,
357    pub tokens: Vec<Token>,
358    /// File size in bytes. Zero for synthetic sub-format sources
359    /// (embedded code blocks) whose bytes are counted by the parent file.
360    #[serde(default)]
361    pub bytes: u64,
362}
363
364impl SourceFile {
365    /// True for a synthetic sub-format source: one embedded language inside a
366    /// host file, stored under `<path>:<format>` (issue #1090). Its tokens keep
367    /// the host file's line numbers, so the lines between two blocks belong to
368    /// the host and not to this format.
369    pub fn is_embedded(&self) -> bool {
370        self.id
371            .strip_suffix(self.format.as_str())
372            .and_then(|rest| rest.strip_suffix(':'))
373            .is_some_and(|path| !path.is_empty())
374    }
375
376    /// Lines of this source that carry code of its own format. For an ordinary
377    /// file that is the last line holding a token — near enough to the file's
378    /// length, and what jscpd has always counted. For an embedded block it is
379    /// only the lines the blocks themselves occupy.
380    pub fn line_count(&self) -> u64 {
381        if self.is_embedded() {
382            covered_lines(self.tokens.iter().map(|t| (t.start.line, t.end.line))) as u64
383        } else {
384            self.tokens.iter().map(|t| t.start.line).max().unwrap_or(0) as u64
385        }
386    }
387}
388
389/// How many distinct lines a run of tokens sits on.
390///
391/// The tokens come in source order and one token may cover several lines, so
392/// this sweeps once and never counts a line twice. For a contiguous run the
393/// answer is the line span; for an embedded block it is much smaller, because
394/// the host language's lines between two blocks hold no token of this source.
395pub fn covered_lines(spans: impl IntoIterator<Item = (u32, u32)>) -> u32 {
396    let mut covered = 0;
397    // Lines are 1-based, so 0 reads as "nothing counted yet".
398    let mut last = 0;
399    for (start, end) in spans {
400        let from = if start > last { start } else { last + 1 };
401        if end >= from {
402            covered += end - from + 1;
403            last = end;
404        }
405    }
406    covered
407}
408
409/// Per-format or total statistics row.
410#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
411#[serde(rename_all = "camelCase")]
412pub struct StatRow {
413    pub lines: u64,
414    pub tokens: u64,
415    pub sources: u64,
416    pub clones: u64,
417    pub duplicated_lines: u64,
418    pub duplicated_tokens: u64,
419    pub percentage: f64,
420    pub percentage_tokens: f64,
421    #[serde(default)]
422    pub new_duplicated_lines: u64,
423    #[serde(default)]
424    pub new_clones: u64,
425}
426
427impl Default for StatRow {
428    fn default() -> Self {
429        Self {
430            lines: 0,
431            tokens: 0,
432            sources: 0,
433            clones: 0,
434            duplicated_lines: 0,
435            duplicated_tokens: 0,
436            percentage: 0.0,
437            percentage_tokens: 0.0,
438            new_duplicated_lines: 0,
439            new_clones: 0,
440        }
441    }
442}
443
444/// Aggregated detection statistics.
445#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
446#[serde(rename_all = "camelCase")]
447pub struct Statistics {
448    pub total: StatRow,
449    pub formats: HashMap<String, StatRow>,
450    pub detection_date: String,
451}
452
453#[cfg(test)]
454mod tests {
455    use super::*;
456    use serde_json;
457
458    #[test]
459    fn statistics_default_total_is_zero() {
460        let stats = Statistics {
461            total: StatRow::default(),
462            formats: HashMap::new(),
463            detection_date: "2026-01-01T00:00:00Z".to_string(),
464        };
465        assert_eq!(stats.total.clones, 0);
466    }
467
468    #[test]
469    fn token_serializes_and_deserializes() {
470        let token = Token {
471            kind: TokenKind::Keyword,
472            value: "function".to_string(),
473            start: Location {
474                line: 1,
475                column: 0,
476                offset: 0,
477            },
478            end: Location {
479                line: 1,
480                column: 8,
481                offset: 8,
482            },
483        };
484        let json = serde_json::to_string(&token).unwrap();
485        let back: Token = serde_json::from_str(&json).unwrap();
486        assert_eq!(token, back);
487    }
488
489    #[test]
490    fn cpd_clone_serializes_with_blame() {
491        let loc = Location {
492            line: 1,
493            column: 0,
494            offset: 0,
495        };
496        let blame = BlameEntry {
497            commit_sha: "abc123".to_string(),
498            author: "Alice".to_string(),
499            timestamp: 1700000000,
500        };
501        let frag = Fragment::new("a.js", loc.clone(), loc, [0, 10]).with_blame(blame);
502        let clone = CpdClone::exact("javascript", frag.clone(), frag, 50);
503        let json = serde_json::to_string(&clone).unwrap();
504        assert!(json.contains("abc123"));
505        assert!(json.contains("fragment_a"));
506    }
507
508    #[test]
509    fn fragment_blame_none_serializes_as_null() {
510        let loc = Location {
511            line: 1,
512            column: 0,
513            offset: 0,
514        };
515        let frag = Fragment {
516            source_id: "b.js".to_string(),
517            source_root: None,
518            start: loc.clone(),
519            end: loc.clone(),
520            range: [0, 5],
521            blame: None,
522        };
523        let json = serde_json::to_string(&frag).unwrap();
524        assert!(json.contains("\"blame\":null"));
525    }
526
527    #[test]
528    fn covered_lines_counts_a_contiguous_run_once() {
529        assert_eq!(covered_lines([(1, 1), (1, 1), (2, 2), (3, 3)]), 3);
530        assert_eq!(covered_lines(std::iter::empty()), 0);
531    }
532
533    #[test]
534    fn covered_lines_skips_the_gap_between_two_blocks() {
535        // Lines 17-26 and 43-52: twenty lines of code, whatever sits between.
536        let block = |from: u32, to: u32| (from..=to).map(|l| (l, l));
537        assert_eq!(covered_lines(block(17, 26).chain(block(43, 52))), 20);
538    }
539
540    #[test]
541    fn covered_lines_handles_a_token_spanning_several_lines() {
542        // A template literal running from line 4 to line 9, then code after it.
543        assert_eq!(covered_lines([(4, 9), (9, 9), (10, 10)]), 7);
544    }
545
546    #[test]
547    fn an_embedded_source_is_recognised_by_its_id() {
548        let source = |id: &str, format: &str| SourceFile {
549            id: id.to_string(),
550            format: format.to_string(),
551            tokens: vec![],
552            bytes: 0,
553        };
554        assert!(source("guide.md:typescript", "typescript").is_embedded());
555        assert!(!source("app.ts", "typescript").is_embedded());
556        // The host's own map is not embedded in anything.
557        assert!(!source("guide.md", "markdown").is_embedded());
558        // A file whose whole name is the format is still a file.
559        assert!(!source("typescript", "typescript").is_embedded());
560    }
561}