Skip to main content

cpd_core/
models.rs

1use serde::{Deserialize, Serialize};
2use std::collections::HashMap;
3
4#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
5#[serde(rename_all = "snake_case")]
6pub enum TokenKind {
7    Keyword,
8    Identifier,
9    Literal,
10    Operator,
11    Punctuation,
12    Comment,
13    BlockComment,
14    Whitespace,
15    Ignore,
16    Other,
17}
18
19impl TokenKind {
20    /// Return a stable byte discriminant for use in token hashing.
21    pub fn discriminant(&self) -> u8 {
22        match self {
23            Self::Keyword => 1,
24            Self::Identifier => 2,
25            Self::Literal => 3,
26            Self::Operator => 4,
27            Self::Punctuation => 5,
28            Self::Comment => 6,
29            Self::BlockComment => 7,
30            Self::Whitespace => 8,
31            Self::Ignore => 9,
32            Self::Other => 10,
33        }
34    }
35}
36
37#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
38pub struct Location {
39    pub line: u32,
40    pub column: u32,
41    pub offset: u32,
42}
43
44#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
45pub struct Token {
46    pub kind: TokenKind,
47    pub value: String,
48    pub start: Location,
49    pub end: Location,
50}
51
52#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
53pub struct BlameEntry {
54    pub commit_sha: String,
55    pub author: String,
56    pub timestamp: i64,
57}
58
59/// How the two fragments of a clone relate at the token level (issue #998).
60#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
61#[serde(rename_all = "lowercase")]
62pub enum CloneKind {
63    /// The fragments are token-for-token identical.
64    #[default]
65    Exact,
66    /// The fragments match only after identifier, literal or annotation
67    /// normalization (`--ignore-identifiers`, `--ignore-literals`,
68    /// `--ignore-annotations`): a Type-2 clone.
69    Renamed,
70    /// A Type-3 near-miss clone: two or more matches of the same file pair
71    /// merged across a gap of unmatched lines (`--max-gap-lines`), or a pair
72    /// of structurally similar functions (`--similarity`). `similar` takes
73    /// precedence over `renamed`: a merge of renamed halves is `similar`.
74    Similar,
75}
76
77impl CloneKind {
78    pub fn is_renamed(self) -> bool {
79        matches!(self, CloneKind::Renamed)
80    }
81
82    pub fn is_similar(self) -> bool {
83        matches!(self, CloneKind::Similar)
84    }
85
86    pub fn as_str(self) -> &'static str {
87        match self {
88            CloneKind::Exact => "exact",
89            CloneKind::Renamed => "renamed",
90            CloneKind::Similar => "similar",
91        }
92    }
93}
94
95#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
96pub struct Fragment {
97    pub source_id: String,
98    #[serde(default, skip_serializing_if = "Option::is_none")]
99    pub source_root: Option<String>,
100    pub start: Location,
101    pub end: Location,
102    pub range: [u32; 2],
103    pub blame: Option<BlameEntry>,
104}
105
106/// How a `similar` clone was produced (issue #999).
107#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
108#[serde(rename_all = "lowercase")]
109pub enum SimilarityMethod {
110    /// Exact matches merged across a gap of unmatched lines
111    /// (`--max-gap-lines`); `similarity` is matched tokens over the span.
112    Gap,
113    /// Whole functions compared by syntax-tree structure (`--similarity`);
114    /// `similarity` is the weighted Jaccard index of node-type shingles.
115    Ast,
116}
117
118impl SimilarityMethod {
119    pub fn as_str(self) -> &'static str {
120        match self {
121            SimilarityMethod::Gap => "gap",
122            SimilarityMethod::Ast => "ast",
123        }
124    }
125}
126
127/// One `--kind` value: a clone kind, or one of the two mechanisms that find
128/// `similar` clones.
129#[derive(Debug, Clone, Copy, PartialEq, Eq)]
130pub enum KindFilter {
131    Exact,
132    Renamed,
133    /// Every `similar` clone, whichever mechanism found it.
134    Similar,
135    /// `similar` clones merged across a gap (`--max-gap-lines`).
136    Gap,
137    /// `similar` function pairs compared by syntax tree (`--similarity`).
138    Ast,
139}
140
141impl KindFilter {
142    pub const NAMES: &'static str = "exact, renamed, similar, gap, ast";
143
144    pub fn as_str(self) -> &'static str {
145        match self {
146            KindFilter::Exact => "exact",
147            KindFilter::Renamed => "renamed",
148            KindFilter::Similar => "similar",
149            KindFilter::Gap => "gap",
150            KindFilter::Ast => "ast",
151        }
152    }
153
154    pub fn matches(self, clone: &CpdClone) -> bool {
155        match self {
156            KindFilter::Exact => clone.kind == CloneKind::Exact,
157            KindFilter::Renamed => clone.kind == CloneKind::Renamed,
158            KindFilter::Similar => clone.kind == CloneKind::Similar,
159            KindFilter::Gap => clone.similarity_method == Some(SimilarityMethod::Gap),
160            KindFilter::Ast => clone.similarity_method == Some(SimilarityMethod::Ast),
161        }
162    }
163}
164
165impl std::str::FromStr for KindFilter {
166    type Err = String;
167
168    fn from_str(s: &str) -> Result<Self, Self::Err> {
169        match s.trim().to_ascii_lowercase().as_str() {
170            "exact" => Ok(KindFilter::Exact),
171            "renamed" => Ok(KindFilter::Renamed),
172            "similar" => Ok(KindFilter::Similar),
173            "gap" => Ok(KindFilter::Gap),
174            "ast" => Ok(KindFilter::Ast),
175            other => Err(format!(
176                "unknown clone kind '{other}': must be one of: {}",
177                KindFilter::NAMES
178            )),
179        }
180    }
181}
182
183#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
184pub struct CpdClone {
185    pub format: String,
186    pub fragment_a: Fragment,
187    pub fragment_b: Fragment,
188    pub token_count: u32,
189    /// True when the clone is absent from the configured baseline (issue #944).
190    /// Always false when no baseline is in use.
191    #[serde(default)]
192    pub is_new: bool,
193    /// `exact` when the raw tokens of both fragments are identical, `renamed`
194    /// when they match only after normalization (issue #998). Always `exact`
195    /// when no normalization option is on.
196    #[serde(default)]
197    pub kind: CloneKind,
198    /// For `similar` clones: matched tokens divided by the tokens of the
199    /// longer merged span, in `(0, 1)`. `None` for exact and renamed clones.
200    #[serde(default, skip_serializing_if = "Option::is_none")]
201    pub similarity: Option<f32>,
202    /// Which mechanism produced a `similar` clone; the two scores are not
203    /// on the same scale, so reporters show it next to the value.
204    #[serde(default, rename = "method", skip_serializing_if = "Option::is_none")]
205    pub similarity_method: Option<SimilarityMethod>,
206    /// Lines inside each fragment's span that are not duplicated code, for
207    /// `fragment_a` and `fragment_b` in that order. Two things land here: the
208    /// lines a `--max-gap-lines` merge left unmatched between its halves, and,
209    /// for an embedded block, the host language's lines lying between two
210    /// blocks of the same fragment (issue #1090). Statistics subtract them
211    /// from the fragment's span.
212    #[serde(skip)]
213    pub unmatched_lines: [u32; 2],
214}
215
216impl Location {
217    pub fn new(line: u32, column: u32, offset: u32) -> Self {
218        Self {
219            line,
220            column,
221            offset,
222        }
223    }
224}
225
226impl Fragment {
227    /// A fragment with no scan root and no blame data, the shape detection
228    /// produces before enrichment.
229    pub fn new(
230        source_id: impl Into<String>,
231        start: Location,
232        end: Location,
233        range: [u32; 2],
234    ) -> Self {
235        Self {
236            source_id: source_id.into(),
237            source_root: None,
238            start,
239            end,
240            range,
241            blame: None,
242        }
243    }
244
245    pub fn with_blame(mut self, blame: BlameEntry) -> Self {
246        self.blame = Some(blame);
247        self
248    }
249}
250
251impl CpdClone {
252    /// Duplicated lines this clone adds to the statistics: the matched lines
253    /// of its primary fragment.
254    pub fn matched_lines(&self) -> u64 {
255        self.fragment_lines(0)
256    }
257
258    /// Matched lines of one fragment — 0 is A, 1 is B. Per-file summaries need
259    /// both, and they must not drift apart.
260    ///
261    /// The span is inclusive, so a clone of lines 10 through 19 is ten lines,
262    /// the same count the reporters print next to it. Whatever the span covers
263    /// but does not duplicate — gap-merge lines, host-language lines between
264    /// two embedded blocks — is already in `unmatched_lines`.
265    pub fn fragment_lines(&self, index: usize) -> u64 {
266        let fragment = if index == 0 {
267            &self.fragment_a
268        } else {
269            &self.fragment_b
270        };
271        let span = fragment.end.line.saturating_sub(fragment.start.line) + 1;
272        span.saturating_sub(self.unmatched_lines[index]) as u64
273    }
274
275    /// An exact clone with no baseline, similarity or gap metadata.
276    pub fn exact(
277        format: impl Into<String>,
278        fragment_a: Fragment,
279        fragment_b: Fragment,
280        token_count: u32,
281    ) -> Self {
282        Self {
283            format: format.into(),
284            fragment_a,
285            fragment_b,
286            token_count,
287            is_new: false,
288            kind: CloneKind::default(),
289            similarity: None,
290            similarity_method: None,
291            unmatched_lines: [0, 0],
292        }
293    }
294    /// `similarity` rounded to three decimals as an f64, the form reporters
295    /// print (an f32 widened to JSON would print as `0.8510638475418091`).
296    pub fn similarity_rounded(&self) -> Option<f64> {
297        self.similarity
298            .map(|s| (f64::from(s) * 1000.0).round() / 1000.0)
299    }
300}
301
302/// Internal detection unit — no heap allocation per token.
303///
304/// Produced by the tokenizer's detection path at tokenize time.
305/// `Token` is used for display, blame, and reporter output;
306/// `DetectionToken` is used only during the clone detection hot path.
307/// The token's value string is not stored — only its pre-computed hash.
308#[derive(Debug, Clone, PartialEq, Eq)]
309pub struct DetectionToken {
310    /// Pre-computed hash of (kind, value) — detection never re-hashes.
311    /// When a normalization option rewrote the value (`$id`, `$str`, `$num`)
312    /// this is the hash of the placeholder.
313    pub hash: u64,
314    /// Hash of the original (kind, value). Equal to `hash` unless a
315    /// normalization option applied; lets detection tell exact clones from
316    /// renamed ones without re-tokenizing.
317    pub raw_hash: u64,
318    pub start: Location,
319    pub end: Location,
320    /// Byte range in the source content: `[start_byte, end_byte]`.
321    pub range: [usize; 2],
322}
323
324/// A source file with pre-tokenized tokens, ready for clone detection.
325#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
326pub struct SourceFile {
327    pub id: String,
328    pub format: String,
329    pub tokens: Vec<Token>,
330    /// File size in bytes. Zero for synthetic sub-format sources
331    /// (embedded code blocks) whose bytes are counted by the parent file.
332    #[serde(default)]
333    pub bytes: u64,
334}
335
336impl SourceFile {
337    /// True for a synthetic sub-format source: one embedded language inside a
338    /// host file, stored under `<path>:<format>` (issue #1090). Its tokens keep
339    /// the host file's line numbers, so the lines between two blocks belong to
340    /// the host and not to this format.
341    pub fn is_embedded(&self) -> bool {
342        self.id
343            .strip_suffix(self.format.as_str())
344            .and_then(|rest| rest.strip_suffix(':'))
345            .is_some_and(|path| !path.is_empty())
346    }
347
348    /// Lines of this source that carry code of its own format. For an ordinary
349    /// file that is the last line holding a token — near enough to the file's
350    /// length, and what jscpd has always counted. For an embedded block it is
351    /// only the lines the blocks themselves occupy.
352    pub fn line_count(&self) -> u64 {
353        if self.is_embedded() {
354            covered_lines(self.tokens.iter().map(|t| (t.start.line, t.end.line))) as u64
355        } else {
356            self.tokens.iter().map(|t| t.start.line).max().unwrap_or(0) as u64
357        }
358    }
359}
360
361/// How many distinct lines a run of tokens sits on.
362///
363/// The tokens come in source order and one token may cover several lines, so
364/// this sweeps once and never counts a line twice. For a contiguous run the
365/// answer is the line span; for an embedded block it is much smaller, because
366/// the host language's lines between two blocks hold no token of this source.
367pub fn covered_lines(spans: impl IntoIterator<Item = (u32, u32)>) -> u32 {
368    let mut covered = 0;
369    // Lines are 1-based, so 0 reads as "nothing counted yet".
370    let mut last = 0;
371    for (start, end) in spans {
372        let from = if start > last { start } else { last + 1 };
373        if end >= from {
374            covered += end - from + 1;
375            last = end;
376        }
377    }
378    covered
379}
380
381/// Per-format or total statistics row.
382#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
383#[serde(rename_all = "camelCase")]
384pub struct StatRow {
385    pub lines: u64,
386    pub tokens: u64,
387    pub sources: u64,
388    pub clones: u64,
389    pub duplicated_lines: u64,
390    pub duplicated_tokens: u64,
391    pub percentage: f64,
392    pub percentage_tokens: f64,
393    #[serde(default)]
394    pub new_duplicated_lines: u64,
395    #[serde(default)]
396    pub new_clones: u64,
397}
398
399impl Default for StatRow {
400    fn default() -> Self {
401        Self {
402            lines: 0,
403            tokens: 0,
404            sources: 0,
405            clones: 0,
406            duplicated_lines: 0,
407            duplicated_tokens: 0,
408            percentage: 0.0,
409            percentage_tokens: 0.0,
410            new_duplicated_lines: 0,
411            new_clones: 0,
412        }
413    }
414}
415
416/// Aggregated detection statistics.
417#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
418#[serde(rename_all = "camelCase")]
419pub struct Statistics {
420    pub total: StatRow,
421    pub formats: HashMap<String, StatRow>,
422    pub detection_date: String,
423}
424
425#[cfg(test)]
426mod tests {
427    use super::*;
428    use serde_json;
429
430    #[test]
431    fn statistics_default_total_is_zero() {
432        let stats = Statistics {
433            total: StatRow::default(),
434            formats: HashMap::new(),
435            detection_date: "2026-01-01T00:00:00Z".to_string(),
436        };
437        assert_eq!(stats.total.clones, 0);
438    }
439
440    #[test]
441    fn token_serializes_and_deserializes() {
442        let token = Token {
443            kind: TokenKind::Keyword,
444            value: "function".to_string(),
445            start: Location {
446                line: 1,
447                column: 0,
448                offset: 0,
449            },
450            end: Location {
451                line: 1,
452                column: 8,
453                offset: 8,
454            },
455        };
456        let json = serde_json::to_string(&token).unwrap();
457        let back: Token = serde_json::from_str(&json).unwrap();
458        assert_eq!(token, back);
459    }
460
461    #[test]
462    fn cpd_clone_serializes_with_blame() {
463        let loc = Location {
464            line: 1,
465            column: 0,
466            offset: 0,
467        };
468        let blame = BlameEntry {
469            commit_sha: "abc123".to_string(),
470            author: "Alice".to_string(),
471            timestamp: 1700000000,
472        };
473        let frag = Fragment::new("a.js", loc.clone(), loc, [0, 10]).with_blame(blame);
474        let clone = CpdClone::exact("javascript", frag.clone(), frag, 50);
475        let json = serde_json::to_string(&clone).unwrap();
476        assert!(json.contains("abc123"));
477        assert!(json.contains("fragment_a"));
478    }
479
480    #[test]
481    fn fragment_blame_none_serializes_as_null() {
482        let loc = Location {
483            line: 1,
484            column: 0,
485            offset: 0,
486        };
487        let frag = Fragment {
488            source_id: "b.js".to_string(),
489            source_root: None,
490            start: loc.clone(),
491            end: loc.clone(),
492            range: [0, 5],
493            blame: None,
494        };
495        let json = serde_json::to_string(&frag).unwrap();
496        assert!(json.contains("\"blame\":null"));
497    }
498
499    #[test]
500    fn covered_lines_counts_a_contiguous_run_once() {
501        assert_eq!(covered_lines([(1, 1), (1, 1), (2, 2), (3, 3)]), 3);
502        assert_eq!(covered_lines(std::iter::empty()), 0);
503    }
504
505    #[test]
506    fn covered_lines_skips_the_gap_between_two_blocks() {
507        // Lines 17-26 and 43-52: twenty lines of code, whatever sits between.
508        let block = |from: u32, to: u32| (from..=to).map(|l| (l, l));
509        assert_eq!(covered_lines(block(17, 26).chain(block(43, 52))), 20);
510    }
511
512    #[test]
513    fn covered_lines_handles_a_token_spanning_several_lines() {
514        // A template literal running from line 4 to line 9, then code after it.
515        assert_eq!(covered_lines([(4, 9), (9, 9), (10, 10)]), 7);
516    }
517
518    #[test]
519    fn an_embedded_source_is_recognised_by_its_id() {
520        let source = |id: &str, format: &str| SourceFile {
521            id: id.to_string(),
522            format: format.to_string(),
523            tokens: vec![],
524            bytes: 0,
525        };
526        assert!(source("guide.md:typescript", "typescript").is_embedded());
527        assert!(!source("app.ts", "typescript").is_embedded());
528        // The host's own map is not embedded in anything.
529        assert!(!source("guide.md", "markdown").is_embedded());
530        // A file whose whole name is the format is still a file.
531        assert!(!source("typescript", "typescript").is_embedded());
532    }
533}