Skip to main content

cpd_core/
models.rs

1use serde::{Deserialize, Serialize};
2use std::collections::HashMap;
3
4#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
5#[serde(rename_all = "snake_case")]
6pub enum TokenKind {
7    Keyword,
8    Identifier,
9    Literal,
10    Operator,
11    Punctuation,
12    Comment,
13    BlockComment,
14    Whitespace,
15    Ignore,
16    Other,
17}
18
19impl TokenKind {
20    /// Return a stable byte discriminant for use in token hashing.
21    pub fn discriminant(&self) -> u8 {
22        match self {
23            Self::Keyword => 1,
24            Self::Identifier => 2,
25            Self::Literal => 3,
26            Self::Operator => 4,
27            Self::Punctuation => 5,
28            Self::Comment => 6,
29            Self::BlockComment => 7,
30            Self::Whitespace => 8,
31            Self::Ignore => 9,
32            Self::Other => 10,
33        }
34    }
35}
36
37#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
38pub struct Location {
39    pub line: u32,
40    pub column: u32,
41    pub offset: u32,
42}
43
44#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
45pub struct Token {
46    pub kind: TokenKind,
47    pub value: String,
48    pub start: Location,
49    pub end: Location,
50}
51
52#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
53pub struct BlameEntry {
54    pub commit_sha: String,
55    pub author: String,
56    pub timestamp: i64,
57}
58
59/// How the two fragments of a clone relate at the token level (issue #998).
60#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
61#[serde(rename_all = "lowercase")]
62pub enum CloneKind {
63    /// The fragments are token-for-token identical.
64    #[default]
65    Exact,
66    /// The fragments match only after identifier, literal or annotation
67    /// normalization (`--ignore-identifiers`, `--ignore-literals`,
68    /// `--ignore-annotations`): a Type-2 clone.
69    Renamed,
70    /// A Type-3 near-miss clone: two or more matches of the same file pair
71    /// merged across a gap of unmatched lines (`--max-gap-lines`), or a pair
72    /// of structurally similar functions (`--similarity`). `similar` takes
73    /// precedence over `renamed`: a merge of renamed halves is `similar`.
74    Similar,
75}
76
77impl CloneKind {
78    pub fn is_renamed(self) -> bool {
79        matches!(self, CloneKind::Renamed)
80    }
81
82    pub fn is_similar(self) -> bool {
83        matches!(self, CloneKind::Similar)
84    }
85
86    pub fn as_str(self) -> &'static str {
87        match self {
88            CloneKind::Exact => "exact",
89            CloneKind::Renamed => "renamed",
90            CloneKind::Similar => "similar",
91        }
92    }
93}
94
95#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
96pub struct Fragment {
97    pub source_id: String,
98    #[serde(default, skip_serializing_if = "Option::is_none")]
99    pub source_root: Option<String>,
100    pub start: Location,
101    pub end: Location,
102    pub range: [u32; 2],
103    pub blame: Option<BlameEntry>,
104}
105
106/// How a `similar` clone was produced (issue #999).
107#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
108#[serde(rename_all = "lowercase")]
109pub enum SimilarityMethod {
110    /// Exact matches merged across a gap of unmatched lines
111    /// (`--max-gap-lines`); `similarity` is matched tokens over the span.
112    Gap,
113    /// Whole functions compared by syntax-tree structure (`--similarity`);
114    /// `similarity` is the weighted Jaccard index of node-type shingles.
115    Ast,
116}
117
118impl SimilarityMethod {
119    pub fn as_str(self) -> &'static str {
120        match self {
121            SimilarityMethod::Gap => "gap",
122            SimilarityMethod::Ast => "ast",
123        }
124    }
125}
126
127/// One `--kind` value: a clone kind, or one of the two mechanisms that find
128/// `similar` clones.
129#[derive(Debug, Clone, Copy, PartialEq, Eq)]
130pub enum KindFilter {
131    Exact,
132    Renamed,
133    /// Every `similar` clone, whichever mechanism found it.
134    Similar,
135    /// `similar` clones merged across a gap (`--max-gap-lines`).
136    Gap,
137    /// `similar` function pairs compared by syntax tree (`--similarity`).
138    Ast,
139}
140
141impl KindFilter {
142    pub const NAMES: &'static str = "exact, renamed, similar, gap, ast";
143
144    pub fn as_str(self) -> &'static str {
145        match self {
146            KindFilter::Exact => "exact",
147            KindFilter::Renamed => "renamed",
148            KindFilter::Similar => "similar",
149            KindFilter::Gap => "gap",
150            KindFilter::Ast => "ast",
151        }
152    }
153
154    pub fn matches(self, clone: &CpdClone) -> bool {
155        match self {
156            KindFilter::Exact => clone.kind == CloneKind::Exact,
157            KindFilter::Renamed => clone.kind == CloneKind::Renamed,
158            KindFilter::Similar => clone.kind == CloneKind::Similar,
159            KindFilter::Gap => clone.similarity_method == Some(SimilarityMethod::Gap),
160            KindFilter::Ast => clone.similarity_method == Some(SimilarityMethod::Ast),
161        }
162    }
163}
164
165impl std::str::FromStr for KindFilter {
166    type Err = String;
167
168    fn from_str(s: &str) -> Result<Self, Self::Err> {
169        match s.trim().to_ascii_lowercase().as_str() {
170            "exact" => Ok(KindFilter::Exact),
171            "renamed" => Ok(KindFilter::Renamed),
172            "similar" => Ok(KindFilter::Similar),
173            "gap" => Ok(KindFilter::Gap),
174            "ast" => Ok(KindFilter::Ast),
175            other => Err(format!(
176                "unknown clone kind '{other}': must be one of: {}",
177                KindFilter::NAMES
178            )),
179        }
180    }
181}
182
183#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
184pub struct CpdClone {
185    pub format: String,
186    pub fragment_a: Fragment,
187    pub fragment_b: Fragment,
188    pub token_count: u32,
189    /// True when the clone is absent from the configured baseline (issue #944).
190    /// Always false when no baseline is in use.
191    #[serde(default)]
192    pub is_new: bool,
193    /// `exact` when the raw tokens of both fragments are identical, `renamed`
194    /// when they match only after normalization (issue #998). Always `exact`
195    /// when no normalization option is on.
196    #[serde(default)]
197    pub kind: CloneKind,
198    /// For `similar` clones: matched tokens divided by the tokens of the
199    /// longer merged span, in `(0, 1)`. `None` for exact and renamed clones.
200    #[serde(default, skip_serializing_if = "Option::is_none")]
201    pub similarity: Option<f32>,
202    /// Which mechanism produced a `similar` clone; the two scores are not
203    /// on the same scale, so reporters show it next to the value.
204    #[serde(default, rename = "method", skip_serializing_if = "Option::is_none")]
205    pub similarity_method: Option<SimilarityMethod>,
206    /// Lines inside each fragment's span that the gap merge (`--max-gap-lines`)
207    /// left unmatched, for `fragment_a` and `fragment_b` in that order.
208    /// Statistics subtract them so gap lines do not count as duplicated.
209    /// `[0, 0]` for every clone the merge pass did not produce.
210    #[serde(skip)]
211    pub unmatched_lines: [u32; 2],
212}
213
214impl Location {
215    pub fn new(line: u32, column: u32, offset: u32) -> Self {
216        Self {
217            line,
218            column,
219            offset,
220        }
221    }
222}
223
224impl Fragment {
225    /// A fragment with no scan root and no blame data, the shape detection
226    /// produces before enrichment.
227    pub fn new(
228        source_id: impl Into<String>,
229        start: Location,
230        end: Location,
231        range: [u32; 2],
232    ) -> Self {
233        Self {
234            source_id: source_id.into(),
235            source_root: None,
236            start,
237            end,
238            range,
239            blame: None,
240        }
241    }
242
243    pub fn with_blame(mut self, blame: BlameEntry) -> Self {
244        self.blame = Some(blame);
245        self
246    }
247}
248
249impl CpdClone {
250    /// Duplicated lines this clone adds to the statistics: the matched lines
251    /// of its primary fragment. A gap-merged clone's unmatched lines are not
252    /// duplicated code and stay out.
253    pub fn matched_lines(&self) -> u64 {
254        self.fragment_a
255            .end
256            .line
257            .saturating_sub(self.fragment_a.start.line)
258            .saturating_sub(self.unmatched_lines[0]) as u64
259    }
260
261    /// An exact clone with no baseline, similarity or gap metadata.
262    pub fn exact(
263        format: impl Into<String>,
264        fragment_a: Fragment,
265        fragment_b: Fragment,
266        token_count: u32,
267    ) -> Self {
268        Self {
269            format: format.into(),
270            fragment_a,
271            fragment_b,
272            token_count,
273            is_new: false,
274            kind: CloneKind::default(),
275            similarity: None,
276            similarity_method: None,
277            unmatched_lines: [0, 0],
278        }
279    }
280    /// `similarity` rounded to three decimals as an f64, the form reporters
281    /// print (an f32 widened to JSON would print as `0.8510638475418091`).
282    pub fn similarity_rounded(&self) -> Option<f64> {
283        self.similarity
284            .map(|s| (f64::from(s) * 1000.0).round() / 1000.0)
285    }
286}
287
288/// Internal detection unit — no heap allocation per token.
289///
290/// Produced by the tokenizer's detection path at tokenize time.
291/// `Token` is used for display, blame, and reporter output;
292/// `DetectionToken` is used only during the clone detection hot path.
293/// The token's value string is not stored — only its pre-computed hash.
294#[derive(Debug, Clone, PartialEq, Eq)]
295pub struct DetectionToken {
296    /// Pre-computed hash of (kind, value) — detection never re-hashes.
297    /// When a normalization option rewrote the value (`$id`, `$str`, `$num`)
298    /// this is the hash of the placeholder.
299    pub hash: u64,
300    /// Hash of the original (kind, value). Equal to `hash` unless a
301    /// normalization option applied; lets detection tell exact clones from
302    /// renamed ones without re-tokenizing.
303    pub raw_hash: u64,
304    pub start: Location,
305    pub end: Location,
306    /// Byte range in the source content: `[start_byte, end_byte]`.
307    pub range: [usize; 2],
308}
309
310/// A source file with pre-tokenized tokens, ready for clone detection.
311#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
312pub struct SourceFile {
313    pub id: String,
314    pub format: String,
315    pub tokens: Vec<Token>,
316    /// File size in bytes. Zero for synthetic sub-format sources
317    /// (embedded code blocks) whose bytes are counted by the parent file.
318    #[serde(default)]
319    pub bytes: u64,
320}
321
322/// Per-format or total statistics row.
323#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
324#[serde(rename_all = "camelCase")]
325pub struct StatRow {
326    pub lines: u64,
327    pub tokens: u64,
328    pub sources: u64,
329    pub clones: u64,
330    pub duplicated_lines: u64,
331    pub duplicated_tokens: u64,
332    pub percentage: f64,
333    pub percentage_tokens: f64,
334    #[serde(default)]
335    pub new_duplicated_lines: u64,
336    #[serde(default)]
337    pub new_clones: u64,
338}
339
340impl Default for StatRow {
341    fn default() -> Self {
342        Self {
343            lines: 0,
344            tokens: 0,
345            sources: 0,
346            clones: 0,
347            duplicated_lines: 0,
348            duplicated_tokens: 0,
349            percentage: 0.0,
350            percentage_tokens: 0.0,
351            new_duplicated_lines: 0,
352            new_clones: 0,
353        }
354    }
355}
356
357/// Aggregated detection statistics.
358#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
359#[serde(rename_all = "camelCase")]
360pub struct Statistics {
361    pub total: StatRow,
362    pub formats: HashMap<String, StatRow>,
363    pub detection_date: String,
364}
365
366#[cfg(test)]
367mod tests {
368    use super::*;
369    use serde_json;
370
371    #[test]
372    fn statistics_default_total_is_zero() {
373        let stats = Statistics {
374            total: StatRow::default(),
375            formats: HashMap::new(),
376            detection_date: "2026-01-01T00:00:00Z".to_string(),
377        };
378        assert_eq!(stats.total.clones, 0);
379    }
380
381    #[test]
382    fn token_serializes_and_deserializes() {
383        let token = Token {
384            kind: TokenKind::Keyword,
385            value: "function".to_string(),
386            start: Location {
387                line: 1,
388                column: 0,
389                offset: 0,
390            },
391            end: Location {
392                line: 1,
393                column: 8,
394                offset: 8,
395            },
396        };
397        let json = serde_json::to_string(&token).unwrap();
398        let back: Token = serde_json::from_str(&json).unwrap();
399        assert_eq!(token, back);
400    }
401
402    #[test]
403    fn cpd_clone_serializes_with_blame() {
404        let loc = Location {
405            line: 1,
406            column: 0,
407            offset: 0,
408        };
409        let blame = BlameEntry {
410            commit_sha: "abc123".to_string(),
411            author: "Alice".to_string(),
412            timestamp: 1700000000,
413        };
414        let frag = Fragment::new("a.js", loc.clone(), loc, [0, 10]).with_blame(blame);
415        let clone = CpdClone::exact("javascript", frag.clone(), frag, 50);
416        let json = serde_json::to_string(&clone).unwrap();
417        assert!(json.contains("abc123"));
418        assert!(json.contains("fragment_a"));
419    }
420
421    #[test]
422    fn fragment_blame_none_serializes_as_null() {
423        let loc = Location {
424            line: 1,
425            column: 0,
426            offset: 0,
427        };
428        let frag = Fragment {
429            source_id: "b.js".to_string(),
430            source_root: None,
431            start: loc.clone(),
432            end: loc.clone(),
433            range: [0, 5],
434            blame: None,
435        };
436        let json = serde_json::to_string(&frag).unwrap();
437        assert!(json.contains("\"blame\":null"));
438    }
439}