Skip to main content

cpd_core/
models.rs

1use serde::{Deserialize, Serialize};
2use std::collections::HashMap;
3
4#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
5#[serde(rename_all = "snake_case")]
6pub enum TokenKind {
7    Keyword,
8    Identifier,
9    Literal,
10    Operator,
11    Punctuation,
12    Comment,
13    BlockComment,
14    Whitespace,
15    Ignore,
16    Other,
17}
18
19impl TokenKind {
20    /// Return a stable byte discriminant for use in token hashing.
21    pub fn discriminant(&self) -> u8 {
22        match self {
23            Self::Keyword => 1,
24            Self::Identifier => 2,
25            Self::Literal => 3,
26            Self::Operator => 4,
27            Self::Punctuation => 5,
28            Self::Comment => 6,
29            Self::BlockComment => 7,
30            Self::Whitespace => 8,
31            Self::Ignore => 9,
32            Self::Other => 10,
33        }
34    }
35}
36
37#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
38pub struct Location {
39    pub line: u32,
40    pub column: u32,
41    pub offset: u32,
42}
43
44#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
45pub struct Token {
46    pub kind: TokenKind,
47    pub value: String,
48    pub start: Location,
49    pub end: Location,
50}
51
52#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
53pub struct BlameEntry {
54    pub commit_sha: String,
55    pub author: String,
56    pub timestamp: i64,
57}
58
59/// How the two fragments of a clone relate at the token level (issue #998).
60#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
61#[serde(rename_all = "lowercase")]
62pub enum CloneKind {
63    /// The fragments are token-for-token identical.
64    #[default]
65    Exact,
66    /// The fragments match only after identifier, literal or annotation
67    /// normalization (`--ignore-identifiers`, `--ignore-literals`,
68    /// `--ignore-annotations`): a Type-2 clone.
69    Renamed,
70    /// A Type-3 near-miss clone: two or more matches of the same file pair
71    /// merged across a gap of unmatched lines (`--max-gap-lines`), or a pair
72    /// of structurally similar functions (`--similarity`). `similar` takes
73    /// precedence over `renamed`: a merge of renamed halves is `similar`.
74    Similar,
75}
76
77impl CloneKind {
78    pub fn is_renamed(self) -> bool {
79        matches!(self, CloneKind::Renamed)
80    }
81
82    pub fn is_similar(self) -> bool {
83        matches!(self, CloneKind::Similar)
84    }
85
86    pub fn is_exact(self) -> bool {
87        matches!(self, CloneKind::Exact)
88    }
89
90    pub fn as_str(self) -> &'static str {
91        match self {
92            CloneKind::Exact => "exact",
93            CloneKind::Renamed => "renamed",
94            CloneKind::Similar => "similar",
95        }
96    }
97}
98
99#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
100pub struct Fragment {
101    pub source_id: String,
102    #[serde(default, skip_serializing_if = "Option::is_none")]
103    pub source_root: Option<String>,
104    pub start: Location,
105    pub end: Location,
106    pub range: [u32; 2],
107    pub blame: Option<BlameEntry>,
108}
109
110/// How a `similar` clone was produced (issue #999).
111#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
112#[serde(rename_all = "lowercase")]
113pub enum SimilarityMethod {
114    /// Exact matches merged across a gap of unmatched lines
115    /// (`--max-gap-lines`); `similarity` is matched tokens over the span.
116    Gap,
117    /// Whole functions compared by syntax-tree structure (`--similarity`);
118    /// `similarity` is the weighted Jaccard index of node-type shingles.
119    Ast,
120}
121
122impl SimilarityMethod {
123    pub fn as_str(self) -> &'static str {
124        match self {
125            SimilarityMethod::Gap => "gap",
126            SimilarityMethod::Ast => "ast",
127        }
128    }
129}
130
131#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
132pub struct CpdClone {
133    pub format: String,
134    pub fragment_a: Fragment,
135    pub fragment_b: Fragment,
136    pub token_count: u32,
137    /// True when the clone is absent from the configured baseline (issue #944).
138    /// Always false when no baseline is in use.
139    #[serde(default)]
140    pub is_new: bool,
141    /// `exact` when the raw tokens of both fragments are identical, `renamed`
142    /// when they match only after normalization (issue #998). Always `exact`
143    /// when no normalization option is on.
144    #[serde(default)]
145    pub kind: CloneKind,
146    /// For `similar` clones: matched tokens divided by the tokens of the
147    /// longer merged span, in `(0, 1)`. `None` for exact and renamed clones.
148    #[serde(default, skip_serializing_if = "Option::is_none")]
149    pub similarity: Option<f32>,
150    /// Which mechanism produced a `similar` clone; the two scores are not
151    /// on the same scale, so reporters show it next to the value.
152    #[serde(default, rename = "method", skip_serializing_if = "Option::is_none")]
153    pub similarity_method: Option<SimilarityMethod>,
154    /// Lines inside each fragment's span that the gap merge (`--max-gap-lines`)
155    /// left unmatched, for `fragment_a` and `fragment_b` in that order.
156    /// Statistics subtract them so gap lines do not count as duplicated.
157    /// `[0, 0]` for every clone the merge pass did not produce.
158    #[serde(skip)]
159    pub unmatched_lines: [u32; 2],
160}
161
162impl CpdClone {
163    /// `similarity` rounded to three decimals as an f64, the form reporters
164    /// print (an f32 widened to JSON would print as `0.8510638475418091`).
165    pub fn similarity_rounded(&self) -> Option<f64> {
166        self.similarity
167            .map(|s| (f64::from(s) * 1000.0).round() / 1000.0)
168    }
169}
170
171/// Internal detection unit — no heap allocation per token.
172///
173/// Produced by the tokenizer's detection path at tokenize time.
174/// `Token` is used for display, blame, and reporter output;
175/// `DetectionToken` is used only during the clone detection hot path.
176/// The token's value string is not stored — only its pre-computed hash.
177#[derive(Debug, Clone, PartialEq, Eq)]
178pub struct DetectionToken {
179    /// Pre-computed hash of (kind, value) — detection never re-hashes.
180    /// When a normalization option rewrote the value (`$id`, `$str`, `$num`)
181    /// this is the hash of the placeholder.
182    pub hash: u64,
183    /// Hash of the original (kind, value). Equal to `hash` unless a
184    /// normalization option applied; lets detection tell exact clones from
185    /// renamed ones without re-tokenizing.
186    pub raw_hash: u64,
187    pub start: Location,
188    pub end: Location,
189    /// Byte range in the source content: `[start_byte, end_byte]`.
190    pub range: [usize; 2],
191}
192
193/// A source file with pre-tokenized tokens, ready for clone detection.
194#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
195pub struct SourceFile {
196    pub id: String,
197    pub format: String,
198    pub tokens: Vec<Token>,
199    /// File size in bytes. Zero for synthetic sub-format sources
200    /// (embedded code blocks) whose bytes are counted by the parent file.
201    #[serde(default)]
202    pub bytes: u64,
203}
204
205/// Per-format or total statistics row.
206#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
207#[serde(rename_all = "camelCase")]
208pub struct StatRow {
209    pub lines: u64,
210    pub tokens: u64,
211    pub sources: u64,
212    pub clones: u64,
213    pub duplicated_lines: u64,
214    pub duplicated_tokens: u64,
215    pub percentage: f64,
216    pub percentage_tokens: f64,
217    #[serde(default)]
218    pub new_duplicated_lines: u64,
219    #[serde(default)]
220    pub new_clones: u64,
221}
222
223impl Default for StatRow {
224    fn default() -> Self {
225        Self {
226            lines: 0,
227            tokens: 0,
228            sources: 0,
229            clones: 0,
230            duplicated_lines: 0,
231            duplicated_tokens: 0,
232            percentage: 0.0,
233            percentage_tokens: 0.0,
234            new_duplicated_lines: 0,
235            new_clones: 0,
236        }
237    }
238}
239
240/// Aggregated detection statistics.
241#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
242#[serde(rename_all = "camelCase")]
243pub struct Statistics {
244    pub total: StatRow,
245    pub formats: HashMap<String, StatRow>,
246    pub detection_date: String,
247}
248
249#[cfg(test)]
250mod tests {
251    use super::*;
252    use serde_json;
253
254    #[test]
255    fn statistics_default_total_is_zero() {
256        let stats = Statistics {
257            total: StatRow::default(),
258            formats: HashMap::new(),
259            detection_date: "2026-01-01T00:00:00Z".to_string(),
260        };
261        assert_eq!(stats.total.clones, 0);
262    }
263
264    #[test]
265    fn token_serializes_and_deserializes() {
266        let token = Token {
267            kind: TokenKind::Keyword,
268            value: "function".to_string(),
269            start: Location {
270                line: 1,
271                column: 0,
272                offset: 0,
273            },
274            end: Location {
275                line: 1,
276                column: 8,
277                offset: 8,
278            },
279        };
280        let json = serde_json::to_string(&token).unwrap();
281        let back: Token = serde_json::from_str(&json).unwrap();
282        assert_eq!(token, back);
283    }
284
285    #[test]
286    fn cpd_clone_serializes_with_blame() {
287        let loc = Location {
288            line: 1,
289            column: 0,
290            offset: 0,
291        };
292        let blame = BlameEntry {
293            commit_sha: "abc123".to_string(),
294            author: "Alice".to_string(),
295            timestamp: 1700000000,
296        };
297        let frag = Fragment {
298            source_id: "a.js".to_string(),
299            source_root: None,
300            start: loc.clone(),
301            end: loc.clone(),
302            range: [0, 10],
303            blame: Some(blame),
304        };
305        let clone = CpdClone {
306            format: "javascript".to_string(),
307            fragment_a: frag.clone(),
308            fragment_b: frag,
309            token_count: 50,
310            is_new: false,
311            kind: Default::default(),
312            similarity: None,
313            similarity_method: None,
314            unmatched_lines: [0, 0],
315        };
316        let json = serde_json::to_string(&clone).unwrap();
317        assert!(json.contains("abc123"));
318        assert!(json.contains("fragment_a"));
319    }
320
321    #[test]
322    fn fragment_blame_none_serializes_as_null() {
323        let loc = Location {
324            line: 1,
325            column: 0,
326            offset: 0,
327        };
328        let frag = Fragment {
329            source_id: "b.js".to_string(),
330            source_root: None,
331            start: loc.clone(),
332            end: loc.clone(),
333            range: [0, 5],
334            blame: None,
335        };
336        let json = serde_json::to_string(&frag).unwrap();
337        assert!(json.contains("\"blame\":null"));
338    }
339}