Skip to main content

cpd_core/
models.rs

1use serde::{Deserialize, Serialize};
2use std::collections::HashMap;
3
4#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
5#[serde(rename_all = "snake_case")]
6pub enum TokenKind {
7    Keyword,
8    Identifier,
9    Literal,
10    Operator,
11    Punctuation,
12    Comment,
13    BlockComment,
14    Whitespace,
15    Ignore,
16    Other,
17}
18
19impl TokenKind {
20    /// Return a stable byte discriminant for use in token hashing.
21    pub fn discriminant(&self) -> u8 {
22        match self {
23            Self::Keyword => 1,
24            Self::Identifier => 2,
25            Self::Literal => 3,
26            Self::Operator => 4,
27            Self::Punctuation => 5,
28            Self::Comment => 6,
29            Self::BlockComment => 7,
30            Self::Whitespace => 8,
31            Self::Ignore => 9,
32            Self::Other => 10,
33        }
34    }
35}
36
37#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
38pub struct Location {
39    pub line: u32,
40    pub column: u32,
41    pub offset: u32,
42}
43
44#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
45pub struct Token {
46    pub kind: TokenKind,
47    pub value: String,
48    pub start: Location,
49    pub end: Location,
50}
51
52#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
53pub struct BlameEntry {
54    pub commit_sha: String,
55    pub author: String,
56    pub timestamp: i64,
57}
58
59/// How the two fragments of a clone relate at the token level (issue #998).
60#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
61#[serde(rename_all = "lowercase")]
62pub enum CloneKind {
63    /// The fragments are token-for-token identical.
64    #[default]
65    Exact,
66    /// The fragments match only after identifier, literal or annotation
67    /// normalization (`--ignore-identifiers`, `--ignore-literals`,
68    /// `--ignore-annotations`): a Type-2 clone.
69    Renamed,
70    /// A Type-3 near-miss clone: two or more matches of the same file pair
71    /// merged across a gap of unmatched lines (`--max-gap-lines`), or a pair
72    /// of structurally similar functions (`--similarity`). `similar` takes
73    /// precedence over `renamed`: a merge of renamed halves is `similar`.
74    Similar,
75}
76
77impl CloneKind {
78    pub fn is_renamed(self) -> bool {
79        matches!(self, CloneKind::Renamed)
80    }
81
82    pub fn is_similar(self) -> bool {
83        matches!(self, CloneKind::Similar)
84    }
85
86    pub fn is_exact(self) -> bool {
87        matches!(self, CloneKind::Exact)
88    }
89
90    pub fn as_str(self) -> &'static str {
91        match self {
92            CloneKind::Exact => "exact",
93            CloneKind::Renamed => "renamed",
94            CloneKind::Similar => "similar",
95        }
96    }
97}
98
99#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
100pub struct Fragment {
101    pub source_id: String,
102    #[serde(default, skip_serializing_if = "Option::is_none")]
103    pub source_root: Option<String>,
104    pub start: Location,
105    pub end: Location,
106    pub range: [u32; 2],
107    pub blame: Option<BlameEntry>,
108}
109
110/// How a `similar` clone was produced (issue #999).
111#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
112#[serde(rename_all = "lowercase")]
113pub enum SimilarityMethod {
114    /// Exact matches merged across a gap of unmatched lines
115    /// (`--max-gap-lines`); `similarity` is matched tokens over the span.
116    Gap,
117    /// Whole functions compared by syntax-tree structure (`--similarity`);
118    /// `similarity` is the weighted Jaccard index of node-type shingles.
119    Ast,
120}
121
122impl SimilarityMethod {
123    pub fn as_str(self) -> &'static str {
124        match self {
125            SimilarityMethod::Gap => "gap",
126            SimilarityMethod::Ast => "ast",
127        }
128    }
129}
130
131#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
132pub struct CpdClone {
133    pub format: String,
134    pub fragment_a: Fragment,
135    pub fragment_b: Fragment,
136    pub token_count: u32,
137    /// True when the clone is absent from the configured baseline (issue #944).
138    /// Always false when no baseline is in use.
139    #[serde(default)]
140    pub is_new: bool,
141    /// `exact` when the raw tokens of both fragments are identical, `renamed`
142    /// when they match only after normalization (issue #998). Always `exact`
143    /// when no normalization option is on.
144    #[serde(default)]
145    pub kind: CloneKind,
146    /// For `similar` clones: matched tokens divided by the tokens of the
147    /// longer merged span, in `(0, 1)`. `None` for exact and renamed clones.
148    #[serde(default, skip_serializing_if = "Option::is_none")]
149    pub similarity: Option<f32>,
150    /// Which mechanism produced a `similar` clone; the two scores are not
151    /// on the same scale, so reporters show it next to the value.
152    #[serde(default, rename = "method", skip_serializing_if = "Option::is_none")]
153    pub similarity_method: Option<SimilarityMethod>,
154    /// Lines inside each fragment's span that the gap merge (`--max-gap-lines`)
155    /// left unmatched, for `fragment_a` and `fragment_b` in that order.
156    /// Statistics subtract them so gap lines do not count as duplicated.
157    /// `[0, 0]` for every clone the merge pass did not produce.
158    #[serde(skip)]
159    pub unmatched_lines: [u32; 2],
160}
161
162impl Location {
163    pub fn new(line: u32, column: u32, offset: u32) -> Self {
164        Self {
165            line,
166            column,
167            offset,
168        }
169    }
170}
171
172impl Fragment {
173    /// A fragment with no scan root and no blame data, the shape detection
174    /// produces before enrichment.
175    pub fn new(
176        source_id: impl Into<String>,
177        start: Location,
178        end: Location,
179        range: [u32; 2],
180    ) -> Self {
181        Self {
182            source_id: source_id.into(),
183            source_root: None,
184            start,
185            end,
186            range,
187            blame: None,
188        }
189    }
190
191    pub fn with_blame(mut self, blame: BlameEntry) -> Self {
192        self.blame = Some(blame);
193        self
194    }
195}
196
197impl CpdClone {
198    /// An exact clone with no baseline, similarity or gap metadata.
199    pub fn exact(
200        format: impl Into<String>,
201        fragment_a: Fragment,
202        fragment_b: Fragment,
203        token_count: u32,
204    ) -> Self {
205        Self {
206            format: format.into(),
207            fragment_a,
208            fragment_b,
209            token_count,
210            is_new: false,
211            kind: CloneKind::default(),
212            similarity: None,
213            similarity_method: None,
214            unmatched_lines: [0, 0],
215        }
216    }
217    /// `similarity` rounded to three decimals as an f64, the form reporters
218    /// print (an f32 widened to JSON would print as `0.8510638475418091`).
219    pub fn similarity_rounded(&self) -> Option<f64> {
220        self.similarity
221            .map(|s| (f64::from(s) * 1000.0).round() / 1000.0)
222    }
223}
224
225/// Internal detection unit — no heap allocation per token.
226///
227/// Produced by the tokenizer's detection path at tokenize time.
228/// `Token` is used for display, blame, and reporter output;
229/// `DetectionToken` is used only during the clone detection hot path.
230/// The token's value string is not stored — only its pre-computed hash.
231#[derive(Debug, Clone, PartialEq, Eq)]
232pub struct DetectionToken {
233    /// Pre-computed hash of (kind, value) — detection never re-hashes.
234    /// When a normalization option rewrote the value (`$id`, `$str`, `$num`)
235    /// this is the hash of the placeholder.
236    pub hash: u64,
237    /// Hash of the original (kind, value). Equal to `hash` unless a
238    /// normalization option applied; lets detection tell exact clones from
239    /// renamed ones without re-tokenizing.
240    pub raw_hash: u64,
241    pub start: Location,
242    pub end: Location,
243    /// Byte range in the source content: `[start_byte, end_byte]`.
244    pub range: [usize; 2],
245}
246
247/// A source file with pre-tokenized tokens, ready for clone detection.
248#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
249pub struct SourceFile {
250    pub id: String,
251    pub format: String,
252    pub tokens: Vec<Token>,
253    /// File size in bytes. Zero for synthetic sub-format sources
254    /// (embedded code blocks) whose bytes are counted by the parent file.
255    #[serde(default)]
256    pub bytes: u64,
257}
258
259/// Per-format or total statistics row.
260#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
261#[serde(rename_all = "camelCase")]
262pub struct StatRow {
263    pub lines: u64,
264    pub tokens: u64,
265    pub sources: u64,
266    pub clones: u64,
267    pub duplicated_lines: u64,
268    pub duplicated_tokens: u64,
269    pub percentage: f64,
270    pub percentage_tokens: f64,
271    #[serde(default)]
272    pub new_duplicated_lines: u64,
273    #[serde(default)]
274    pub new_clones: u64,
275}
276
277impl Default for StatRow {
278    fn default() -> Self {
279        Self {
280            lines: 0,
281            tokens: 0,
282            sources: 0,
283            clones: 0,
284            duplicated_lines: 0,
285            duplicated_tokens: 0,
286            percentage: 0.0,
287            percentage_tokens: 0.0,
288            new_duplicated_lines: 0,
289            new_clones: 0,
290        }
291    }
292}
293
294/// Aggregated detection statistics.
295#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
296#[serde(rename_all = "camelCase")]
297pub struct Statistics {
298    pub total: StatRow,
299    pub formats: HashMap<String, StatRow>,
300    pub detection_date: String,
301}
302
303#[cfg(test)]
304mod tests {
305    use super::*;
306    use serde_json;
307
308    #[test]
309    fn statistics_default_total_is_zero() {
310        let stats = Statistics {
311            total: StatRow::default(),
312            formats: HashMap::new(),
313            detection_date: "2026-01-01T00:00:00Z".to_string(),
314        };
315        assert_eq!(stats.total.clones, 0);
316    }
317
318    #[test]
319    fn token_serializes_and_deserializes() {
320        let token = Token {
321            kind: TokenKind::Keyword,
322            value: "function".to_string(),
323            start: Location {
324                line: 1,
325                column: 0,
326                offset: 0,
327            },
328            end: Location {
329                line: 1,
330                column: 8,
331                offset: 8,
332            },
333        };
334        let json = serde_json::to_string(&token).unwrap();
335        let back: Token = serde_json::from_str(&json).unwrap();
336        assert_eq!(token, back);
337    }
338
339    #[test]
340    fn cpd_clone_serializes_with_blame() {
341        let loc = Location {
342            line: 1,
343            column: 0,
344            offset: 0,
345        };
346        let blame = BlameEntry {
347            commit_sha: "abc123".to_string(),
348            author: "Alice".to_string(),
349            timestamp: 1700000000,
350        };
351        let frag = Fragment::new("a.js", loc.clone(), loc, [0, 10]).with_blame(blame);
352        let clone = CpdClone::exact("javascript", frag.clone(), frag, 50);
353        let json = serde_json::to_string(&clone).unwrap();
354        assert!(json.contains("abc123"));
355        assert!(json.contains("fragment_a"));
356    }
357
358    #[test]
359    fn fragment_blame_none_serializes_as_null() {
360        let loc = Location {
361            line: 1,
362            column: 0,
363            offset: 0,
364        };
365        let frag = Fragment {
366            source_id: "b.js".to_string(),
367            source_root: None,
368            start: loc.clone(),
369            end: loc.clone(),
370            range: [0, 5],
371            blame: None,
372        };
373        let json = serde_json::to_string(&frag).unwrap();
374        assert!(json.contains("\"blame\":null"));
375    }
376}