Skip to main content

cpd_core/
models.rs

1use serde::{Deserialize, Serialize};
2use std::collections::HashMap;
3
4#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
5#[serde(rename_all = "snake_case")]
6pub enum TokenKind {
7    Keyword,
8    Identifier,
9    Literal,
10    Operator,
11    Punctuation,
12    Comment,
13    BlockComment,
14    Whitespace,
15    Ignore,
16    Other,
17}
18
19impl TokenKind {
20    /// Return a stable byte discriminant for use in token hashing.
21    pub fn discriminant(&self) -> u8 {
22        match self {
23            Self::Keyword => 1,
24            Self::Identifier => 2,
25            Self::Literal => 3,
26            Self::Operator => 4,
27            Self::Punctuation => 5,
28            Self::Comment => 6,
29            Self::BlockComment => 7,
30            Self::Whitespace => 8,
31            Self::Ignore => 9,
32            Self::Other => 10,
33        }
34    }
35}
36
37#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
38pub struct Location {
39    pub line: u32,
40    pub column: u32,
41    pub offset: u32,
42}
43
44#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
45pub struct Token {
46    pub kind: TokenKind,
47    pub value: String,
48    pub start: Location,
49    pub end: Location,
50}
51
52#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
53pub struct BlameEntry {
54    pub commit_sha: String,
55    pub author: String,
56    pub timestamp: i64,
57}
58
59#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
60pub struct Fragment {
61    pub source_id: String,
62    #[serde(default, skip_serializing_if = "Option::is_none")]
63    pub source_root: Option<String>,
64    pub start: Location,
65    pub end: Location,
66    pub range: [u32; 2],
67    pub blame: Option<BlameEntry>,
68}
69
70#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
71pub struct CpdClone {
72    pub format: String,
73    pub fragment_a: Fragment,
74    pub fragment_b: Fragment,
75    pub token_count: u32,
76}
77
78/// Internal detection unit — no heap allocation per token.
79///
80/// Produced by the tokenizer's detection path at tokenize time.
81/// `Token` is used for display, blame, and reporter output;
82/// `DetectionToken` is used only during the clone detection hot path.
83/// The token's value string is not stored — only its pre-computed hash.
84#[derive(Debug, Clone, PartialEq, Eq)]
85pub struct DetectionToken {
86    /// Pre-computed hash of (kind, value) — detection never re-hashes.
87    pub hash: u64,
88    pub start: Location,
89    pub end: Location,
90    /// Byte range in the source content: `[start_byte, end_byte]`.
91    pub range: [usize; 2],
92}
93
94/// A source file with pre-tokenized tokens, ready for clone detection.
95#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
96pub struct SourceFile {
97    pub id: String,
98    pub format: String,
99    pub tokens: Vec<Token>,
100    /// File size in bytes. Zero for synthetic sub-format sources
101    /// (embedded code blocks) whose bytes are counted by the parent file.
102    #[serde(default)]
103    pub bytes: u64,
104}
105
106/// Per-format or total statistics row.
107#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
108#[serde(rename_all = "camelCase")]
109pub struct StatRow {
110    pub lines: u64,
111    pub tokens: u64,
112    pub sources: u64,
113    pub clones: u64,
114    pub duplicated_lines: u64,
115    pub duplicated_tokens: u64,
116    pub percentage: f64,
117    pub percentage_tokens: f64,
118    #[serde(default)]
119    pub new_duplicated_lines: u64,
120    #[serde(default)]
121    pub new_clones: u64,
122}
123
124impl Default for StatRow {
125    fn default() -> Self {
126        Self {
127            lines: 0,
128            tokens: 0,
129            sources: 0,
130            clones: 0,
131            duplicated_lines: 0,
132            duplicated_tokens: 0,
133            percentage: 0.0,
134            percentage_tokens: 0.0,
135            new_duplicated_lines: 0,
136            new_clones: 0,
137        }
138    }
139}
140
141/// Aggregated detection statistics.
142#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
143#[serde(rename_all = "camelCase")]
144pub struct Statistics {
145    pub total: StatRow,
146    pub formats: HashMap<String, StatRow>,
147    pub detection_date: String,
148}
149
150#[cfg(test)]
151mod tests {
152    use super::*;
153    use serde_json;
154
155    #[test]
156    fn statistics_default_total_is_zero() {
157        let stats = Statistics {
158            total: StatRow::default(),
159            formats: HashMap::new(),
160            detection_date: "2026-01-01T00:00:00Z".to_string(),
161        };
162        assert_eq!(stats.total.clones, 0);
163    }
164
165    #[test]
166    fn token_serializes_and_deserializes() {
167        let token = Token {
168            kind: TokenKind::Keyword,
169            value: "function".to_string(),
170            start: Location {
171                line: 1,
172                column: 0,
173                offset: 0,
174            },
175            end: Location {
176                line: 1,
177                column: 8,
178                offset: 8,
179            },
180        };
181        let json = serde_json::to_string(&token).unwrap();
182        let back: Token = serde_json::from_str(&json).unwrap();
183        assert_eq!(token, back);
184    }
185
186    #[test]
187    fn cpd_clone_serializes_with_blame() {
188        let loc = Location {
189            line: 1,
190            column: 0,
191            offset: 0,
192        };
193        let blame = BlameEntry {
194            commit_sha: "abc123".to_string(),
195            author: "Alice".to_string(),
196            timestamp: 1700000000,
197        };
198        let frag = Fragment {
199            source_id: "a.js".to_string(),
200            source_root: None,
201            start: loc.clone(),
202            end: loc.clone(),
203            range: [0, 10],
204            blame: Some(blame),
205        };
206        let clone = CpdClone {
207            format: "javascript".to_string(),
208            fragment_a: frag.clone(),
209            fragment_b: frag,
210            token_count: 50,
211        };
212        let json = serde_json::to_string(&clone).unwrap();
213        assert!(json.contains("abc123"));
214        assert!(json.contains("fragment_a"));
215    }
216
217    #[test]
218    fn fragment_blame_none_serializes_as_null() {
219        let loc = Location {
220            line: 1,
221            column: 0,
222            offset: 0,
223        };
224        let frag = Fragment {
225            source_id: "b.js".to_string(),
226            source_root: None,
227            start: loc.clone(),
228            end: loc.clone(),
229            range: [0, 5],
230            blame: None,
231        };
232        let json = serde_json::to_string(&frag).unwrap();
233        assert!(json.contains("\"blame\":null"));
234    }
235}