Skip to main content

cpd_core/
models.rs

1use serde::{Deserialize, Serialize};
2use std::collections::HashMap;
3
4#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
5#[serde(rename_all = "snake_case")]
6pub enum TokenKind {
7    Keyword,
8    Identifier,
9    Literal,
10    Operator,
11    Punctuation,
12    Comment,
13    BlockComment,
14    Whitespace,
15    Ignore,
16    Other,
17}
18
19impl TokenKind {
20    /// Return a stable byte discriminant for use in token hashing.
21    pub fn discriminant(&self) -> u8 {
22        match self {
23            Self::Keyword => 1,
24            Self::Identifier => 2,
25            Self::Literal => 3,
26            Self::Operator => 4,
27            Self::Punctuation => 5,
28            Self::Comment => 6,
29            Self::BlockComment => 7,
30            Self::Whitespace => 8,
31            Self::Ignore => 9,
32            Self::Other => 10,
33        }
34    }
35}
36
37#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
38pub struct Location {
39    pub line: u32,
40    pub column: u32,
41    pub offset: u32,
42}
43
44#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
45pub struct Token {
46    pub kind: TokenKind,
47    pub value: String,
48    pub start: Location,
49    pub end: Location,
50}
51
52#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
53pub struct BlameEntry {
54    pub commit_sha: String,
55    pub author: String,
56    pub timestamp: i64,
57}
58
59#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
60pub struct Fragment {
61    pub source_id: String,
62    #[serde(default, skip_serializing_if = "Option::is_none")]
63    pub source_root: Option<String>,
64    pub start: Location,
65    pub end: Location,
66    pub range: [u32; 2],
67    pub blame: Option<BlameEntry>,
68}
69
70#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
71pub struct CpdClone {
72    pub format: String,
73    pub fragment_a: Fragment,
74    pub fragment_b: Fragment,
75    pub token_count: u32,
76}
77
78/// Internal detection unit — no heap allocation per token.
79///
80/// Produced by the tokenizer's detection path at tokenize time.
81/// `Token` is used for display, blame, and reporter output;
82/// `DetectionToken` is used only during the clone detection hot path.
83/// The token's value string is not stored — only its pre-computed hash.
84#[derive(Debug, Clone, PartialEq, Eq)]
85pub struct DetectionToken {
86    /// Pre-computed hash of (kind, value) — detection never re-hashes.
87    pub hash: u64,
88    pub start: Location,
89    pub end: Location,
90    /// Byte range in the source content: `[start_byte, end_byte]`.
91    pub range: [usize; 2],
92}
93
94/// A source file with pre-tokenized tokens, ready for clone detection.
95#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
96pub struct SourceFile {
97    pub id: String,
98    pub format: String,
99    pub tokens: Vec<Token>,
100}
101
102/// Per-format or total statistics row.
103#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
104#[serde(rename_all = "camelCase")]
105pub struct StatRow {
106    pub lines: u64,
107    pub tokens: u64,
108    pub sources: u64,
109    pub clones: u64,
110    pub duplicated_lines: u64,
111    pub duplicated_tokens: u64,
112    pub percentage: f64,
113    pub percentage_tokens: f64,
114    #[serde(default)]
115    pub new_duplicated_lines: u64,
116    #[serde(default)]
117    pub new_clones: u64,
118}
119
120impl Default for StatRow {
121    fn default() -> Self {
122        Self {
123            lines: 0,
124            tokens: 0,
125            sources: 0,
126            clones: 0,
127            duplicated_lines: 0,
128            duplicated_tokens: 0,
129            percentage: 0.0,
130            percentage_tokens: 0.0,
131            new_duplicated_lines: 0,
132            new_clones: 0,
133        }
134    }
135}
136
137/// Aggregated detection statistics.
138#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
139#[serde(rename_all = "camelCase")]
140pub struct Statistics {
141    pub total: StatRow,
142    pub formats: HashMap<String, StatRow>,
143    pub detection_date: String,
144}
145
146#[cfg(test)]
147mod tests {
148    use super::*;
149    use serde_json;
150
151    #[test]
152    fn statistics_default_total_is_zero() {
153        let stats = Statistics {
154            total: StatRow::default(),
155            formats: HashMap::new(),
156            detection_date: "2026-01-01T00:00:00Z".to_string(),
157        };
158        assert_eq!(stats.total.clones, 0);
159    }
160
161    #[test]
162    fn token_serializes_and_deserializes() {
163        let token = Token {
164            kind: TokenKind::Keyword,
165            value: "function".to_string(),
166            start: Location {
167                line: 1,
168                column: 0,
169                offset: 0,
170            },
171            end: Location {
172                line: 1,
173                column: 8,
174                offset: 8,
175            },
176        };
177        let json = serde_json::to_string(&token).unwrap();
178        let back: Token = serde_json::from_str(&json).unwrap();
179        assert_eq!(token, back);
180    }
181
182    #[test]
183    fn cpd_clone_serializes_with_blame() {
184        let loc = Location {
185            line: 1,
186            column: 0,
187            offset: 0,
188        };
189        let blame = BlameEntry {
190            commit_sha: "abc123".to_string(),
191            author: "Alice".to_string(),
192            timestamp: 1700000000,
193        };
194        let frag = Fragment {
195            source_id: "a.js".to_string(),
196            source_root: None,
197            start: loc.clone(),
198            end: loc.clone(),
199            range: [0, 10],
200            blame: Some(blame),
201        };
202        let clone = CpdClone {
203            format: "javascript".to_string(),
204            fragment_a: frag.clone(),
205            fragment_b: frag,
206            token_count: 50,
207        };
208        let json = serde_json::to_string(&clone).unwrap();
209        assert!(json.contains("abc123"));
210        assert!(json.contains("fragment_a"));
211    }
212
213    #[test]
214    fn fragment_blame_none_serializes_as_null() {
215        let loc = Location {
216            line: 1,
217            column: 0,
218            offset: 0,
219        };
220        let frag = Fragment {
221            source_id: "b.js".to_string(),
222            source_root: None,
223            start: loc.clone(),
224            end: loc.clone(),
225            range: [0, 5],
226            blame: None,
227        };
228        let json = serde_json::to_string(&frag).unwrap();
229        assert!(json.contains("\"blame\":null"));
230    }
231}