Skip to main content

cpd_core/
models.rs

1use serde::{Deserialize, Serialize};
2use std::collections::HashMap;
3
4#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
5#[serde(rename_all = "snake_case")]
6pub enum TokenKind {
7    Keyword,
8    Identifier,
9    Literal,
10    Operator,
11    Punctuation,
12    Comment,
13    BlockComment,
14    Whitespace,
15    Ignore,
16    Other,
17}
18
19impl TokenKind {
20    /// Return a stable byte discriminant for use in token hashing.
21    pub fn discriminant(&self) -> u8 {
22        match self {
23            Self::Keyword => 1,
24            Self::Identifier => 2,
25            Self::Literal => 3,
26            Self::Operator => 4,
27            Self::Punctuation => 5,
28            Self::Comment => 6,
29            Self::BlockComment => 7,
30            Self::Whitespace => 8,
31            Self::Ignore => 9,
32            Self::Other => 10,
33        }
34    }
35}
36
37#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
38pub struct Location {
39    pub line: u32,
40    pub column: u32,
41    pub offset: u32,
42}
43
44#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
45pub struct Token {
46    pub kind: TokenKind,
47    pub value: String,
48    pub start: Location,
49    pub end: Location,
50}
51
52#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
53pub struct BlameEntry {
54    pub commit_sha: String,
55    pub author: String,
56    pub timestamp: i64,
57}
58
59#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
60pub struct Fragment {
61    pub source_id: String,
62    #[serde(default, skip_serializing_if = "Option::is_none")]
63    pub source_root: Option<String>,
64    pub start: Location,
65    pub end: Location,
66    pub range: [u32; 2],
67    pub blame: Option<BlameEntry>,
68}
69
70#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
71pub struct CpdClone {
72    pub format: String,
73    pub fragment_a: Fragment,
74    pub fragment_b: Fragment,
75    pub token_count: u32,
76    /// True when the clone is absent from the configured baseline (issue #944).
77    /// Always false when no baseline is in use.
78    #[serde(default)]
79    pub is_new: bool,
80}
81
82/// Internal detection unit — no heap allocation per token.
83///
84/// Produced by the tokenizer's detection path at tokenize time.
85/// `Token` is used for display, blame, and reporter output;
86/// `DetectionToken` is used only during the clone detection hot path.
87/// The token's value string is not stored — only its pre-computed hash.
88#[derive(Debug, Clone, PartialEq, Eq)]
89pub struct DetectionToken {
90    /// Pre-computed hash of (kind, value) — detection never re-hashes.
91    pub hash: u64,
92    pub start: Location,
93    pub end: Location,
94    /// Byte range in the source content: `[start_byte, end_byte]`.
95    pub range: [usize; 2],
96}
97
98/// A source file with pre-tokenized tokens, ready for clone detection.
99#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
100pub struct SourceFile {
101    pub id: String,
102    pub format: String,
103    pub tokens: Vec<Token>,
104    /// File size in bytes. Zero for synthetic sub-format sources
105    /// (embedded code blocks) whose bytes are counted by the parent file.
106    #[serde(default)]
107    pub bytes: u64,
108}
109
110/// Per-format or total statistics row.
111#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
112#[serde(rename_all = "camelCase")]
113pub struct StatRow {
114    pub lines: u64,
115    pub tokens: u64,
116    pub sources: u64,
117    pub clones: u64,
118    pub duplicated_lines: u64,
119    pub duplicated_tokens: u64,
120    pub percentage: f64,
121    pub percentage_tokens: f64,
122    #[serde(default)]
123    pub new_duplicated_lines: u64,
124    #[serde(default)]
125    pub new_clones: u64,
126}
127
128impl Default for StatRow {
129    fn default() -> Self {
130        Self {
131            lines: 0,
132            tokens: 0,
133            sources: 0,
134            clones: 0,
135            duplicated_lines: 0,
136            duplicated_tokens: 0,
137            percentage: 0.0,
138            percentage_tokens: 0.0,
139            new_duplicated_lines: 0,
140            new_clones: 0,
141        }
142    }
143}
144
145/// Aggregated detection statistics.
146#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
147#[serde(rename_all = "camelCase")]
148pub struct Statistics {
149    pub total: StatRow,
150    pub formats: HashMap<String, StatRow>,
151    pub detection_date: String,
152}
153
154#[cfg(test)]
155mod tests {
156    use super::*;
157    use serde_json;
158
159    #[test]
160    fn statistics_default_total_is_zero() {
161        let stats = Statistics {
162            total: StatRow::default(),
163            formats: HashMap::new(),
164            detection_date: "2026-01-01T00:00:00Z".to_string(),
165        };
166        assert_eq!(stats.total.clones, 0);
167    }
168
169    #[test]
170    fn token_serializes_and_deserializes() {
171        let token = Token {
172            kind: TokenKind::Keyword,
173            value: "function".to_string(),
174            start: Location {
175                line: 1,
176                column: 0,
177                offset: 0,
178            },
179            end: Location {
180                line: 1,
181                column: 8,
182                offset: 8,
183            },
184        };
185        let json = serde_json::to_string(&token).unwrap();
186        let back: Token = serde_json::from_str(&json).unwrap();
187        assert_eq!(token, back);
188    }
189
190    #[test]
191    fn cpd_clone_serializes_with_blame() {
192        let loc = Location {
193            line: 1,
194            column: 0,
195            offset: 0,
196        };
197        let blame = BlameEntry {
198            commit_sha: "abc123".to_string(),
199            author: "Alice".to_string(),
200            timestamp: 1700000000,
201        };
202        let frag = Fragment {
203            source_id: "a.js".to_string(),
204            source_root: None,
205            start: loc.clone(),
206            end: loc.clone(),
207            range: [0, 10],
208            blame: Some(blame),
209        };
210        let clone = CpdClone {
211            format: "javascript".to_string(),
212            fragment_a: frag.clone(),
213            fragment_b: frag,
214            token_count: 50,
215            is_new: false,
216        };
217        let json = serde_json::to_string(&clone).unwrap();
218        assert!(json.contains("abc123"));
219        assert!(json.contains("fragment_a"));
220    }
221
222    #[test]
223    fn fragment_blame_none_serializes_as_null() {
224        let loc = Location {
225            line: 1,
226            column: 0,
227            offset: 0,
228        };
229        let frag = Fragment {
230            source_id: "b.js".to_string(),
231            source_root: None,
232            start: loc.clone(),
233            end: loc.clone(),
234            range: [0, 5],
235            blame: None,
236        };
237        let json = serde_json::to_string(&frag).unwrap();
238        assert!(json.contains("\"blame\":null"));
239    }
240}