1use serde::{Deserialize, Serialize};
2use std::collections::HashMap;
3
4#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
5#[serde(rename_all = "snake_case")]
6pub enum TokenKind {
7 Keyword,
8 Identifier,
9 Literal,
10 Operator,
11 Punctuation,
12 Comment,
13 BlockComment,
14 Whitespace,
15 Ignore,
16 Other,
17}
18
19impl TokenKind {
20 pub fn discriminant(&self) -> u8 {
22 match self {
23 Self::Keyword => 1,
24 Self::Identifier => 2,
25 Self::Literal => 3,
26 Self::Operator => 4,
27 Self::Punctuation => 5,
28 Self::Comment => 6,
29 Self::BlockComment => 7,
30 Self::Whitespace => 8,
31 Self::Ignore => 9,
32 Self::Other => 10,
33 }
34 }
35}
36
37#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
38pub struct Location {
39 pub line: u32,
40 pub column: u32,
41 pub offset: u32,
42}
43
44#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
45pub struct Token {
46 pub kind: TokenKind,
47 pub value: String,
48 pub start: Location,
49 pub end: Location,
50}
51
52#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
53pub struct BlameEntry {
54 pub commit_sha: String,
55 pub author: String,
56 pub timestamp: i64,
57}
58
59#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
61#[serde(rename_all = "lowercase")]
62pub enum CloneKind {
63 #[default]
65 Exact,
66 Renamed,
70 Similar,
75}
76
77impl CloneKind {
78 pub fn is_renamed(self) -> bool {
79 matches!(self, CloneKind::Renamed)
80 }
81
82 pub fn is_similar(self) -> bool {
83 matches!(self, CloneKind::Similar)
84 }
85
86 pub fn as_str(self) -> &'static str {
87 match self {
88 CloneKind::Exact => "exact",
89 CloneKind::Renamed => "renamed",
90 CloneKind::Similar => "similar",
91 }
92 }
93}
94
95#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
96pub struct Fragment {
97 pub source_id: String,
98 #[serde(default, skip_serializing_if = "Option::is_none")]
99 pub source_root: Option<String>,
100 pub start: Location,
101 pub end: Location,
102 pub range: [u32; 2],
103 pub blame: Option<BlameEntry>,
104}
105
106#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
108#[serde(rename_all = "lowercase")]
109pub enum SimilarityMethod {
110 Gap,
113 Ast,
116}
117
118impl SimilarityMethod {
119 pub fn as_str(self) -> &'static str {
120 match self {
121 SimilarityMethod::Gap => "gap",
122 SimilarityMethod::Ast => "ast",
123 }
124 }
125}
126
127#[derive(Debug, Clone, Copy, PartialEq, Eq)]
130pub enum KindFilter {
131 Exact,
132 Renamed,
133 Similar,
135 Gap,
137 Ast,
139}
140
141impl KindFilter {
142 pub const NAMES: &'static str = "exact, renamed, similar, gap, ast";
143
144 pub fn as_str(self) -> &'static str {
145 match self {
146 KindFilter::Exact => "exact",
147 KindFilter::Renamed => "renamed",
148 KindFilter::Similar => "similar",
149 KindFilter::Gap => "gap",
150 KindFilter::Ast => "ast",
151 }
152 }
153
154 pub fn matches(self, clone: &CpdClone) -> bool {
155 match self {
156 KindFilter::Exact => clone.kind == CloneKind::Exact,
157 KindFilter::Renamed => clone.kind == CloneKind::Renamed,
158 KindFilter::Similar => clone.kind == CloneKind::Similar,
159 KindFilter::Gap => clone.similarity_method == Some(SimilarityMethod::Gap),
160 KindFilter::Ast => clone.similarity_method == Some(SimilarityMethod::Ast),
161 }
162 }
163}
164
165impl std::str::FromStr for KindFilter {
166 type Err = String;
167
168 fn from_str(s: &str) -> Result<Self, Self::Err> {
169 match s.trim().to_ascii_lowercase().as_str() {
170 "exact" => Ok(KindFilter::Exact),
171 "renamed" => Ok(KindFilter::Renamed),
172 "similar" => Ok(KindFilter::Similar),
173 "gap" => Ok(KindFilter::Gap),
174 "ast" => Ok(KindFilter::Ast),
175 other => Err(format!(
176 "unknown clone kind '{other}': must be one of: {}",
177 KindFilter::NAMES
178 )),
179 }
180 }
181}
182
183#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
184pub struct CpdClone {
185 pub format: String,
186 pub fragment_a: Fragment,
187 pub fragment_b: Fragment,
188 pub token_count: u32,
189 #[serde(default)]
192 pub is_new: bool,
193 #[serde(default)]
197 pub kind: CloneKind,
198 #[serde(default, skip_serializing_if = "Option::is_none")]
201 pub similarity: Option<f32>,
202 #[serde(default, rename = "method", skip_serializing_if = "Option::is_none")]
205 pub similarity_method: Option<SimilarityMethod>,
206 #[serde(skip)]
213 pub unmatched_lines: [u32; 2],
214}
215
216impl Location {
217 pub fn new(line: u32, column: u32, offset: u32) -> Self {
218 Self {
219 line,
220 column,
221 offset,
222 }
223 }
224}
225
226impl Fragment {
227 pub fn new(
230 source_id: impl Into<String>,
231 start: Location,
232 end: Location,
233 range: [u32; 2],
234 ) -> Self {
235 Self {
236 source_id: source_id.into(),
237 source_root: None,
238 start,
239 end,
240 range,
241 blame: None,
242 }
243 }
244
245 pub fn with_blame(mut self, blame: BlameEntry) -> Self {
246 self.blame = Some(blame);
247 self
248 }
249}
250
251impl CpdClone {
252 pub fn matched_lines(&self) -> u64 {
255 self.fragment_lines(0)
256 }
257
258 pub fn fragment_lines(&self, index: usize) -> u64 {
266 let fragment = if index == 0 {
267 &self.fragment_a
268 } else {
269 &self.fragment_b
270 };
271 let span = fragment.end.line.saturating_sub(fragment.start.line) + 1;
272 span.saturating_sub(self.unmatched_lines[index]) as u64
273 }
274
275 pub fn exact(
277 format: impl Into<String>,
278 fragment_a: Fragment,
279 fragment_b: Fragment,
280 token_count: u32,
281 ) -> Self {
282 Self {
283 format: format.into(),
284 fragment_a,
285 fragment_b,
286 token_count,
287 is_new: false,
288 kind: CloneKind::default(),
289 similarity: None,
290 similarity_method: None,
291 unmatched_lines: [0, 0],
292 }
293 }
294 pub fn similarity_rounded(&self) -> Option<f64> {
297 self.similarity
298 .map(|s| (f64::from(s) * 1000.0).round() / 1000.0)
299 }
300}
301
302#[derive(Debug, Clone, PartialEq, Eq)]
309pub struct DetectionToken {
310 pub hash: u64,
314 pub raw_hash: u64,
318 pub start: Location,
319 pub end: Location,
320 pub range: [usize; 2],
322}
323
324#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
326pub struct SourceFile {
327 pub id: String,
328 pub format: String,
329 pub tokens: Vec<Token>,
330 #[serde(default)]
333 pub bytes: u64,
334}
335
336impl SourceFile {
337 pub fn is_embedded(&self) -> bool {
342 self.id
343 .strip_suffix(self.format.as_str())
344 .and_then(|rest| rest.strip_suffix(':'))
345 .is_some_and(|path| !path.is_empty())
346 }
347
348 pub fn line_count(&self) -> u64 {
353 if self.is_embedded() {
354 covered_lines(self.tokens.iter().map(|t| (t.start.line, t.end.line))) as u64
355 } else {
356 self.tokens.iter().map(|t| t.start.line).max().unwrap_or(0) as u64
357 }
358 }
359}
360
361pub fn covered_lines(spans: impl IntoIterator<Item = (u32, u32)>) -> u32 {
368 let mut covered = 0;
369 let mut last = 0;
371 for (start, end) in spans {
372 let from = if start > last { start } else { last + 1 };
373 if end >= from {
374 covered += end - from + 1;
375 last = end;
376 }
377 }
378 covered
379}
380
381#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
383#[serde(rename_all = "camelCase")]
384pub struct StatRow {
385 pub lines: u64,
386 pub tokens: u64,
387 pub sources: u64,
388 pub clones: u64,
389 pub duplicated_lines: u64,
390 pub duplicated_tokens: u64,
391 pub percentage: f64,
392 pub percentage_tokens: f64,
393 #[serde(default)]
394 pub new_duplicated_lines: u64,
395 #[serde(default)]
396 pub new_clones: u64,
397}
398
399impl Default for StatRow {
400 fn default() -> Self {
401 Self {
402 lines: 0,
403 tokens: 0,
404 sources: 0,
405 clones: 0,
406 duplicated_lines: 0,
407 duplicated_tokens: 0,
408 percentage: 0.0,
409 percentage_tokens: 0.0,
410 new_duplicated_lines: 0,
411 new_clones: 0,
412 }
413 }
414}
415
416#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
418#[serde(rename_all = "camelCase")]
419pub struct Statistics {
420 pub total: StatRow,
421 pub formats: HashMap<String, StatRow>,
422 pub detection_date: String,
423}
424
425#[cfg(test)]
426mod tests {
427 use super::*;
428 use serde_json;
429
430 #[test]
431 fn statistics_default_total_is_zero() {
432 let stats = Statistics {
433 total: StatRow::default(),
434 formats: HashMap::new(),
435 detection_date: "2026-01-01T00:00:00Z".to_string(),
436 };
437 assert_eq!(stats.total.clones, 0);
438 }
439
440 #[test]
441 fn token_serializes_and_deserializes() {
442 let token = Token {
443 kind: TokenKind::Keyword,
444 value: "function".to_string(),
445 start: Location {
446 line: 1,
447 column: 0,
448 offset: 0,
449 },
450 end: Location {
451 line: 1,
452 column: 8,
453 offset: 8,
454 },
455 };
456 let json = serde_json::to_string(&token).unwrap();
457 let back: Token = serde_json::from_str(&json).unwrap();
458 assert_eq!(token, back);
459 }
460
461 #[test]
462 fn cpd_clone_serializes_with_blame() {
463 let loc = Location {
464 line: 1,
465 column: 0,
466 offset: 0,
467 };
468 let blame = BlameEntry {
469 commit_sha: "abc123".to_string(),
470 author: "Alice".to_string(),
471 timestamp: 1700000000,
472 };
473 let frag = Fragment::new("a.js", loc.clone(), loc, [0, 10]).with_blame(blame);
474 let clone = CpdClone::exact("javascript", frag.clone(), frag, 50);
475 let json = serde_json::to_string(&clone).unwrap();
476 assert!(json.contains("abc123"));
477 assert!(json.contains("fragment_a"));
478 }
479
480 #[test]
481 fn fragment_blame_none_serializes_as_null() {
482 let loc = Location {
483 line: 1,
484 column: 0,
485 offset: 0,
486 };
487 let frag = Fragment {
488 source_id: "b.js".to_string(),
489 source_root: None,
490 start: loc.clone(),
491 end: loc.clone(),
492 range: [0, 5],
493 blame: None,
494 };
495 let json = serde_json::to_string(&frag).unwrap();
496 assert!(json.contains("\"blame\":null"));
497 }
498
499 #[test]
500 fn covered_lines_counts_a_contiguous_run_once() {
501 assert_eq!(covered_lines([(1, 1), (1, 1), (2, 2), (3, 3)]), 3);
502 assert_eq!(covered_lines(std::iter::empty()), 0);
503 }
504
505 #[test]
506 fn covered_lines_skips_the_gap_between_two_blocks() {
507 let block = |from: u32, to: u32| (from..=to).map(|l| (l, l));
509 assert_eq!(covered_lines(block(17, 26).chain(block(43, 52))), 20);
510 }
511
512 #[test]
513 fn covered_lines_handles_a_token_spanning_several_lines() {
514 assert_eq!(covered_lines([(4, 9), (9, 9), (10, 10)]), 7);
516 }
517
518 #[test]
519 fn an_embedded_source_is_recognised_by_its_id() {
520 let source = |id: &str, format: &str| SourceFile {
521 id: id.to_string(),
522 format: format.to_string(),
523 tokens: vec![],
524 bytes: 0,
525 };
526 assert!(source("guide.md:typescript", "typescript").is_embedded());
527 assert!(!source("app.ts", "typescript").is_embedded());
528 assert!(!source("guide.md", "markdown").is_embedded());
530 assert!(!source("typescript", "typescript").is_embedded());
532 }
533}