Skip to main content

cpd_core/
summary.rs

1// summary.rs — opt-in codebase summary: per-file metrics, folder rollup, top-N lists.
2//
3// Everything in this module runs only when `--summary` is enabled, after
4// detection has finished, over data already held in memory (SourceFile tokens
5// and detected clones). Nothing in the detection hot path calls into it.
6
7use crate::models::{CpdClone, SourceFile};
8use serde::{Deserialize, Serialize};
9use std::collections::HashMap;
10
11/// Metric used to rank files and folders in the summary.
12#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)]
13#[serde(rename_all = "lowercase")]
14pub enum SummaryMetric {
15    #[default]
16    Tokens,
17    Lines,
18    Size,
19    Complexity,
20}
21
22impl std::str::FromStr for SummaryMetric {
23    type Err = String;
24
25    fn from_str(s: &str) -> Result<Self, Self::Err> {
26        match s {
27            "tokens" => Ok(Self::Tokens),
28            "lines" => Ok(Self::Lines),
29            "size" => Ok(Self::Size),
30            "complexity" => Ok(Self::Complexity),
31            other => Err(format!(
32                "invalid summary metric '{other}': must be one of: tokens, lines, size, complexity"
33            )),
34        }
35    }
36}
37
38impl std::fmt::Display for SummaryMetric {
39    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
40        let s = match self {
41            Self::Tokens => "tokens",
42            Self::Lines => "lines",
43            Self::Size => "size",
44            Self::Complexity => "complexity",
45        };
46        f.write_str(s)
47    }
48}
49
50/// Per-file summary row.
51#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
52#[serde(rename_all = "camelCase")]
53pub struct FileSummary {
54    pub path: String,
55    pub format: String,
56    pub lines: u64,
57    pub tokens: u64,
58    pub bytes: u64,
59    pub duplicated_lines: u64,
60    pub duplicated_tokens: u64,
61    /// Cyclomatic-complexity estimate: 1 + count of decision-point tokens
62    /// (`if`, `for`, `while`, `case`, `catch`, `&&`, `||`, `?`, …).
63    pub complexity: u64,
64}
65
66/// Per-folder rollup. Files are counted in their direct parent directory only
67/// (no cumulative ancestor totals), so every file contributes to exactly one
68/// folder row and rows are directly comparable.
69#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
70#[serde(rename_all = "camelCase")]
71pub struct FolderSummary {
72    pub path: String,
73    pub files: u64,
74    pub lines: u64,
75    pub tokens: u64,
76    pub bytes: u64,
77    pub duplicated_lines: u64,
78    /// Sum of per-file complexity estimates (divide by `files` for the mean).
79    pub complexity: u64,
80}
81
82/// Codebase summary: top files and folder rollup.
83#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
84#[serde(rename_all = "camelCase")]
85pub struct Summary {
86    /// Primary sort metric.
87    pub by: SummaryMetric,
88    /// Top-N files by `by`, descending. Every row carries all metrics
89    /// (tokens, lines, bytes, complexity, duplication) so one list serves
90    /// every lens; re-run with a different `--summary-by` to re-rank.
91    pub files: Vec<FileSummary>,
92    /// Top-N folders by `by`, direct-parent aggregation.
93    pub folders: Vec<FolderSummary>,
94    /// Total number of files analyzed (before top-N truncation).
95    pub total_files: u64,
96    /// Total number of folders (before top-N truncation).
97    pub total_folders: u64,
98}
99
100/// Decision-point tokens counted by the complexity estimate. Conservative,
101/// language-agnostic list: branch/loop keywords and short-circuit operators
102/// that appear as standalone tokens across supported languages.
103///
104/// Matching is ASCII-case-insensitive so case-insensitive and
105/// uppercase-keyword languages (SQL, PL/SQL, Fortran, COBOL, BASIC, Pascal)
106/// count too. The occasional identifier spelled like a keyword slightly
107/// inflates an estimate that is only used for ranking.
108fn is_decision_token(value: &str) -> bool {
109    let bytes = value.as_bytes();
110    if bytes.is_empty() || bytes.len() > 7 {
111        return false;
112    }
113    let mut lower = [0u8; 7];
114    for (dst, b) in lower.iter_mut().zip(bytes) {
115        *dst = b.to_ascii_lowercase();
116    }
117    matches!(
118        &lower[..bytes.len()],
119        b"if"
120            | b"elif"
121            | b"elsif"
122            | b"elseif"
123            | b"unless"
124            | b"for"
125            | b"foreach"
126            | b"while"
127            | b"until"
128            | b"case"
129            | b"cond"
130            | b"when"
131            | b"catch"
132            | b"rescue"
133            | b"except"
134            | b"andalso"
135            | b"orelse"
136            | b"&&"
137            | b"||"
138            | b"and"
139            | b"or"
140            | b"?"
141            | b"??"
142    )
143}
144
145/// A synthetic source is the per-sub-format shadow of a multi-format file
146/// (markdown/vue/svelte embedded code); its id is `<parent-id>:<format>` and
147/// its metrics are already covered by the parent entry.
148fn is_synthetic(source: &SourceFile) -> bool {
149    source
150        .id
151        .strip_suffix(source.format.as_str())
152        .is_some_and(|prefix| prefix.ends_with(':'))
153}
154
155fn metric_of(file: &FileSummary, by: SummaryMetric) -> u64 {
156    match by {
157        SummaryMetric::Tokens => file.tokens,
158        SummaryMetric::Lines => file.lines,
159        SummaryMetric::Size => file.bytes,
160        SummaryMetric::Complexity => file.complexity,
161    }
162}
163
164fn folder_metric_of(folder: &FolderSummary, by: SummaryMetric) -> u64 {
165    match by {
166        SummaryMetric::Tokens => folder.tokens,
167        SummaryMetric::Lines => folder.lines,
168        SummaryMetric::Size => folder.bytes,
169        SummaryMetric::Complexity => folder.complexity,
170    }
171}
172
173/// Parent directory of a path, with separators normalized to `/`.
174/// Files at the scan root map to `"."`.
175fn parent_dir(path: &str) -> String {
176    let normalized = path.replace('\\', "/");
177    match normalized.rfind('/') {
178        Some(0) => "/".to_string(),
179        Some(idx) => normalized[..idx].to_string(),
180        None => ".".to_string(),
181    }
182}
183
184/// Compute the summary from detection results.
185///
186/// `display_path` maps a source id (canonical absolute path) to the path shown
187/// in reports — the same relativization applied to clone fragments, so
188/// per-file duplication matching works on identical strings.
189pub fn compute_summary(
190    sources: &[SourceFile],
191    clones: &[CpdClone],
192    top: usize,
193    by: SummaryMetric,
194    display_path: impl Fn(&str) -> String,
195) -> Summary {
196    // Per-file duplication, keyed by display path. Both fragments of a clone
197    // count toward their file: the question here is "where does duplicated
198    // code live", not the de-duplicated total that Statistics reports.
199    let mut dup: HashMap<String, (u64, u64)> = HashMap::new();
200    for clone in clones {
201        for fragment in [&clone.fragment_a, &clone.fragment_b] {
202            // Sub-format fragments carry a `<path>:<format>` id; fold them
203            // into the parent file.
204            let path = fragment
205                .source_id
206                .strip_suffix(&format!(":{}", clone.format))
207                .unwrap_or(&fragment.source_id);
208            let entry = dup.entry(path.to_string()).or_default();
209            entry.0 += fragment.end.line.saturating_sub(fragment.start.line) as u64;
210            entry.1 += clone.token_count as u64;
211        }
212    }
213
214    let mut files: Vec<FileSummary> = sources
215        .iter()
216        .filter(|s| !is_synthetic(s))
217        .map(|source| {
218            let path = display_path(&source.id);
219            // Same line metric as Statistics: max token start line.
220            let lines = source
221                .tokens
222                .iter()
223                .map(|t| t.start.line)
224                .max()
225                .unwrap_or(0) as u64;
226            let decisions = source
227                .tokens
228                .iter()
229                .filter(|t| is_decision_token(&t.value))
230                .count() as u64;
231            let (duplicated_lines, duplicated_tokens) = dup.get(&path).copied().unwrap_or_default();
232            FileSummary {
233                lines,
234                tokens: source.tokens.len() as u64,
235                bytes: source.bytes,
236                duplicated_lines,
237                duplicated_tokens,
238                complexity: 1 + decisions,
239                format: source.format.clone(),
240                path,
241            }
242        })
243        .collect();
244
245    let total_files = files.len() as u64;
246
247    // Folder rollup over ALL files (before top-N truncation).
248    let mut folder_map: HashMap<String, FolderSummary> = HashMap::new();
249    for file in &files {
250        let dir = parent_dir(&file.path);
251        let entry = folder_map
252            .entry(dir.clone())
253            .or_insert_with(|| FolderSummary {
254                path: dir,
255                files: 0,
256                lines: 0,
257                tokens: 0,
258                bytes: 0,
259                duplicated_lines: 0,
260                complexity: 0,
261            });
262        entry.files += 1;
263        entry.lines += file.lines;
264        entry.tokens += file.tokens;
265        entry.bytes += file.bytes;
266        entry.duplicated_lines += file.duplicated_lines;
267        entry.complexity += file.complexity;
268    }
269    let total_folders = folder_map.len() as u64;
270
271    // Top-N files by the primary metric: `--summary-top N` always yields at
272    // most N rows (least surprise). Other lenses are one `--summary-by` away;
273    // every row still carries all metrics.
274    files.sort_by(|a, b| {
275        metric_of(b, by)
276            .cmp(&metric_of(a, by))
277            .then_with(|| a.path.cmp(&b.path))
278    });
279    files.truncate(top);
280
281    let mut folders: Vec<FolderSummary> = folder_map.into_values().collect();
282    folders.sort_by(|a, b| {
283        folder_metric_of(b, by)
284            .cmp(&folder_metric_of(a, by))
285            .then_with(|| a.path.cmp(&b.path))
286    });
287    folders.truncate(top);
288
289    Summary {
290        by,
291        files,
292        folders,
293        total_files,
294        total_folders,
295    }
296}
297
298#[cfg(test)]
299mod tests {
300    use super::*;
301    use crate::models::{CpdClone, Fragment, Location, Token, TokenKind};
302
303    fn loc(line: u32) -> Location {
304        Location {
305            line,
306            column: 0,
307            offset: 0,
308        }
309    }
310
311    fn token(value: &str, line: u32) -> Token {
312        Token {
313            kind: TokenKind::Keyword,
314            value: value.to_string(),
315            start: loc(line),
316            end: loc(line),
317        }
318    }
319
320    fn source(id: &str, format: &str, values: &[&str], bytes: u64) -> SourceFile {
321        SourceFile {
322            id: id.to_string(),
323            format: format.to_string(),
324            tokens: values
325                .iter()
326                .enumerate()
327                .map(|(i, v)| token(v, i as u32 + 1))
328                .collect(),
329            bytes,
330        }
331    }
332
333    fn clone_between(format: &str, a: &str, b: &str, lines: u32, tokens: u32) -> CpdClone {
334        let fragment = |id: &str| Fragment {
335            source_id: id.to_string(),
336            source_root: None,
337            start: loc(1),
338            end: loc(1 + lines),
339            range: [0, tokens],
340            blame: None,
341        };
342        CpdClone {
343            format: format.to_string(),
344            fragment_a: fragment(a),
345            fragment_b: fragment(b),
346            token_count: tokens,
347        }
348    }
349
350    fn identity(path: &str) -> String {
351        path.to_string()
352    }
353
354    #[test]
355    fn empty_input_produces_empty_summary() {
356        let summary = compute_summary(&[], &[], 10, SummaryMetric::Tokens, identity);
357        assert!(summary.files.is_empty());
358        assert!(summary.folders.is_empty());
359        assert_eq!(summary.total_files, 0);
360        assert_eq!(summary.total_folders, 0);
361    }
362
363    #[test]
364    fn files_sorted_by_primary_metric() {
365        let sources = vec![
366            source("src/small.js", "javascript", &["a", "b"], 10),
367            source("src/big.js", "javascript", &["a", "b", "c", "d"], 20),
368        ];
369        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
370        assert_eq!(summary.files[0].path, "src/big.js");
371        assert_eq!(summary.files[0].tokens, 4);
372        assert_eq!(summary.total_files, 2);
373    }
374
375    #[test]
376    fn top_n_is_exact_row_count_by_primary_metric() {
377        // huge.js wins on tokens, fat.js wins on size — top=1 by tokens must
378        // yield exactly one row: huge.js. `--summary-top N` never surprises
379        // with more than N rows; other metrics are served by --summary-by.
380        let sources = vec![
381            source("huge.js", "javascript", &["a", "b", "c", "d", "e"], 1),
382            source("fat.js", "javascript", &["a"], 9999),
383        ];
384        let summary = compute_summary(&sources, &[], 1, SummaryMetric::Tokens, identity);
385        assert_eq!(summary.files.len(), 1);
386        assert_eq!(summary.files[0].path, "huge.js");
387        assert_eq!(summary.total_files, 2, "truncation stays visible");
388
389        let by_size = compute_summary(&sources, &[], 1, SummaryMetric::Size, identity);
390        assert_eq!(by_size.files[0].path, "fat.js");
391    }
392
393    #[test]
394    fn complexity_counts_decision_tokens() {
395        let sources = vec![source(
396            "a.js",
397            "javascript",
398            &["if", "x", "&&", "y", "for", "z", "else"],
399            10,
400        )];
401        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
402        // 1 + (if, &&, for) = 4; "else" is not a decision point.
403        assert_eq!(summary.files[0].complexity, 4);
404    }
405
406    #[test]
407    fn complexity_is_case_insensitive() {
408        // SQL / PL/SQL / Fortran style uppercase keywords.
409        let sources = vec![source(
410            "a.sql",
411            "sql",
412            &["IF", "x", "OR", "y", "WHEN", "THEN", "If"],
413            10,
414        )];
415        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
416        // 1 + (IF, OR, WHEN, If) = 5; THEN is not a decision point.
417        assert_eq!(summary.files[0].complexity, 5);
418    }
419
420    #[test]
421    fn decision_token_edge_cases() {
422        assert!(is_decision_token("unless"));
423        assert!(is_decision_token("ELSEIF"));
424        assert!(is_decision_token("andalso"));
425        assert!(!is_decision_token(""));
426        assert!(!is_decision_token("iffy"));
427        assert!(!is_decision_token("conditionally"), "length-capped");
428        assert!(!is_decision_token("форматирование"), "non-ASCII ignored");
429    }
430
431    #[test]
432    fn folder_rollup_uses_direct_parent() {
433        let sources = vec![
434            source("src/app/a.js", "javascript", &["x"], 5),
435            source("src/app/b.js", "javascript", &["x", "y"], 5),
436            source("src/c.js", "javascript", &["x"], 5),
437            source("root.js", "javascript", &["x"], 5),
438        ];
439        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
440        assert_eq!(summary.total_folders, 3);
441        let app = summary
442            .folders
443            .iter()
444            .find(|f| f.path == "src/app")
445            .expect("src/app folder");
446        assert_eq!(app.files, 2);
447        assert_eq!(app.tokens, 3);
448        let root = summary.folders.iter().find(|f| f.path == ".");
449        assert!(root.is_some(), "root files grouped under '.'");
450    }
451
452    #[test]
453    fn duplication_attributed_to_both_fragments() {
454        let sources = vec![
455            source("a.js", "javascript", &["x", "y", "z"], 5),
456            source("b.js", "javascript", &["x", "y", "z"], 5),
457        ];
458        let clones = vec![clone_between("javascript", "a.js", "b.js", 9, 30)];
459        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, identity);
460        for path in ["a.js", "b.js"] {
461            let file = summary.files.iter().find(|f| f.path == path).unwrap();
462            assert_eq!(file.duplicated_lines, 9, "{path} duplicated lines");
463            assert_eq!(file.duplicated_tokens, 30, "{path} duplicated tokens");
464        }
465    }
466
467    #[test]
468    fn synthetic_sub_format_sources_are_skipped() {
469        let sources = vec![
470            source("doc.md", "markdown", &["x", "y"], 100),
471            source("doc.md:javascript", "javascript", &["x"], 0),
472        ];
473        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
474        assert_eq!(summary.total_files, 1);
475        assert_eq!(summary.files[0].path, "doc.md");
476    }
477
478    #[test]
479    fn sub_format_clone_folds_into_parent_file() {
480        let sources = vec![source("doc.md", "markdown", &["x", "y"], 100)];
481        let clones = vec![clone_between(
482            "javascript",
483            "doc.md:javascript",
484            "doc.md:javascript",
485            4,
486            20,
487        )];
488        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, identity);
489        assert_eq!(
490            summary.files[0].duplicated_lines, 8,
491            "both fragments fold in"
492        );
493    }
494
495    #[test]
496    fn display_path_applied_before_dup_matching() {
497        let sources = vec![source("/abs/root/a.js", "javascript", &["x"], 5)];
498        let clones = vec![clone_between("javascript", "a.js", "a.js", 2, 10)];
499        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, |p| {
500            p.strip_prefix("/abs/root/").unwrap_or(p).to_string()
501        });
502        assert_eq!(summary.files[0].path, "a.js");
503        assert_eq!(summary.files[0].duplicated_lines, 4);
504    }
505
506    #[test]
507    fn folders_truncated_to_top_n_but_total_reported() {
508        let sources: Vec<SourceFile> = (0..5)
509            .map(|i| source(&format!("dir{i}/f.js"), "javascript", &["x"], 1))
510            .collect();
511        let summary = compute_summary(&sources, &[], 2, SummaryMetric::Tokens, identity);
512        assert_eq!(summary.folders.len(), 2);
513        assert_eq!(summary.total_folders, 5);
514    }
515
516    #[test]
517    fn metric_parses_from_str() {
518        assert_eq!(
519            "complexity".parse::<SummaryMetric>().unwrap(),
520            SummaryMetric::Complexity
521        );
522        assert!("bogus".parse::<SummaryMetric>().is_err());
523    }
524
525    #[test]
526    fn summary_serializes_camel_case() {
527        let sources = vec![source("a.js", "javascript", &["x"], 5)];
528        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Size, identity);
529        let json = serde_json::to_string(&summary).unwrap();
530        assert!(json.contains("\"totalFiles\""));
531        assert!(json.contains("\"duplicatedLines\""));
532        assert!(json.contains("\"by\":\"size\""));
533        assert!(!json.contains("total_files"));
534    }
535}