Skip to main content

cpd_core/
summary.rs

1// summary.rs — opt-in codebase summary: per-file metrics, folder rollup, top-N lists.
2//
3// Everything in this module runs only when `--summary` is enabled, after
4// detection has finished, over data already held in memory (SourceFile tokens
5// and detected clones). Nothing in the detection hot path calls into it.
6
7use crate::models::{CpdClone, SourceFile, Token, TokenKind};
8use serde::{Deserialize, Serialize};
9use std::collections::HashMap;
10
11/// Metric used to rank files and folders in the summary.
12#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)]
13#[serde(rename_all = "lowercase")]
14pub enum SummaryMetric {
15    #[default]
16    Tokens,
17    Lines,
18    Size,
19    Complexity,
20}
21
22impl std::str::FromStr for SummaryMetric {
23    type Err = String;
24
25    fn from_str(s: &str) -> Result<Self, Self::Err> {
26        match s {
27            "tokens" => Ok(Self::Tokens),
28            "lines" => Ok(Self::Lines),
29            "size" => Ok(Self::Size),
30            "complexity" => Ok(Self::Complexity),
31            other => Err(format!(
32                "invalid summary metric '{other}': must be one of: tokens, lines, size, complexity"
33            )),
34        }
35    }
36}
37
38impl std::fmt::Display for SummaryMetric {
39    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
40        let s = match self {
41            Self::Tokens => "tokens",
42            Self::Lines => "lines",
43            Self::Size => "size",
44            Self::Complexity => "complexity",
45        };
46        f.write_str(s)
47    }
48}
49
50/// Per-file summary row.
51#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
52#[serde(rename_all = "camelCase")]
53pub struct FileSummary {
54    pub path: String,
55    pub format: String,
56    pub lines: u64,
57    pub tokens: u64,
58    pub bytes: u64,
59    pub duplicated_lines: u64,
60    pub duplicated_tokens: u64,
61    /// Cyclomatic-complexity estimate: 1 + count of decision-point tokens
62    /// (`if`, `for`, `while`, `case`, `catch`, `&&`, `||`, `?`, …).
63    pub complexity: u64,
64}
65
66/// Per-folder rollup. Files are counted in their direct parent directory only
67/// (no cumulative ancestor totals), so every file contributes to exactly one
68/// folder row and rows are directly comparable.
69#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
70#[serde(rename_all = "camelCase")]
71pub struct FolderSummary {
72    pub path: String,
73    pub files: u64,
74    pub lines: u64,
75    pub tokens: u64,
76    pub bytes: u64,
77    pub duplicated_lines: u64,
78    /// Sum of per-file complexity estimates (divide by `files` for the mean).
79    pub complexity: u64,
80}
81
82/// Codebase summary: top files and folder rollup.
83#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
84#[serde(rename_all = "camelCase")]
85pub struct Summary {
86    /// Primary sort metric.
87    pub by: SummaryMetric,
88    /// Top-N files by `by`, descending. Every row carries all metrics
89    /// (tokens, lines, bytes, complexity, duplication) so one list serves
90    /// every lens; re-run with a different `--summary-by` to re-rank.
91    pub files: Vec<FileSummary>,
92    /// Top-N folders by `by`, direct-parent aggregation.
93    pub folders: Vec<FolderSummary>,
94    /// Total number of files analyzed (before top-N truncation).
95    pub total_files: u64,
96    /// Total number of folders (before top-N truncation).
97    pub total_folders: u64,
98}
99
100/// Decision-point tokens counted by the complexity estimate. Conservative,
101/// language-agnostic list: branch/loop keywords and short-circuit operators
102/// that appear as standalone tokens across supported languages.
103///
104/// Matching is ASCII-case-insensitive so case-insensitive and
105/// uppercase-keyword languages (SQL, PL/SQL, Fortran, COBOL, BASIC, Pascal)
106/// count too. The occasional identifier spelled like a keyword slightly
107/// inflates an estimate that is only used for ranking.
108///
109/// Operators reach this function already joined — see [`joined_token`]. Only
110/// the JavaScript tokenizer emits `&&` as one token; the generic one splits
111/// every punctuation run into single characters, so without that joining the
112/// short-circuit arms here would be unreachable for every other format.
113fn is_decision_token(value: &str) -> bool {
114    let bytes = value.as_bytes();
115    if bytes.is_empty() || bytes.len() > 7 {
116        return false;
117    }
118    let mut lower = [0u8; 7];
119    for (dst, b) in lower.iter_mut().zip(bytes) {
120        *dst = b.to_ascii_lowercase();
121    }
122    matches!(
123        &lower[..bytes.len()],
124        b"if"
125            | b"elif"
126            | b"elsif"
127            | b"elseif"
128            | b"unless"
129            | b"for"
130            | b"foreach"
131            | b"while"
132            | b"until"
133            | b"case"
134            | b"cond"
135            | b"when"
136            | b"catch"
137            | b"rescue"
138            | b"except"
139            | b"andalso"
140            | b"orelse"
141            | b"&&"
142            | b"||"
143            | b"and"
144            | b"or"
145            | b"?"
146            | b"??"
147    )
148}
149
150/// What a bare `?` means in a language.
151#[derive(Clone, Copy, PartialEq, Eq)]
152enum QuestionMark {
153    /// It only ever opens a ternary: C, Java, PHP, Go templates, and most else.
154    Ternary,
155    /// It also marks an optional, so `String?` and `x?.y` are types and
156    /// accesses rather than branches. Neither TypeScript nor Swift has an
157    /// Elvis operator, so `?:` there is an optional property, not a branch.
158    Optional,
159    /// As above, but `?:` *is* the Elvis operator and does branch: Kotlin,
160    /// Groovy.
161    OptionalWithElvis,
162}
163
164/// How one language spells its branches, beyond the shared keyword set.
165///
166/// The shared set is right for most of the formats jscpd knows. An entry here
167/// exists only where a language branches on something the set has no word for
168/// (`match` arms in Rust, `guard` in Swift, `select` in Go) or spells
169/// something in the set so differently that counting it is simply wrong.
170#[derive(Clone, Copy)]
171struct DecisionRules {
172    /// Branch tokens beyond the shared set, matched after joining.
173    extra: &'static [&'static str],
174    question: QuestionMark,
175    /// Tokens that open a function body. Cyclomatic complexity is one path per
176    /// function; where a language has no single reliable marker this stays
177    /// empty and the estimate keeps its per-file baseline of one.
178    declarations: &'static [&'static str],
179    /// The C family declares a function as `head(args) {` with no keyword at
180    /// all, so its functions are counted from that shape instead.
181    braced_declarations: bool,
182    /// Whether `"""` and `'''` delimit a string that can span lines; see
183    /// [`has_triple_quoted_strings`].
184    triple_quoted_strings: bool,
185    /// Tokens that open a group of arms counted one by one through `extra`.
186    /// Each cancels one arm, because N arms are N paths and so N - 1 branches,
187    /// the same way a `switch` counts its `case` labels but not its `default`.
188    arm_groups: &'static [&'static str],
189}
190
191impl Default for DecisionRules {
192    fn default() -> Self {
193        Self {
194            extra: &[],
195            question: QuestionMark::Ternary,
196            declarations: &[],
197            braced_declarations: false,
198            triple_quoted_strings: false,
199            arm_groups: &[],
200        }
201    }
202}
203
204/// Formats whose strings can span lines between `"""` or `'''` delimiters.
205///
206/// Where a doubled quote is how a quote is escaped — C# verbatim strings,
207/// VB.NET, SQL, Pascal — three quotes in a row are ordinary string content: the
208/// regex `@"""((?:\\.|[^""\\])*)"""` is one line of C#. Reading such a run as
209/// a delimiter opens a string that never closes and swallows the rest of the
210/// file, which is why this is a list rather than a default.
211fn has_triple_quoted_strings(format: &str) -> bool {
212    matches!(
213        format,
214        "python" | "kotlin" | "scala" | "groovy" | "swift" | "java" | "julia" | "elixir" | "dart"
215    )
216}
217
218/// False for prose and data formats, whose "if" and `||` are words and
219/// version ranges rather than branches. The same test decides whether a
220/// format counts as the project's code at all — in complexity here, and in
221/// `health::compute`'s and a format-level duplication breakdown's "code
222/// files" — since a format that can never have a branch can never have
223/// complexity above zero either way.
224pub fn has_control_flow(format: &str) -> bool {
225    !matches!(
226        format,
227        "markdown"
228            | "asciidoc"
229            | "rest"
230            | "textile"
231            | "wiki"
232            | "txt"
233            | "log"
234            | "csv"
235            | "json"
236            | "json5"
237            | "yaml"
238            | "toml"
239            | "ini"
240            | "properties"
241            | "editorconfig"
242            | "ignore"
243            | "diff"
244            | "gettext"
245    )
246}
247
248fn rules_for(format: &str) -> DecisionRules {
249    DecisionRules {
250        triple_quoted_strings: has_triple_quoted_strings(format),
251        ..language_rules(format)
252    }
253}
254
255fn language_rules(format: &str) -> DecisionRules {
256    match format {
257        // A `match` has no keyword per arm, only the `=>` each one is written
258        // with. Every arm is counted and the `match` itself takes one back: a
259        // three-arm match is three paths, which is two branches.
260        "rust" => DecisionRules {
261            extra: &["=>"],
262            declarations: &["fn"],
263            arm_groups: &["match"],
264            ..Default::default()
265        },
266        "swift" => DecisionRules {
267            extra: &["guard"],
268            question: QuestionMark::Optional,
269            declarations: &["func"],
270            ..Default::default()
271        },
272        // `=>` opens an arrow function here rather than a branch, so it counts
273        // toward the function tally instead.
274        "typescript" | "tsx" | "flow" | "javascript" | "jsx" => DecisionRules {
275            question: QuestionMark::Optional,
276            declarations: &["function", "=>"],
277            ..Default::default()
278        },
279        "kotlin" => DecisionRules {
280            question: QuestionMark::OptionalWithElvis,
281            declarations: &["fun"],
282            ..Default::default()
283        },
284        "groovy" => DecisionRules {
285            question: QuestionMark::OptionalWithElvis,
286            declarations: &["def"],
287            ..Default::default()
288        },
289        "csharp" => DecisionRules {
290            question: QuestionMark::Optional,
291            braced_declarations: true,
292            ..Default::default()
293        },
294        "c" | "c-header" | "cpp" | "cpp-header" | "java" | "objectivec" | "clike" => {
295            DecisionRules {
296                braced_declarations: true,
297                ..Default::default()
298            }
299        }
300        "go" => DecisionRules {
301            extra: &["select"],
302            declarations: &["func"],
303            ..Default::default()
304        },
305        "scala" | "python" | "ruby" | "crystal" => DecisionRules {
306            declarations: &["def"],
307            ..Default::default()
308        },
309        "php" | "lua" => DecisionRules {
310            declarations: &["function"],
311            ..Default::default()
312        },
313        "erlang" => DecisionRules {
314            extra: &["receive"],
315            ..Default::default()
316        },
317        _ => DecisionRules::default(),
318    }
319}
320
321/// Two-character operators the generic tokenizer hands over as two tokens.
322const JOINED_OPERATORS: &[&str] = &["&&", "||", "??", "?.", "?:", "=>"];
323
324/// One token's text, or the two-character operator that two adjacent tokens
325/// spell between them.
326///
327/// The pair cannot borrow: each token owns its own `String`, so two adjacent
328/// characters of the source are not adjacent in memory. Two bytes on the stack
329/// avoid an allocation per operator.
330enum Scanned<'a> {
331    Single(&'a str),
332    Pair([u8; 2]),
333}
334
335impl Scanned<'_> {
336    fn text(&self) -> &str {
337        match self {
338            Self::Single(text) => text,
339            Self::Pair(bytes) => std::str::from_utf8(bytes).unwrap_or(""),
340        }
341    }
342}
343
344/// The token at `at`, joined with the next one when the two are adjacent
345/// single-character punctuation spelling a two-character operator.
346///
347/// The generic tokenizer splits `&&` into two `&` tokens, so a scan that looks
348/// at one token at a time can never see a short-circuit operator at all.
349/// Returns what to classify and the index to continue from.
350fn joined_token(tokens: &[Token], at: usize) -> (Scanned<'_>, usize) {
351    let current = &tokens[at];
352    let single = (Scanned::Single(current.value.as_str()), at + 1);
353    let Some(next) = tokens.get(at + 1) else {
354        return single;
355    };
356    let (Some(&left), Some(&right)) = (
357        current.value.as_bytes().first(),
358        next.value.as_bytes().first(),
359    ) else {
360        return single;
361    };
362    // Adjacent in the source, and both exactly one punctuation character.
363    if current.end.offset != next.start.offset
364        || current.value.len() != 1
365        || next.value.len() != 1
366        || !left.is_ascii_punctuation()
367        || !right.is_ascii_punctuation()
368    {
369        return single;
370    }
371    let pair = [left, right];
372    match std::str::from_utf8(&pair).is_ok_and(|text| JOINED_OPERATORS.contains(&text)) {
373        true => (Scanned::Pair(pair), at + 2),
374        false => single,
375    }
376}
377
378/// Which tokens lie inside a triple-quoted string.
379///
380/// The tokenizer is line-based: it closes every string at the end of the line
381/// it started on, so the body of a `"""` string arrives as ordinary words, and
382/// a docstring saying "if the value is big" lands three branches on the file.
383///
384/// The quotes themselves survive as literal tokens, and a run of adjacent
385/// literals spells exactly the quote characters its line holds: a PEP 257
386/// opener `"""Summary.` arrives as `""` and `"Summary.`, a closer on its own
387/// line as `""` and `"`, and a one-line docstring as `""`, `"One."`, `""`. So
388/// an odd number of triple quotes in a run opens or closes the string, and an
389/// even number leaves it as it was.
390///
391/// Reading a pair of tokens alone is not enough, and was the first version of
392/// this: with the summary written on the opening line the opener is not a bare
393/// quote, so the *closing* `"""` looked like an opener and swallowed every
394/// branch that followed it.
395fn triple_quoted(tokens: &[Token]) -> Vec<bool> {
396    const DELIMITERS: [&str; 2] = ["\"\"\"", "'''"];
397    let mut inside = vec![false; tokens.len()];
398    let mut open: Option<&str> = None;
399    let mut at = 0usize;
400    while at < tokens.len() {
401        if tokens[at].kind != TokenKind::Literal {
402            inside[at] = open.is_some();
403            at += 1;
404            continue;
405        }
406        let start = at;
407        let mut text = tokens[at].value.clone();
408        at += 1;
409        while at < tokens.len()
410            && tokens[at].kind == TokenKind::Literal
411            && tokens[at - 1].end.offset == tokens[at].start.offset
412        {
413            text.push_str(&tokens[at].value);
414            at += 1;
415        }
416        let was_open = open.is_some();
417        for delimiter in DELIMITERS {
418            let toggles = text.matches(delimiter).count() % 2 == 1;
419            match open {
420                Some(current) if current == delimiter && toggles => open = None,
421                None if toggles => open = Some(delimiter),
422                _ => {}
423            }
424        }
425        // The run is string text whichever way it turned the state.
426        inside[start..at].fill(was_open || open.is_some());
427    }
428    inside
429}
430
431/// Words a file binds as names of its own.
432///
433/// `case`, `when` and `cond` are branch keywords in some languages and
434/// perfectly ordinary variable names in others — the shared list cannot tell
435/// them apart, and the tokenizer is no help: its `classify_word` returns
436/// `Identifier` for every word, so `if` in Python is tagged exactly like a
437/// variable called `case`.
438///
439/// What a file *does* reveal is which words it assigns to, reads off an
440/// object, or lists as a parameter or argument. A word this file writes
441/// `case = 3`, `x.case` or `f(case, when)` for is that file's own name,
442/// whatever the language reserves, and counting it as a branch is what
443/// inflated the estimate elevenfold on a file of five such assignments.
444fn locally_bound(tokens: &[Token]) -> Vec<&str> {
445    let mut bound = Vec::new();
446    for (at, token) in tokens.iter().enumerate() {
447        if !is_decision_token(&token.value) {
448            continue;
449        }
450        // `obj.case` — a member, never the keyword.
451        let after_dot = at
452            .checked_sub(1)
453            .is_some_and(|previous| tokens[previous].value == ".");
454        // `case = 3`, but not `case == 3` or `case => 3`.
455        let assigned = tokens.get(at + 1).is_some_and(|next| next.value == "=")
456            && tokens
457                .get(at + 2)
458                .is_none_or(|after| !matches!(after.value.as_str(), "=" | ">"));
459        // `def headline(case, when)` or `headline(case, when)`: a keyword is
460        // never written directly before a comma or a closing paren, a name in
461        // a parameter or argument list always is. Words only — `Ok(parse()?)`
462        // is Rust's `?` doing its job, not a name.
463        let listed = token.value.starts_with(|c: char| c.is_ascii_alphabetic())
464            && tokens
465                .get(at + 1)
466                .is_some_and(|next| matches!(next.value.as_str(), "," | ")"));
467        if after_dot || assigned || listed {
468            bound.push(token.value.as_str());
469        }
470    }
471    bound.sort_unstable();
472    bound.dedup();
473    bound
474}
475
476/// Keywords that take a parenthesised head and a block, so that `) {` after
477/// one of them opens a branch rather than a function body.
478const PARENTHESISED_STATEMENTS: &[&str] = &[
479    "switch",
480    "using",
481    "lock",
482    "synchronized",
483    "with",
484    "do",
485    "try",
486    "return",
487    "sizeof",
488    "typeof",
489    "new",
490    "throw",
491    "await",
492    "yield",
493    "fixed",
494    "unsafe",
495];
496
497/// Decision points and function count for one file.
498///
499/// Cyclomatic complexity is one path per function plus one per branch. Where a
500/// language has no reliable function marker the tally falls back to the
501/// per-file baseline of one, which is what this estimate has always used.
502fn scan_complexity(tokens: &[Token], rules: &DecisionRules) -> (u64, u64) {
503    let (mut decisions, mut functions, mut groups) = (0u64, 0u64, 0u64);
504    let mut at = 0usize;
505    let in_string = match rules.triple_quoted_strings {
506        true => triple_quoted(tokens),
507        false => vec![false; tokens.len()],
508    };
509    let bound = locally_bound(tokens);
510    // What the token before each open paren was, so a `) {` can be told from
511    // the head it closes: a function signature, or `if (…) {`.
512    let mut heads: Vec<&str> = Vec::new();
513    let mut last_head: Option<&str> = None;
514    while at < tokens.len() {
515        // Skip the body of a string that spans lines; see `triple_quoted`.
516        if in_string[at] {
517            at += 1;
518            continue;
519        }
520        let (scanned, next) = joined_token(tokens, at);
521        let text = scanned.text();
522        // `String?` writes the optional against the type it belongs to;
523        // `cond ? a : b` puts the ternary in the open. Nothing else separates
524        // the two without parsing, and the convention is near-universal.
525        let attached = at
526            .checked_sub(1)
527            .and_then(|previous| tokens.get(previous))
528            .is_some_and(|previous| previous.end.offset == tokens[at].start.offset);
529        at = next;
530        // C, C++, Java and C# open a function with `) {` and no keyword of
531        // their own. Tracking what preceded each `(` is enough to tell that
532        // from the `) {` of an `if` or a `switch`, and it is the only reason
533        // those languages kept a per-file baseline.
534        if rules.braced_declarations {
535            match text {
536                "(" => {
537                    let head = at
538                        .checked_sub(2)
539                        .and_then(|before| tokens.get(before))
540                        .map(|token| token.value.as_str())
541                        .unwrap_or("");
542                    heads.push(head);
543                }
544                ")" => last_head = heads.pop(),
545                "{" => {
546                    if let Some(head) = last_head.take()
547                        && !head.is_empty()
548                        && head.chars().all(|c| c.is_alphanumeric() || c == '_')
549                        && !is_decision_token(head)
550                        && !PARENTHESISED_STATEMENTS
551                            .iter()
552                            .any(|s| s.eq_ignore_ascii_case(head))
553                    {
554                        functions += 1;
555                    }
556                }
557                _ => last_head = None,
558            }
559        }
560        if rules
561            .declarations
562            .iter()
563            .any(|d| d.eq_ignore_ascii_case(text))
564        {
565            functions += 1;
566            continue;
567        }
568        if rules
569            .arm_groups
570            .iter()
571            .any(|g| g.eq_ignore_ascii_case(text))
572        {
573            groups += 1;
574            continue;
575        }
576        if rules.extra.iter().any(|e| e.eq_ignore_ascii_case(text)) {
577            decisions += 1;
578            continue;
579        }
580        // A word this file binds as a name is that file's name, not a keyword.
581        if bound.binary_search(&text).is_ok() {
582            continue;
583        }
584        let counts = match text {
585            // `?.` never branches, and `?:` branches only where it is the
586            // Elvis operator rather than an optional property.
587            "?." => false,
588            "?:" => rules.question != QuestionMark::Optional,
589            "?" => rules.question == QuestionMark::Ternary || !attached,
590            other => is_decision_token(other),
591        };
592        if counts {
593            decisions += 1;
594        }
595    }
596    (decisions.saturating_sub(groups), functions)
597}
598
599/// A synthetic source is the per-sub-format shadow of a multi-format file
600/// (markdown/vue/svelte embedded code); its id is `<parent-id>:<format>` and
601/// its metrics are already covered by the parent entry.
602fn is_synthetic(source: &SourceFile) -> bool {
603    source
604        .id
605        .strip_suffix(source.format.as_str())
606        .is_some_and(|prefix| prefix.ends_with(':'))
607}
608
609fn metric_of(file: &FileSummary, by: SummaryMetric) -> u64 {
610    match by {
611        SummaryMetric::Tokens => file.tokens,
612        SummaryMetric::Lines => file.lines,
613        SummaryMetric::Size => file.bytes,
614        SummaryMetric::Complexity => file.complexity,
615    }
616}
617
618fn folder_metric_of(folder: &FolderSummary, by: SummaryMetric) -> u64 {
619    match by {
620        SummaryMetric::Tokens => folder.tokens,
621        SummaryMetric::Lines => folder.lines,
622        SummaryMetric::Size => folder.bytes,
623        SummaryMetric::Complexity => folder.complexity,
624    }
625}
626
627/// Parent directory of a path, with separators normalized to `/`.
628/// Files at the scan root map to `"."`.
629fn parent_dir(path: &str) -> String {
630    let normalized = path.replace('\\', "/");
631    match normalized.rfind('/') {
632        Some(0) => "/".to_string(),
633        Some(idx) => normalized[..idx].to_string(),
634        None => ".".to_string(),
635    }
636}
637
638/// Compute the summary from detection results.
639///
640/// `display_path` maps a source id (canonical absolute path) to the path shown
641/// in reports — the same relativization applied to clone fragments, so
642/// per-file duplication matching works on identical strings.
643pub fn compute_summary(
644    sources: &[SourceFile],
645    clones: &[CpdClone],
646    top: usize,
647    by: SummaryMetric,
648    display_path: impl Fn(&str) -> String,
649) -> Summary {
650    // Per-file duplication, keyed by display path. Both fragments of a clone
651    // count toward their file: the question here is "where does duplicated
652    // code live", not the de-duplicated total that Statistics reports.
653    let mut dup: HashMap<String, (u64, u64)> = HashMap::new();
654    for clone in clones {
655        for (fragment, unmatched) in [
656            (&clone.fragment_a, clone.unmatched_lines[0]),
657            (&clone.fragment_b, clone.unmatched_lines[1]),
658        ] {
659            // Sub-format fragments carry a `<path>:<format>` id; fold them
660            // into the parent file.
661            let path = fragment
662                .source_id
663                .strip_suffix(&format!(":{}", clone.format))
664                .unwrap_or(&fragment.source_id);
665            let entry = dup.entry(path.to_string()).or_default();
666            // Gap lines of a merged clone are not duplicated code.
667            entry.0 += fragment
668                .end
669                .line
670                .saturating_sub(fragment.start.line)
671                .saturating_sub(unmatched) as u64;
672            entry.1 += clone.token_count as u64;
673        }
674    }
675
676    let mut files: Vec<FileSummary> = sources
677        .iter()
678        .filter(|s| !is_synthetic(s))
679        .map(|source| {
680            let path = display_path(&source.id);
681            // Same line metric as Statistics: max token start line.
682            let lines = source
683                .tokens
684                .iter()
685                .map(|t| t.start.line)
686                .max()
687                .unwrap_or(0) as u64;
688            let (decisions, functions) =
689                scan_complexity(&source.tokens, &rules_for(&source.format));
690            let (duplicated_lines, duplicated_tokens) = dup.get(&path).copied().unwrap_or_default();
691            FileSummary {
692                lines,
693                tokens: source.tokens.len() as u64,
694                bytes: source.bytes,
695                duplicated_lines,
696                duplicated_tokens,
697                // One path per function, or the per-file baseline where the
698                // language has no marker the scan can trust. Prose and data
699                // have no paths: an "if" in a README is a word, and a lock
700                // file full of `||` version ranges is not code.
701                complexity: match has_control_flow(&source.format) {
702                    true => functions.max(1) + decisions,
703                    false => 0,
704                },
705                format: source.format.clone(),
706                path,
707            }
708        })
709        .collect();
710
711    let total_files = files.len() as u64;
712
713    // Folder rollup over ALL files (before top-N truncation).
714    let mut folder_map: HashMap<String, FolderSummary> = HashMap::new();
715    for file in &files {
716        let dir = parent_dir(&file.path);
717        let entry = folder_map
718            .entry(dir.clone())
719            .or_insert_with(|| FolderSummary {
720                path: dir,
721                files: 0,
722                lines: 0,
723                tokens: 0,
724                bytes: 0,
725                duplicated_lines: 0,
726                complexity: 0,
727            });
728        entry.files += 1;
729        entry.lines += file.lines;
730        entry.tokens += file.tokens;
731        entry.bytes += file.bytes;
732        entry.duplicated_lines += file.duplicated_lines;
733        entry.complexity += file.complexity;
734    }
735    let total_folders = folder_map.len() as u64;
736
737    // Top-N files by the primary metric: `--summary-top N` always yields at
738    // most N rows (least surprise). Other lenses are one `--summary-by` away;
739    // every row still carries all metrics.
740    files.sort_by(|a, b| {
741        metric_of(b, by)
742            .cmp(&metric_of(a, by))
743            .then_with(|| a.path.cmp(&b.path))
744    });
745    files.truncate(top);
746
747    let mut folders: Vec<FolderSummary> = folder_map.into_values().collect();
748    folders.sort_by(|a, b| {
749        folder_metric_of(b, by)
750            .cmp(&folder_metric_of(a, by))
751            .then_with(|| a.path.cmp(&b.path))
752    });
753    folders.truncate(top);
754
755    Summary {
756        by,
757        files,
758        folders,
759        total_files,
760        total_folders,
761    }
762}
763
764#[cfg(test)]
765mod tests {
766    use super::*;
767    use crate::models::{CpdClone, Fragment, Location, Token, TokenKind};
768
769    fn loc(line: u32) -> Location {
770        Location {
771            line,
772            column: 0,
773            offset: 0,
774        }
775    }
776
777    fn token(value: &str, line: u32) -> Token {
778        Token {
779            kind: TokenKind::Keyword,
780            value: value.to_string(),
781            start: loc(line),
782            end: loc(line),
783        }
784    }
785
786    fn source(id: &str, format: &str, values: &[&str], bytes: u64) -> SourceFile {
787        SourceFile {
788            id: id.to_string(),
789            format: format.to_string(),
790            tokens: values
791                .iter()
792                .enumerate()
793                .map(|(i, v)| token(v, i as u32 + 1))
794                .collect(),
795            bytes,
796        }
797    }
798
799    fn clone_between(format: &str, a: &str, b: &str, lines: u32, tokens: u32) -> CpdClone {
800        let fragment = |id: &str| Fragment {
801            source_id: id.to_string(),
802            source_root: None,
803            start: loc(1),
804            end: loc(1 + lines),
805            range: [0, tokens],
806            blame: None,
807        };
808        CpdClone {
809            format: format.to_string(),
810            fragment_a: fragment(a),
811            fragment_b: fragment(b),
812            token_count: tokens,
813            is_new: false,
814            kind: Default::default(),
815            similarity: None,
816            similarity_method: None,
817            unmatched_lines: [0, 0],
818        }
819    }
820
821    fn identity(path: &str) -> String {
822        path.to_string()
823    }
824
825    #[test]
826    fn empty_input_produces_empty_summary() {
827        let summary = compute_summary(&[], &[], 10, SummaryMetric::Tokens, identity);
828        assert!(summary.files.is_empty());
829        assert!(summary.folders.is_empty());
830        assert_eq!(summary.total_files, 0);
831        assert_eq!(summary.total_folders, 0);
832    }
833
834    #[test]
835    fn prose_and_data_have_no_complexity() {
836        let words = [
837            "If", "you", "need", "it", "or", "while", "waiting", "for", "a", "case",
838        ];
839        let sources = vec![
840            source("README.md", "markdown", &words, 10),
841            source(
842                "pnpm-lock.yaml",
843                "yaml",
844                &["version", ":", "^1", "||", "^2"],
845                10,
846            ),
847            source("guide.rst", "rest", &words, 10),
848            source("notes.py", "python", &words, 10),
849        ];
850        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
851        let cx = |path: &str| {
852            summary
853                .files
854                .iter()
855                .find(|f| f.path == path)
856                .unwrap()
857                .complexity
858        };
859        assert_eq!(cx("README.md"), 0, "a word is not a branch");
860        assert_eq!(cx("pnpm-lock.yaml"), 0, "a version range is not a branch");
861        assert_eq!(cx("guide.rst"), 0, "reStructuredText is prose too");
862        assert!(cx("notes.py") > 1, "the same words in code still count");
863    }
864
865    #[test]
866    fn files_sorted_by_primary_metric() {
867        let sources = vec![
868            source("src/small.js", "javascript", &["a", "b"], 10),
869            source("src/big.js", "javascript", &["a", "b", "c", "d"], 20),
870        ];
871        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
872        assert_eq!(summary.files[0].path, "src/big.js");
873        assert_eq!(summary.files[0].tokens, 4);
874        assert_eq!(summary.total_files, 2);
875    }
876
877    #[test]
878    fn top_n_is_exact_row_count_by_primary_metric() {
879        // huge.js wins on tokens, fat.js wins on size — top=1 by tokens must
880        // yield exactly one row: huge.js. `--summary-top N` never surprises
881        // with more than N rows; other metrics are served by --summary-by.
882        let sources = vec![
883            source("huge.js", "javascript", &["a", "b", "c", "d", "e"], 1),
884            source("fat.js", "javascript", &["a"], 9999),
885        ];
886        let summary = compute_summary(&sources, &[], 1, SummaryMetric::Tokens, identity);
887        assert_eq!(summary.files.len(), 1);
888        assert_eq!(summary.files[0].path, "huge.js");
889        assert_eq!(summary.total_files, 2, "truncation stays visible");
890
891        let by_size = compute_summary(&sources, &[], 1, SummaryMetric::Size, identity);
892        assert_eq!(by_size.files[0].path, "fat.js");
893    }
894
895    /// Tokens laid out over a real source string the way the generic
896    /// tokenizer hands them over: one per word, one per punctuation
897    /// character, with the offsets that make adjacency visible.
898    fn lay_out(id: &str, format: &str, code: &str) -> SourceFile {
899        let mut tokens = Vec::new();
900        let bytes = code.as_bytes();
901        let mut at = 0usize;
902        while at < bytes.len() {
903            let byte = bytes[at];
904            if byte.is_ascii_whitespace() {
905                at += 1;
906                continue;
907            }
908            let start = at;
909            if byte.is_ascii_alphanumeric() || byte == b'_' {
910                while at < bytes.len() && (bytes[at].is_ascii_alphanumeric() || bytes[at] == b'_') {
911                    at += 1;
912                }
913            } else {
914                at += 1;
915            }
916            let at32 = |offset: usize| Location {
917                line: 1,
918                column: offset as u32,
919                offset: offset as u32,
920            };
921            tokens.push(Token {
922                kind: TokenKind::Identifier,
923                value: code[start..at].to_string(),
924                start: at32(start),
925                end: at32(at),
926            });
927        }
928        SourceFile {
929            id: id.to_string(),
930            format: format.to_string(),
931            tokens,
932            bytes: code.len() as u64,
933        }
934    }
935
936    fn cx(format: &str, code: &str) -> u64 {
937        let sources = vec![lay_out("a", format, code)];
938        compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity).files[0].complexity
939    }
940
941    #[test]
942    fn short_circuit_operators_count_when_split_into_characters() {
943        // The generic tokenizer hands `&&` over as two `&` tokens, so without
944        // joining them the short-circuit arms of the shared list never match
945        // for any format but JavaScript.
946        assert_eq!(cx("c", "int f() { return a && b; }"), 2);
947        assert_eq!(cx("c", "int f() { return a || b; }"), 2);
948        assert_eq!(cx("c", "int f() { return a && b || c; }"), 3);
949        // A single `&` is a bitwise and, not a branch.
950        assert_eq!(cx("c", "int f() { return a & b; }"), 1);
951    }
952
953    #[test]
954    fn an_optional_is_not_a_ternary() {
955        // `String?` writes the question mark against its type; a ternary puts
956        // it in the open. Nothing else tells them apart without parsing.
957        assert_eq!(cx("swift", "func f(x: String?) -> Int { return 1 }"), 1);
958        assert_eq!(cx("swift", "func f(x: Foo) { let y = x?.bar }"), 1);
959        assert_eq!(
960            cx("swift", "func f(x: Int) -> Int { return x > 0 ? 1 : 2 }"),
961            2
962        );
963        // Nil-coalescing is a branch in any spelling.
964        assert_eq!(
965            cx("swift", "func f(a: Int?, b: Int) -> Int { return a ?? b }"),
966            2
967        );
968        // A language where `?` only ever opens a ternary is unaffected.
969        assert_eq!(cx("c", "int f(int x) { return x ? 1 : 2; }"), 2);
970    }
971
972    #[test]
973    fn a_match_arm_is_the_branch_not_the_match() {
974        // Three arms are three paths: one for the function, two branches.
975        let three_arms = "fn f(x: i32) -> i32 { match x { 0 => 1, 1 => 2, _ => 3 } }";
976        assert_eq!(cx("rust", three_arms), 3);
977        // A guard is a branch of its own on top of the arm it guards.
978        let guarded = "fn f(x: i32, ok: bool) -> i32 { match x { 0 if ok => 1, 0 => 2, _ => 3 } }";
979        assert_eq!(cx("rust", guarded), 4);
980        // A match with a single arm does not branch at all.
981        assert_eq!(cx("rust", "fn f(x: i32) -> i32 { match x { _ => 0 } }"), 1);
982        // `=>` opens an arrow function in JavaScript, so it is a declaration
983        // there rather than a branch.
984        assert_eq!(cx("javascript", "const f = (x) => x + 1;"), 1);
985    }
986
987    #[test]
988    fn complexity_counts_one_path_per_function() {
989        // Four functions, three of them with a single branch.
990        let code = "def a(n):\n if n: pass\ndef b(n):\n if n: pass\ndef c(n):\n if n: pass\ndef d(n):\n pass\n";
991        assert_eq!(cx("python", code), 7);
992        // A language with no marker the scan can trust keeps the per-file
993        // baseline of one rather than guessing.
994        assert_eq!(cx("cobol", "IF x THEN y"), 2);
995    }
996
997    #[test]
998    fn guard_and_select_are_branches_where_they_exist() {
999        assert_eq!(
1000            cx("swift", "func f(x: Int) { guard x > 0 else { return } }"),
1001            2
1002        );
1003        assert_eq!(cx("go", "func f() { select { } }"), 2);
1004        // `guard` is an ordinary word elsewhere.
1005        assert_eq!(cx("c", "int f() { int guard = 1; return guard; }"), 1);
1006    }
1007
1008    #[test]
1009    fn elvis_branches_only_where_the_language_has_one() {
1010        // Kotlin's `?:` is the elvis operator.
1011        assert_eq!(
1012            cx("kotlin", "fun f(a: Int?, b: Int): Int { return a ?: b }"),
1013            2
1014        );
1015        // TypeScript has no elvis; `a?: T` marks an optional property.
1016        assert_eq!(cx("typescript", "function f(a?: string) { return a; }"), 1);
1017    }
1018
1019    #[test]
1020    fn a_docstring_is_not_a_pile_of_branches() {
1021        // The real tokenizer closes every string at the end of its line, so a
1022        // docstring leaves a literal holding just its opening quote and its
1023        // body arrives as ordinary words.
1024        let quote = "\u{22}";
1025        let tokens = vec![
1026            lit_token("def", 0, 3, TokenKind::Identifier),
1027            lit_token("f", 4, 5, TokenKind::Identifier),
1028            lit_token(&quote.repeat(2), 6, 8, TokenKind::Literal),
1029            lit_token(quote, 8, 9, TokenKind::Literal),
1030            lit_token("if", 10, 12, TokenKind::Identifier),
1031            lit_token("for", 13, 16, TokenKind::Identifier),
1032            lit_token("while", 17, 22, TokenKind::Identifier),
1033            lit_token(&quote.repeat(2), 23, 25, TokenKind::Literal),
1034            lit_token(quote, 25, 26, TokenKind::Literal),
1035            lit_token("if", 27, 29, TokenKind::Identifier),
1036        ];
1037        let sources = vec![SourceFile {
1038            id: "a".into(),
1039            format: "python".into(),
1040            tokens,
1041            bytes: 30,
1042        }];
1043        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1044        // One `def`, and only the `if` outside the docstring.
1045        assert_eq!(summary.files[0].complexity, 2);
1046    }
1047
1048    #[test]
1049    fn a_docstring_with_its_summary_on_the_opening_line_closes_where_it_ends() {
1050        // PEP 257 style, as the tokenizer really hands it over: the opener is
1051        // `""` + `"Summary.` rather than a bare quote, so a reader that looked
1052        // for `""` + `"` alone took the *closing* quotes for an opener and
1053        // swallowed every branch after them.
1054        let q = "\u{22}";
1055        let tokens = vec![
1056            lit_token("def", 0, 3, TokenKind::Identifier),
1057            lit_token("f", 4, 5, TokenKind::Identifier),
1058            lit_token(&q.repeat(2), 10, 12, TokenKind::Literal),
1059            lit_token(&format!("{q}Summary."), 12, 21, TokenKind::Literal),
1060            lit_token("if", 30, 32, TokenKind::Identifier),
1061            lit_token("and", 33, 36, TokenKind::Identifier),
1062            lit_token(&q.repeat(2), 40, 42, TokenKind::Literal),
1063            lit_token(q, 42, 43, TokenKind::Literal),
1064            lit_token("if", 50, 52, TokenKind::Identifier),
1065            lit_token("x", 53, 54, TokenKind::Identifier),
1066            // A one-line docstring is balanced on its own line.
1067            lit_token(&q.repeat(2), 60, 62, TokenKind::Literal),
1068            lit_token(&format!("{q}One.{q}"), 62, 68, TokenKind::Literal),
1069            lit_token(&q.repeat(2), 68, 70, TokenKind::Literal),
1070            lit_token("for", 75, 78, TokenKind::Identifier),
1071        ];
1072        let sources = vec![SourceFile {
1073            id: "a".into(),
1074            format: "python".into(),
1075            tokens,
1076            bytes: 80,
1077        }];
1078        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1079        // One `def`, plus the `if` and the `for` written as code.
1080        assert_eq!(summary.files[0].complexity, 3);
1081    }
1082
1083    #[test]
1084    fn a_doubled_quote_escape_is_not_a_triple_quoted_string() {
1085        // C# verbatim strings escape a quote by doubling it, so a run spelling
1086        // `"""x` is a quote followed by `x`, not a string that runs on. Read
1087        // as one, it swallowed the rest of a 413-branch file.
1088        let q = "\u{22}";
1089        let file = |format: &str| SourceFile {
1090            id: "a".into(),
1091            format: format.into(),
1092            tokens: vec![
1093                lit_token("x", 0, 1, TokenKind::Identifier),
1094                lit_token(&q.repeat(2), 4, 6, TokenKind::Literal),
1095                lit_token(&format!("{q}x"), 6, 8, TokenKind::Literal),
1096                lit_token("if", 10, 12, TokenKind::Identifier),
1097                lit_token("y", 13, 14, TokenKind::Identifier),
1098            ],
1099            bytes: 20,
1100        };
1101        let cx_of = |format: &str| {
1102            compute_summary(
1103                &[file(format)],
1104                &[],
1105                10,
1106                SummaryMetric::Complexity,
1107                identity,
1108            )
1109            .files[0]
1110                .complexity
1111        };
1112        assert_eq!(cx_of("csharp"), 2, "the `if` after the escape is code");
1113        // The same run in Python really does open a string.
1114        assert_eq!(cx_of("python"), 1);
1115    }
1116
1117    fn lit_token(value: &str, start: u32, end: u32, kind: TokenKind) -> Token {
1118        Token {
1119            kind,
1120            value: value.to_string(),
1121            start: Location {
1122                line: 1,
1123                column: start,
1124                offset: start,
1125            },
1126            end: Location {
1127                line: 1,
1128                column: end,
1129                offset: end,
1130            },
1131        }
1132    }
1133
1134    #[test]
1135    fn a_name_the_file_binds_is_not_a_keyword() {
1136        // `case` and `when` are branch keywords somewhere and ordinary
1137        // variables elsewhere; the tokenizer tags both `Identifier`. A file
1138        // that assigns to the word has settled the question for itself.
1139        let code = "def run(c):\n case = 3\n when = 4\n return case + when\n";
1140        assert_eq!(cx("python", code), 1);
1141        // Reading it off an object is the same evidence.
1142        assert_eq!(cx("python", "def f(o):\n return o.case\n"), 1);
1143        // So is listing it as a parameter or an argument.
1144        let listed = "def headline(case, when):\n return fmt(case, when)\n";
1145        assert_eq!(cx("python", listed), 1);
1146        // Rust's `?` before a closing paren is still a branch: only words are
1147        // taken as names.
1148        let question = "fn f(s: &str) -> Result<u8, E> { Ok(parse(s)?) }";
1149        assert_eq!(cx("rust", question), 2);
1150        // Without that evidence the keyword still counts: one `def`, plus
1151        // `case` and `when`.
1152        let ruby = "def f(x)\n case x\n when 1 then 2\n end\nend";
1153        assert_eq!(cx("ruby", ruby), 3);
1154    }
1155
1156    #[test]
1157    fn the_c_family_declares_functions_without_a_keyword() {
1158        // `head(args) {` is the only marker C, C++, Java and C# give, and it
1159        // has to be told apart from the `) {` of a control statement.
1160        let code =
1161            "int add(int a, int b) { if (a > b) { return a; } return b; }\nvoid noop(void) { }\n";
1162        assert_eq!(cx("c", code), 3, "two functions plus one if");
1163        let java = "class T { int f(int a) { if (a > 0 && a < 10) { return a; } return 0; } }";
1164        assert_eq!(cx("java", java), 3, "one function, one if, one &&");
1165    }
1166
1167    #[test]
1168    fn a_parenthesised_statement_is_not_a_function() {
1169        // `switch (x) {` and `for (…) {` end in `) {` just like a signature.
1170        // One function plus the single `case`; the `switch` head adds neither.
1171        assert_eq!(
1172            cx(
1173                "c",
1174                "int f(int x) { switch (x) { case 1: return 1; } return 0; }"
1175            ),
1176            2
1177        );
1178        assert_eq!(
1179            cx("c", "void f(void) { for (int i = 0; i < 3; i++) { } }"),
1180            2
1181        );
1182        assert_eq!(cx("c", "void f(void) { while (x) { } }"), 2);
1183        // A language that declares functions by keyword is unaffected by this.
1184        assert_eq!(cx("go", "func f() { if x { } }"), 2);
1185    }
1186
1187    #[test]
1188    fn complexity_counts_decision_tokens() {
1189        let sources = vec![source(
1190            "a.js",
1191            "javascript",
1192            &["if", "x", "&&", "y", "for", "z", "else"],
1193            10,
1194        )];
1195        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1196        // 1 + (if, &&, for) = 4; "else" is not a decision point.
1197        assert_eq!(summary.files[0].complexity, 4);
1198    }
1199
1200    #[test]
1201    fn complexity_is_case_insensitive() {
1202        // SQL / PL/SQL / Fortran style uppercase keywords.
1203        let sources = vec![source(
1204            "a.sql",
1205            "sql",
1206            &["IF", "x", "OR", "y", "WHEN", "THEN", "If"],
1207            10,
1208        )];
1209        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1210        // 1 + (IF, OR, WHEN, If) = 5; THEN is not a decision point.
1211        assert_eq!(summary.files[0].complexity, 5);
1212    }
1213
1214    #[test]
1215    fn decision_token_edge_cases() {
1216        assert!(is_decision_token("unless"));
1217        assert!(is_decision_token("ELSEIF"));
1218        assert!(is_decision_token("andalso"));
1219        assert!(!is_decision_token(""));
1220        assert!(!is_decision_token("iffy"));
1221        assert!(!is_decision_token("conditionally"), "length-capped");
1222        assert!(!is_decision_token("форматирование"), "non-ASCII ignored");
1223    }
1224
1225    #[test]
1226    fn folder_rollup_uses_direct_parent() {
1227        let sources = vec![
1228            source("src/app/a.js", "javascript", &["x"], 5),
1229            source("src/app/b.js", "javascript", &["x", "y"], 5),
1230            source("src/c.js", "javascript", &["x"], 5),
1231            source("root.js", "javascript", &["x"], 5),
1232        ];
1233        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
1234        assert_eq!(summary.total_folders, 3);
1235        let app = summary
1236            .folders
1237            .iter()
1238            .find(|f| f.path == "src/app")
1239            .expect("src/app folder");
1240        assert_eq!(app.files, 2);
1241        assert_eq!(app.tokens, 3);
1242        let root = summary.folders.iter().find(|f| f.path == ".");
1243        assert!(root.is_some(), "root files grouped under '.'");
1244    }
1245
1246    #[test]
1247    fn duplication_attributed_to_both_fragments() {
1248        let sources = vec![
1249            source("a.js", "javascript", &["x", "y", "z"], 5),
1250            source("b.js", "javascript", &["x", "y", "z"], 5),
1251        ];
1252        let clones = vec![clone_between("javascript", "a.js", "b.js", 9, 30)];
1253        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, identity);
1254        for path in ["a.js", "b.js"] {
1255            let file = summary.files.iter().find(|f| f.path == path).unwrap();
1256            assert_eq!(file.duplicated_lines, 9, "{path} duplicated lines");
1257            assert_eq!(file.duplicated_tokens, 30, "{path} duplicated tokens");
1258        }
1259    }
1260
1261    #[test]
1262    fn synthetic_sub_format_sources_are_skipped() {
1263        let sources = vec![
1264            source("doc.md", "markdown", &["x", "y"], 100),
1265            source("doc.md:javascript", "javascript", &["x"], 0),
1266        ];
1267        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
1268        assert_eq!(summary.total_files, 1);
1269        assert_eq!(summary.files[0].path, "doc.md");
1270    }
1271
1272    #[test]
1273    fn sub_format_clone_folds_into_parent_file() {
1274        let sources = vec![source("doc.md", "markdown", &["x", "y"], 100)];
1275        let clones = vec![clone_between(
1276            "javascript",
1277            "doc.md:javascript",
1278            "doc.md:javascript",
1279            4,
1280            20,
1281        )];
1282        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, identity);
1283        assert_eq!(
1284            summary.files[0].duplicated_lines, 8,
1285            "both fragments fold in"
1286        );
1287    }
1288
1289    #[test]
1290    fn gap_lines_of_a_merged_clone_stay_out_of_file_duplication() {
1291        let sources = vec![
1292            source("a.js", "javascript", &["x"; 20], 10),
1293            source("b.js", "javascript", &["x"; 20], 10),
1294        ];
1295        let mut merged = clone_between("javascript", "a.js", "b.js", 10, 60);
1296        merged.unmatched_lines = [0, 3];
1297        let summary = compute_summary(&sources, &[merged], 10, SummaryMetric::Tokens, identity);
1298        let dup = |path: &str| {
1299            summary
1300                .files
1301                .iter()
1302                .find(|f| f.path == path)
1303                .unwrap()
1304                .duplicated_lines
1305        };
1306        assert_eq!(dup("a.js"), 10);
1307        assert_eq!(dup("b.js"), 7, "three gap lines in b are not duplicated");
1308    }
1309
1310    #[test]
1311    fn display_path_applied_before_dup_matching() {
1312        let sources = vec![source("/abs/root/a.js", "javascript", &["x"], 5)];
1313        let clones = vec![clone_between("javascript", "a.js", "a.js", 2, 10)];
1314        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, |p| {
1315            p.strip_prefix("/abs/root/").unwrap_or(p).to_string()
1316        });
1317        assert_eq!(summary.files[0].path, "a.js");
1318        assert_eq!(summary.files[0].duplicated_lines, 4);
1319    }
1320
1321    #[test]
1322    fn folders_truncated_to_top_n_but_total_reported() {
1323        let sources: Vec<SourceFile> = (0..5)
1324            .map(|i| source(&format!("dir{i}/f.js"), "javascript", &["x"], 1))
1325            .collect();
1326        let summary = compute_summary(&sources, &[], 2, SummaryMetric::Tokens, identity);
1327        assert_eq!(summary.folders.len(), 2);
1328        assert_eq!(summary.total_folders, 5);
1329    }
1330
1331    #[test]
1332    fn metric_parses_from_str() {
1333        assert_eq!(
1334            "complexity".parse::<SummaryMetric>().unwrap(),
1335            SummaryMetric::Complexity
1336        );
1337        assert!("bogus".parse::<SummaryMetric>().is_err());
1338    }
1339
1340    #[test]
1341    fn summary_serializes_camel_case() {
1342        let sources = vec![source("a.js", "javascript", &["x"], 5)];
1343        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Size, identity);
1344        let json = serde_json::to_string(&summary).unwrap();
1345        assert!(json.contains("\"totalFiles\""));
1346        assert!(json.contains("\"duplicatedLines\""));
1347        assert!(json.contains("\"by\":\"size\""));
1348        assert!(!json.contains("total_files"));
1349    }
1350}