Skip to main content

cpd_core/
summary.rs

1// summary.rs — opt-in codebase summary: per-file metrics, folder rollup, top-N lists.
2//
3// Everything in this module runs only when `--summary` is enabled, after
4// detection has finished, over data already held in memory (SourceFile tokens
5// and detected clones). Nothing in the detection hot path calls into it.
6
7use crate::models::{CpdClone, SourceFile, Token, TokenKind};
8use serde::{Deserialize, Serialize};
9use std::collections::HashMap;
10
11/// Metric used to rank files and folders in the summary.
12#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)]
13#[serde(rename_all = "lowercase")]
14pub enum SummaryMetric {
15    #[default]
16    Tokens,
17    Lines,
18    Size,
19    Complexity,
20}
21
22impl std::str::FromStr for SummaryMetric {
23    type Err = String;
24
25    fn from_str(s: &str) -> Result<Self, Self::Err> {
26        match s {
27            "tokens" => Ok(Self::Tokens),
28            "lines" => Ok(Self::Lines),
29            "size" => Ok(Self::Size),
30            "complexity" => Ok(Self::Complexity),
31            other => Err(format!(
32                "invalid summary metric '{other}': must be one of: tokens, lines, size, complexity"
33            )),
34        }
35    }
36}
37
38impl std::fmt::Display for SummaryMetric {
39    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
40        let s = match self {
41            Self::Tokens => "tokens",
42            Self::Lines => "lines",
43            Self::Size => "size",
44            Self::Complexity => "complexity",
45        };
46        f.write_str(s)
47    }
48}
49
50/// Per-file summary row.
51#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
52#[serde(rename_all = "camelCase")]
53pub struct FileSummary {
54    pub path: String,
55    pub format: String,
56    pub lines: u64,
57    pub tokens: u64,
58    pub bytes: u64,
59    pub duplicated_lines: u64,
60    pub duplicated_tokens: u64,
61    /// Cyclomatic-complexity estimate: 1 + count of decision-point tokens
62    /// (`if`, `for`, `while`, `case`, `catch`, `&&`, `||`, `?`, …).
63    pub complexity: u64,
64}
65
66/// Per-folder rollup. Files are counted in their direct parent directory only
67/// (no cumulative ancestor totals), so every file contributes to exactly one
68/// folder row and rows are directly comparable.
69#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
70#[serde(rename_all = "camelCase")]
71pub struct FolderSummary {
72    pub path: String,
73    pub files: u64,
74    pub lines: u64,
75    pub tokens: u64,
76    pub bytes: u64,
77    pub duplicated_lines: u64,
78    /// Sum of per-file complexity estimates (divide by `files` for the mean).
79    pub complexity: u64,
80}
81
82/// Codebase summary: top files and folder rollup.
83#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
84#[serde(rename_all = "camelCase")]
85pub struct Summary {
86    /// Primary sort metric.
87    pub by: SummaryMetric,
88    /// Top-N files by `by`, descending. Every row carries all metrics
89    /// (tokens, lines, bytes, complexity, duplication) so one list serves
90    /// every lens; re-run with a different `--summary-by` to re-rank.
91    pub files: Vec<FileSummary>,
92    /// Top-N folders by `by`, direct-parent aggregation.
93    pub folders: Vec<FolderSummary>,
94    /// Total number of files analyzed (before top-N truncation).
95    pub total_files: u64,
96    /// Total number of folders (before top-N truncation).
97    pub total_folders: u64,
98}
99
100/// Decision-point tokens counted by the complexity estimate. Conservative,
101/// language-agnostic list: branch/loop keywords and short-circuit operators
102/// that appear as standalone tokens across supported languages.
103///
104/// Matching is ASCII-case-insensitive so case-insensitive and
105/// uppercase-keyword languages (SQL, PL/SQL, Fortran, COBOL, BASIC, Pascal)
106/// count too. The occasional identifier spelled like a keyword slightly
107/// inflates an estimate that is only used for ranking.
108///
109/// Operators reach this function already joined — see [`joined_token`]. Only
110/// the JavaScript tokenizer emits `&&` as one token; the generic one splits
111/// every punctuation run into single characters, so without that joining the
112/// short-circuit arms here would be unreachable for every other format.
113fn is_decision_token(value: &str) -> bool {
114    let bytes = value.as_bytes();
115    if bytes.is_empty() || bytes.len() > 7 {
116        return false;
117    }
118    let mut lower = [0u8; 7];
119    for (dst, b) in lower.iter_mut().zip(bytes) {
120        *dst = b.to_ascii_lowercase();
121    }
122    matches!(
123        &lower[..bytes.len()],
124        b"if"
125            | b"elif"
126            | b"elsif"
127            | b"elseif"
128            | b"unless"
129            | b"for"
130            | b"foreach"
131            | b"while"
132            | b"until"
133            | b"case"
134            | b"cond"
135            | b"when"
136            | b"catch"
137            | b"rescue"
138            | b"except"
139            | b"andalso"
140            | b"orelse"
141            | b"&&"
142            | b"||"
143            | b"and"
144            | b"or"
145            | b"?"
146            | b"??"
147    )
148}
149
150/// What a bare `?` means in a language.
151#[derive(Clone, Copy, PartialEq, Eq)]
152enum QuestionMark {
153    /// It only ever opens a ternary: C, Java, PHP, Go templates, and most else.
154    Ternary,
155    /// It also marks an optional, so `String?` and `x?.y` are types and
156    /// accesses rather than branches. Neither TypeScript nor Swift has an
157    /// Elvis operator, so `?:` there is an optional property, not a branch.
158    Optional,
159    /// As above, but `?:` *is* the Elvis operator and does branch: Kotlin,
160    /// Groovy.
161    OptionalWithElvis,
162}
163
164/// How one language spells its branches, beyond the shared keyword set.
165///
166/// The shared set is right for most of the formats jscpd knows. An entry here
167/// exists only where a language branches on something the set has no word for
168/// (`match` arms in Rust, `guard` in Swift, `select` in Go) or spells
169/// something in the set so differently that counting it is simply wrong.
170#[derive(Clone, Copy)]
171struct DecisionRules {
172    /// Branch tokens beyond the shared set, matched after joining.
173    extra: &'static [&'static str],
174    question: QuestionMark,
175    /// Tokens that open a function body. Cyclomatic complexity is one path per
176    /// function; where a language has no single reliable marker this stays
177    /// empty and the estimate keeps its per-file baseline of one.
178    declarations: &'static [&'static str],
179    /// The C family declares a function as `head(args) {` with no keyword at
180    /// all, so its functions are counted from that shape instead.
181    braced_declarations: bool,
182    /// Whether `"""` and `'''` delimit a string that can span lines; see
183    /// [`has_triple_quoted_strings`].
184    triple_quoted_strings: bool,
185    /// Tokens that open a group of arms counted one by one through `extra`.
186    /// Each cancels one arm, because N arms are N paths and so N - 1 branches,
187    /// the same way a `switch` counts its `case` labels but not its `default`.
188    arm_groups: &'static [&'static str],
189}
190
191impl Default for DecisionRules {
192    fn default() -> Self {
193        Self {
194            extra: &[],
195            question: QuestionMark::Ternary,
196            declarations: &[],
197            braced_declarations: false,
198            triple_quoted_strings: false,
199            arm_groups: &[],
200        }
201    }
202}
203
204/// Formats whose strings can span lines between `"""` or `'''` delimiters.
205///
206/// Where a doubled quote is how a quote is escaped — C# verbatim strings,
207/// VB.NET, SQL, Pascal — three quotes in a row are ordinary string content: the
208/// regex `@"""((?:\\.|[^""\\])*)"""` is one line of C#. Reading such a run as
209/// a delimiter opens a string that never closes and swallows the rest of the
210/// file, which is why this is a list rather than a default.
211fn has_triple_quoted_strings(format: &str) -> bool {
212    matches!(
213        format,
214        "python" | "kotlin" | "scala" | "groovy" | "swift" | "java" | "julia" | "elixir" | "dart"
215    )
216}
217
218/// False for prose and data formats, whose "if" and `||` are words and
219/// version ranges rather than branches. One half of [`is_code`]: a format
220/// that can never have a branch can never have complexity above zero.
221pub fn has_control_flow(format: &str) -> bool {
222    !matches!(
223        format,
224        "markdown"
225            | "asciidoc"
226            | "rest"
227            | "textile"
228            | "wiki"
229            | "txt"
230            | "log"
231            | "csv"
232            | "json"
233            | "json5"
234            | "yaml"
235            | "toml"
236            | "ini"
237            | "properties"
238            | "editorconfig"
239            | "ignore"
240            | "diff"
241            | "gettext"
242    )
243}
244
245/// Markup, stylesheets, declarative schemas and the templating languages
246/// built on top of markup: a duplicated template or style rule repeating is
247/// not the maintenance problem duplicated programming logic is, so it does
248/// not count toward the health score's duplication share at all — the same
249/// treatment prose and data files get, just decided per clone rather than
250/// per file, since a `.svelte` or `.vue` file's markup and style blocks are
251/// tokenized separately from its script block. The other half of
252/// [`is_code`]: whole files in these formats have no complexity either — the
253/// "if" in an HTML attribute and the "and" in a media query are words, not
254/// branches.
255///
256/// These are the tokenizer's own format *names*
257/// (`cpd-tokenizer/src/formats.rs`), not file extensions: html/htm/xml/svg
258/// all tokenize as `markup` (`html` is the name a component file's markup
259/// block and a Markdown html snippet carry), `.puml`/`.plantuml` as
260/// `plant-uml`, `.tpl` as `smarty`, `.jade` as `pug`, and `.vtl` as
261/// `velocity` — matching on the extension instead of the name a clone's
262/// `format` field actually carries would silently never exclude anything.
263pub fn is_markup(format: &str) -> bool {
264    matches!(
265        format,
266        "markup"
267            | "html"
268            | "css"
269            | "scss"
270            | "sass"
271            | "less"
272            | "stylus"
273            | "razor"
274            | "haml"
275            | "pug"
276            | "handlebars"
277            | "erb"
278            | "liquid"
279            | "twig"
280            | "velocity"
281            | "ftl"
282            | "soy"
283            | "smarty"
284            | "tt2"
285            | "protobuf"
286            | "plant-uml"
287            | "mermaid"
288            | "django"
289            | "aspnet"
290    )
291}
292
293/// Whether a format counts as the project's code: prose and data
294/// ([`has_control_flow`]) and markup ([`is_markup`]) do not. This one test
295/// decides where complexity can be above zero, and through that which files
296/// are `health::compute`'s "code files"; a format-level duplication
297/// breakdown leaves the same formats out.
298pub fn is_code(format: &str) -> bool {
299    has_control_flow(format) && !is_markup(format)
300}
301
302fn rules_for(format: &str) -> DecisionRules {
303    DecisionRules {
304        triple_quoted_strings: has_triple_quoted_strings(format),
305        ..language_rules(format)
306    }
307}
308
309fn language_rules(format: &str) -> DecisionRules {
310    match format {
311        // A `match` has no keyword per arm, only the `=>` each one is written
312        // with. Every arm is counted and the `match` itself takes one back: a
313        // three-arm match is three paths, which is two branches.
314        "rust" => DecisionRules {
315            extra: &["=>"],
316            declarations: &["fn"],
317            arm_groups: &["match"],
318            ..Default::default()
319        },
320        "swift" => DecisionRules {
321            extra: &["guard"],
322            question: QuestionMark::Optional,
323            declarations: &["func"],
324            ..Default::default()
325        },
326        // `=>` opens an arrow function here rather than a branch, so it counts
327        // toward the function tally instead.
328        "typescript" | "tsx" | "flow" | "javascript" | "jsx" => DecisionRules {
329            question: QuestionMark::Optional,
330            declarations: &["function", "=>"],
331            ..Default::default()
332        },
333        "kotlin" => DecisionRules {
334            question: QuestionMark::OptionalWithElvis,
335            declarations: &["fun"],
336            ..Default::default()
337        },
338        "groovy" => DecisionRules {
339            question: QuestionMark::OptionalWithElvis,
340            declarations: &["def"],
341            ..Default::default()
342        },
343        "csharp" => DecisionRules {
344            question: QuestionMark::Optional,
345            braced_declarations: true,
346            ..Default::default()
347        },
348        "c" | "c-header" | "cpp" | "cpp-header" | "java" | "objectivec" | "clike" => {
349            DecisionRules {
350                braced_declarations: true,
351                ..Default::default()
352            }
353        }
354        "go" => DecisionRules {
355            extra: &["select"],
356            declarations: &["func"],
357            ..Default::default()
358        },
359        "scala" | "python" | "ruby" | "crystal" => DecisionRules {
360            declarations: &["def"],
361            ..Default::default()
362        },
363        "php" | "lua" => DecisionRules {
364            declarations: &["function"],
365            ..Default::default()
366        },
367        "erlang" => DecisionRules {
368            extra: &["receive"],
369            ..Default::default()
370        },
371        _ => DecisionRules::default(),
372    }
373}
374
375/// Two-character operators the generic tokenizer hands over as two tokens.
376const JOINED_OPERATORS: &[&str] = &["&&", "||", "??", "?.", "?:", "=>"];
377
378/// One token's text, or the two-character operator that two adjacent tokens
379/// spell between them.
380///
381/// The pair cannot borrow: each token owns its own `String`, so two adjacent
382/// characters of the source are not adjacent in memory. Two bytes on the stack
383/// avoid an allocation per operator.
384enum Scanned<'a> {
385    Single(&'a str),
386    Pair([u8; 2]),
387}
388
389impl Scanned<'_> {
390    fn text(&self) -> &str {
391        match self {
392            Self::Single(text) => text,
393            Self::Pair(bytes) => std::str::from_utf8(bytes).unwrap_or(""),
394        }
395    }
396}
397
398/// The token at `at`, joined with the next one when the two are adjacent
399/// single-character punctuation spelling a two-character operator.
400///
401/// The generic tokenizer splits `&&` into two `&` tokens, so a scan that looks
402/// at one token at a time can never see a short-circuit operator at all.
403/// Returns what to classify and the index to continue from.
404fn joined_token(tokens: &[Token], at: usize) -> (Scanned<'_>, usize) {
405    let current = &tokens[at];
406    let single = (Scanned::Single(current.value.as_str()), at + 1);
407    let Some(next) = tokens.get(at + 1) else {
408        return single;
409    };
410    let (Some(&left), Some(&right)) = (
411        current.value.as_bytes().first(),
412        next.value.as_bytes().first(),
413    ) else {
414        return single;
415    };
416    // Adjacent in the source, and both exactly one punctuation character.
417    if current.end.offset != next.start.offset
418        || current.value.len() != 1
419        || next.value.len() != 1
420        || !left.is_ascii_punctuation()
421        || !right.is_ascii_punctuation()
422    {
423        return single;
424    }
425    let pair = [left, right];
426    match std::str::from_utf8(&pair).is_ok_and(|text| JOINED_OPERATORS.contains(&text)) {
427        true => (Scanned::Pair(pair), at + 2),
428        false => single,
429    }
430}
431
432/// Which tokens lie inside a triple-quoted string.
433///
434/// The tokenizer is line-based: it closes every string at the end of the line
435/// it started on, so the body of a `"""` string arrives as ordinary words, and
436/// a docstring saying "if the value is big" lands three branches on the file.
437///
438/// The quotes themselves survive as literal tokens, and a run of adjacent
439/// literals spells exactly the quote characters its line holds: a PEP 257
440/// opener `"""Summary.` arrives as `""` and `"Summary.`, a closer on its own
441/// line as `""` and `"`, and a one-line docstring as `""`, `"One."`, `""`. So
442/// an odd number of triple quotes in a run opens or closes the string, and an
443/// even number leaves it as it was.
444///
445/// Reading a pair of tokens alone is not enough, and was the first version of
446/// this: with the summary written on the opening line the opener is not a bare
447/// quote, so the *closing* `"""` looked like an opener and swallowed every
448/// branch that followed it.
449fn triple_quoted(tokens: &[Token]) -> Vec<bool> {
450    const DELIMITERS: [&str; 2] = ["\"\"\"", "'''"];
451    let mut inside = vec![false; tokens.len()];
452    let mut open: Option<&str> = None;
453    let mut at = 0usize;
454    while at < tokens.len() {
455        if tokens[at].kind != TokenKind::Literal {
456            inside[at] = open.is_some();
457            at += 1;
458            continue;
459        }
460        let start = at;
461        let mut text = tokens[at].value.clone();
462        at += 1;
463        while at < tokens.len()
464            && tokens[at].kind == TokenKind::Literal
465            && tokens[at - 1].end.offset == tokens[at].start.offset
466        {
467            text.push_str(&tokens[at].value);
468            at += 1;
469        }
470        let was_open = open.is_some();
471        for delimiter in DELIMITERS {
472            let toggles = text.matches(delimiter).count() % 2 == 1;
473            match open {
474                Some(current) if current == delimiter && toggles => open = None,
475                None if toggles => open = Some(delimiter),
476                _ => {}
477            }
478        }
479        // The run is string text whichever way it turned the state.
480        inside[start..at].fill(was_open || open.is_some());
481    }
482    inside
483}
484
485/// Words a file binds as names of its own.
486///
487/// `case`, `when` and `cond` are branch keywords in some languages and
488/// perfectly ordinary variable names in others — the shared list cannot tell
489/// them apart, and the tokenizer is no help: its `classify_word` returns
490/// `Identifier` for every word, so `if` in Python is tagged exactly like a
491/// variable called `case`.
492///
493/// What a file *does* reveal is which words it assigns to, reads off an
494/// object, or lists as a parameter or argument. A word this file writes
495/// `case = 3`, `x.case` or `f(case, when)` for is that file's own name,
496/// whatever the language reserves, and counting it as a branch is what
497/// inflated the estimate elevenfold on a file of five such assignments.
498fn locally_bound(tokens: &[Token]) -> Vec<&str> {
499    let mut bound = Vec::new();
500    for (at, token) in tokens.iter().enumerate() {
501        if !is_decision_token(&token.value) {
502            continue;
503        }
504        // `obj.case` — a member, never the keyword.
505        let after_dot = at
506            .checked_sub(1)
507            .is_some_and(|previous| tokens[previous].value == ".");
508        // `case = 3`, but not `case == 3` or `case => 3`.
509        let assigned = tokens.get(at + 1).is_some_and(|next| next.value == "=")
510            && tokens
511                .get(at + 2)
512                .is_none_or(|after| !matches!(after.value.as_str(), "=" | ">"));
513        // `def headline(case, when)` or `headline(case, when)`: a keyword is
514        // never written directly before a comma or a closing paren, a name in
515        // a parameter or argument list always is. Words only — `Ok(parse()?)`
516        // is Rust's `?` doing its job, not a name.
517        let listed = token.value.starts_with(|c: char| c.is_ascii_alphabetic())
518            && tokens
519                .get(at + 1)
520                .is_some_and(|next| matches!(next.value.as_str(), "," | ")"));
521        if after_dot || assigned || listed {
522            bound.push(token.value.as_str());
523        }
524    }
525    bound.sort_unstable();
526    bound.dedup();
527    bound
528}
529
530/// Keywords that take a parenthesised head and a block, so that `) {` after
531/// one of them opens a branch rather than a function body.
532const PARENTHESISED_STATEMENTS: &[&str] = &[
533    "switch",
534    "using",
535    "lock",
536    "synchronized",
537    "with",
538    "do",
539    "try",
540    "return",
541    "sizeof",
542    "typeof",
543    "new",
544    "throw",
545    "await",
546    "yield",
547    "fixed",
548    "unsafe",
549];
550
551/// Decision points and function count for one file.
552///
553/// Cyclomatic complexity is one path per function plus one per branch. Where a
554/// language has no reliable function marker the tally falls back to the
555/// per-file baseline of one, which is what this estimate has always used.
556fn scan_complexity(tokens: &[Token], rules: &DecisionRules) -> (u64, u64) {
557    let (mut decisions, mut functions, mut groups) = (0u64, 0u64, 0u64);
558    let mut at = 0usize;
559    let in_string = match rules.triple_quoted_strings {
560        true => triple_quoted(tokens),
561        false => vec![false; tokens.len()],
562    };
563    let bound = locally_bound(tokens);
564    // What the token before each open paren was, so a `) {` can be told from
565    // the head it closes: a function signature, or `if (…) {`.
566    let mut heads: Vec<&str> = Vec::new();
567    let mut last_head: Option<&str> = None;
568    while at < tokens.len() {
569        // Skip the body of a string that spans lines; see `triple_quoted`.
570        if in_string[at] {
571            at += 1;
572            continue;
573        }
574        let (scanned, next) = joined_token(tokens, at);
575        let text = scanned.text();
576        // `String?` writes the optional against the type it belongs to;
577        // `cond ? a : b` puts the ternary in the open. Nothing else separates
578        // the two without parsing, and the convention is near-universal.
579        let attached = at
580            .checked_sub(1)
581            .and_then(|previous| tokens.get(previous))
582            .is_some_and(|previous| previous.end.offset == tokens[at].start.offset);
583        at = next;
584        // C, C++, Java and C# open a function with `) {` and no keyword of
585        // their own. Tracking what preceded each `(` is enough to tell that
586        // from the `) {` of an `if` or a `switch`, and it is the only reason
587        // those languages kept a per-file baseline.
588        if rules.braced_declarations {
589            match text {
590                "(" => {
591                    let head = at
592                        .checked_sub(2)
593                        .and_then(|before| tokens.get(before))
594                        .map(|token| token.value.as_str())
595                        .unwrap_or("");
596                    heads.push(head);
597                }
598                ")" => last_head = heads.pop(),
599                "{" => {
600                    if let Some(head) = last_head.take()
601                        && !head.is_empty()
602                        && head.chars().all(|c| c.is_alphanumeric() || c == '_')
603                        && !is_decision_token(head)
604                        && !PARENTHESISED_STATEMENTS
605                            .iter()
606                            .any(|s| s.eq_ignore_ascii_case(head))
607                    {
608                        functions += 1;
609                    }
610                }
611                _ => last_head = None,
612            }
613        }
614        if rules
615            .declarations
616            .iter()
617            .any(|d| d.eq_ignore_ascii_case(text))
618        {
619            functions += 1;
620            continue;
621        }
622        if rules
623            .arm_groups
624            .iter()
625            .any(|g| g.eq_ignore_ascii_case(text))
626        {
627            groups += 1;
628            continue;
629        }
630        if rules.extra.iter().any(|e| e.eq_ignore_ascii_case(text)) {
631            decisions += 1;
632            continue;
633        }
634        // A word this file binds as a name is that file's name, not a keyword.
635        if bound.binary_search(&text).is_ok() {
636            continue;
637        }
638        let counts = match text {
639            // `?.` never branches, and `?:` branches only where it is the
640            // Elvis operator rather than an optional property.
641            "?." => false,
642            "?:" => rules.question != QuestionMark::Optional,
643            "?" => rules.question == QuestionMark::Ternary || !attached,
644            other => is_decision_token(other),
645        };
646        if counts {
647            decisions += 1;
648        }
649    }
650    (decisions.saturating_sub(groups), functions)
651}
652
653/// A synthetic source is the per-sub-format shadow of a multi-format file
654/// (markdown/vue/svelte embedded code); its id is `<parent-id>:<format>` and
655/// its metrics are already covered by the parent entry.
656fn is_synthetic(source: &SourceFile) -> bool {
657    source
658        .id
659        .strip_suffix(source.format.as_str())
660        .is_some_and(|prefix| prefix.ends_with(':'))
661}
662
663fn metric_of(file: &FileSummary, by: SummaryMetric) -> u64 {
664    match by {
665        SummaryMetric::Tokens => file.tokens,
666        SummaryMetric::Lines => file.lines,
667        SummaryMetric::Size => file.bytes,
668        SummaryMetric::Complexity => file.complexity,
669    }
670}
671
672fn folder_metric_of(folder: &FolderSummary, by: SummaryMetric) -> u64 {
673    match by {
674        SummaryMetric::Tokens => folder.tokens,
675        SummaryMetric::Lines => folder.lines,
676        SummaryMetric::Size => folder.bytes,
677        SummaryMetric::Complexity => folder.complexity,
678    }
679}
680
681/// Parent directory of a path, with separators normalized to `/`.
682/// Files at the scan root map to `"."`.
683fn parent_dir(path: &str) -> String {
684    let normalized = path.replace('\\', "/");
685    match normalized.rfind('/') {
686        Some(0) => "/".to_string(),
687        Some(idx) => normalized[..idx].to_string(),
688        None => ".".to_string(),
689    }
690}
691
692/// Compute the summary from detection results.
693///
694/// `display_path` maps a source id (canonical absolute path) to the path shown
695/// in reports — the same relativization applied to clone fragments, so
696/// per-file duplication matching works on identical strings.
697pub fn compute_summary(
698    sources: &[SourceFile],
699    clones: &[CpdClone],
700    top: usize,
701    by: SummaryMetric,
702    display_path: impl Fn(&str) -> String,
703) -> Summary {
704    // Per-file duplication, keyed by display path. Both fragments of a clone
705    // count toward their file: the question here is "where does duplicated
706    // code live", not the de-duplicated total that Statistics reports.
707    let mut dup: HashMap<String, (u64, u64)> = HashMap::new();
708    for clone in clones {
709        for (fragment, unmatched) in [
710            (&clone.fragment_a, clone.unmatched_lines[0]),
711            (&clone.fragment_b, clone.unmatched_lines[1]),
712        ] {
713            // Sub-format fragments carry a `<path>:<format>` id; fold them
714            // into the parent file.
715            let path = fragment
716                .source_id
717                .strip_suffix(&format!(":{}", clone.format))
718                .unwrap_or(&fragment.source_id);
719            let entry = dup.entry(path.to_string()).or_default();
720            // Gap lines of a merged clone are not duplicated code.
721            entry.0 += fragment
722                .end
723                .line
724                .saturating_sub(fragment.start.line)
725                .saturating_sub(unmatched) as u64;
726            entry.1 += clone.token_count as u64;
727        }
728    }
729
730    let mut files: Vec<FileSummary> = sources
731        .iter()
732        .filter(|s| !is_synthetic(s))
733        .map(|source| {
734            let path = display_path(&source.id);
735            // Same line metric as Statistics: max token start line.
736            let lines = source
737                .tokens
738                .iter()
739                .map(|t| t.start.line)
740                .max()
741                .unwrap_or(0) as u64;
742            let (decisions, functions) =
743                scan_complexity(&source.tokens, &rules_for(&source.format));
744            let (duplicated_lines, duplicated_tokens) = dup.get(&path).copied().unwrap_or_default();
745            FileSummary {
746                lines,
747                tokens: source.tokens.len() as u64,
748                bytes: source.bytes,
749                duplicated_lines,
750                duplicated_tokens,
751                // One path per function, or the per-file baseline where the
752                // language has no marker the scan can trust. Prose, data and
753                // markup have no paths: an "if" in a README or an HTML
754                // attribute is a word, and a lock file full of `||` version
755                // ranges is not code.
756                complexity: match is_code(&source.format) {
757                    true => functions.max(1) + decisions,
758                    false => 0,
759                },
760                format: source.format.clone(),
761                path,
762            }
763        })
764        .collect();
765
766    let total_files = files.len() as u64;
767
768    // Folder rollup over ALL files (before top-N truncation).
769    let mut folder_map: HashMap<String, FolderSummary> = HashMap::new();
770    for file in &files {
771        let dir = parent_dir(&file.path);
772        let entry = folder_map
773            .entry(dir.clone())
774            .or_insert_with(|| FolderSummary {
775                path: dir,
776                files: 0,
777                lines: 0,
778                tokens: 0,
779                bytes: 0,
780                duplicated_lines: 0,
781                complexity: 0,
782            });
783        entry.files += 1;
784        entry.lines += file.lines;
785        entry.tokens += file.tokens;
786        entry.bytes += file.bytes;
787        entry.duplicated_lines += file.duplicated_lines;
788        entry.complexity += file.complexity;
789    }
790    let total_folders = folder_map.len() as u64;
791
792    // Top-N files by the primary metric: `--summary-top N` always yields at
793    // most N rows (least surprise). Other lenses are one `--summary-by` away;
794    // every row still carries all metrics.
795    files.sort_by(|a, b| {
796        metric_of(b, by)
797            .cmp(&metric_of(a, by))
798            .then_with(|| a.path.cmp(&b.path))
799    });
800    files.truncate(top);
801
802    let mut folders: Vec<FolderSummary> = folder_map.into_values().collect();
803    folders.sort_by(|a, b| {
804        folder_metric_of(b, by)
805            .cmp(&folder_metric_of(a, by))
806            .then_with(|| a.path.cmp(&b.path))
807    });
808    folders.truncate(top);
809
810    Summary {
811        by,
812        files,
813        folders,
814        total_files,
815        total_folders,
816    }
817}
818
819#[cfg(test)]
820mod tests {
821    use super::*;
822    use crate::models::{CpdClone, Fragment, Location, Token, TokenKind};
823
824    fn loc(line: u32) -> Location {
825        Location {
826            line,
827            column: 0,
828            offset: 0,
829        }
830    }
831
832    fn token(value: &str, line: u32) -> Token {
833        Token {
834            kind: TokenKind::Keyword,
835            value: value.to_string(),
836            start: loc(line),
837            end: loc(line),
838        }
839    }
840
841    fn source(id: &str, format: &str, values: &[&str], bytes: u64) -> SourceFile {
842        SourceFile {
843            id: id.to_string(),
844            format: format.to_string(),
845            tokens: values
846                .iter()
847                .enumerate()
848                .map(|(i, v)| token(v, i as u32 + 1))
849                .collect(),
850            bytes,
851        }
852    }
853
854    fn clone_between(format: &str, a: &str, b: &str, lines: u32, tokens: u32) -> CpdClone {
855        let fragment = |id: &str| Fragment {
856            source_id: id.to_string(),
857            source_root: None,
858            start: loc(1),
859            end: loc(1 + lines),
860            range: [0, tokens],
861            blame: None,
862        };
863        CpdClone {
864            format: format.to_string(),
865            fragment_a: fragment(a),
866            fragment_b: fragment(b),
867            token_count: tokens,
868            is_new: false,
869            kind: Default::default(),
870            similarity: None,
871            similarity_method: None,
872            unmatched_lines: [0, 0],
873        }
874    }
875
876    fn identity(path: &str) -> String {
877        path.to_string()
878    }
879
880    #[test]
881    fn empty_input_produces_empty_summary() {
882        let summary = compute_summary(&[], &[], 10, SummaryMetric::Tokens, identity);
883        assert!(summary.files.is_empty());
884        assert!(summary.folders.is_empty());
885        assert_eq!(summary.total_files, 0);
886        assert_eq!(summary.total_folders, 0);
887    }
888
889    #[test]
890    fn prose_and_data_have_no_complexity() {
891        let words = [
892            "If", "you", "need", "it", "or", "while", "waiting", "for", "a", "case",
893        ];
894        let sources = vec![
895            source("README.md", "markdown", &words, 10),
896            source(
897                "pnpm-lock.yaml",
898                "yaml",
899                &["version", ":", "^1", "||", "^2"],
900                10,
901            ),
902            source("guide.rst", "rest", &words, 10),
903            source("notes.py", "python", &words, 10),
904        ];
905        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
906        let cx = |path: &str| {
907            summary
908                .files
909                .iter()
910                .find(|f| f.path == path)
911                .unwrap()
912                .complexity
913        };
914        assert_eq!(cx("README.md"), 0, "a word is not a branch");
915        assert_eq!(cx("pnpm-lock.yaml"), 0, "a version range is not a branch");
916        assert_eq!(cx("guide.rst"), 0, "reStructuredText is prose too");
917        assert!(cx("notes.py") > 1, "the same words in code still count");
918    }
919
920    #[test]
921    fn markup_styles_and_templates_have_no_complexity() {
922        // The same exclusion the duplication share applies (`is_markup`):
923        // an "if" in markup text or a media query's "and" is a word, not a
924        // branch. `markup` is what html/xml files tokenize as; `html` is the
925        // name a component file's markup block carries.
926        let words = [
927            "<", "a", ">", "If", "you", "click", "or", "wait", "<", "/", "a", ">",
928        ];
929        let sources = vec![
930            source("index.html", "markup", &words, 10),
931            source("snippet.html", "html", &words, 10),
932            source(
933                "site.css",
934                "css",
935                &["@", "media", "screen", "and", "(", "print", ")"],
936                10,
937            ),
938            source("card.twig", "twig", &["{", "%", "if", "user", "%", "}"], 10),
939        ];
940        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
941        assert!(
942            summary.files.iter().all(|f| f.complexity == 0),
943            "markup formats must have complexity 0: {:?}",
944            summary.files
945        );
946    }
947
948    #[test]
949    fn files_sorted_by_primary_metric() {
950        let sources = vec![
951            source("src/small.js", "javascript", &["a", "b"], 10),
952            source("src/big.js", "javascript", &["a", "b", "c", "d"], 20),
953        ];
954        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
955        assert_eq!(summary.files[0].path, "src/big.js");
956        assert_eq!(summary.files[0].tokens, 4);
957        assert_eq!(summary.total_files, 2);
958    }
959
960    #[test]
961    fn top_n_is_exact_row_count_by_primary_metric() {
962        // huge.js wins on tokens, fat.js wins on size — top=1 by tokens must
963        // yield exactly one row: huge.js. `--summary-top N` never surprises
964        // with more than N rows; other metrics are served by --summary-by.
965        let sources = vec![
966            source("huge.js", "javascript", &["a", "b", "c", "d", "e"], 1),
967            source("fat.js", "javascript", &["a"], 9999),
968        ];
969        let summary = compute_summary(&sources, &[], 1, SummaryMetric::Tokens, identity);
970        assert_eq!(summary.files.len(), 1);
971        assert_eq!(summary.files[0].path, "huge.js");
972        assert_eq!(summary.total_files, 2, "truncation stays visible");
973
974        let by_size = compute_summary(&sources, &[], 1, SummaryMetric::Size, identity);
975        assert_eq!(by_size.files[0].path, "fat.js");
976    }
977
978    /// Tokens laid out over a real source string the way the generic
979    /// tokenizer hands them over: one per word, one per punctuation
980    /// character, with the offsets that make adjacency visible.
981    fn lay_out(id: &str, format: &str, code: &str) -> SourceFile {
982        let mut tokens = Vec::new();
983        let bytes = code.as_bytes();
984        let mut at = 0usize;
985        while at < bytes.len() {
986            let byte = bytes[at];
987            if byte.is_ascii_whitespace() {
988                at += 1;
989                continue;
990            }
991            let start = at;
992            if byte.is_ascii_alphanumeric() || byte == b'_' {
993                while at < bytes.len() && (bytes[at].is_ascii_alphanumeric() || bytes[at] == b'_') {
994                    at += 1;
995                }
996            } else {
997                at += 1;
998            }
999            let at32 = |offset: usize| Location {
1000                line: 1,
1001                column: offset as u32,
1002                offset: offset as u32,
1003            };
1004            tokens.push(Token {
1005                kind: TokenKind::Identifier,
1006                value: code[start..at].to_string(),
1007                start: at32(start),
1008                end: at32(at),
1009            });
1010        }
1011        SourceFile {
1012            id: id.to_string(),
1013            format: format.to_string(),
1014            tokens,
1015            bytes: code.len() as u64,
1016        }
1017    }
1018
1019    fn cx(format: &str, code: &str) -> u64 {
1020        let sources = vec![lay_out("a", format, code)];
1021        compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity).files[0].complexity
1022    }
1023
1024    #[test]
1025    fn short_circuit_operators_count_when_split_into_characters() {
1026        // The generic tokenizer hands `&&` over as two `&` tokens, so without
1027        // joining them the short-circuit arms of the shared list never match
1028        // for any format but JavaScript.
1029        assert_eq!(cx("c", "int f() { return a && b; }"), 2);
1030        assert_eq!(cx("c", "int f() { return a || b; }"), 2);
1031        assert_eq!(cx("c", "int f() { return a && b || c; }"), 3);
1032        // A single `&` is a bitwise and, not a branch.
1033        assert_eq!(cx("c", "int f() { return a & b; }"), 1);
1034    }
1035
1036    #[test]
1037    fn an_optional_is_not_a_ternary() {
1038        // `String?` writes the question mark against its type; a ternary puts
1039        // it in the open. Nothing else tells them apart without parsing.
1040        assert_eq!(cx("swift", "func f(x: String?) -> Int { return 1 }"), 1);
1041        assert_eq!(cx("swift", "func f(x: Foo) { let y = x?.bar }"), 1);
1042        assert_eq!(
1043            cx("swift", "func f(x: Int) -> Int { return x > 0 ? 1 : 2 }"),
1044            2
1045        );
1046        // Nil-coalescing is a branch in any spelling.
1047        assert_eq!(
1048            cx("swift", "func f(a: Int?, b: Int) -> Int { return a ?? b }"),
1049            2
1050        );
1051        // A language where `?` only ever opens a ternary is unaffected.
1052        assert_eq!(cx("c", "int f(int x) { return x ? 1 : 2; }"), 2);
1053    }
1054
1055    #[test]
1056    fn a_match_arm_is_the_branch_not_the_match() {
1057        // Three arms are three paths: one for the function, two branches.
1058        let three_arms = "fn f(x: i32) -> i32 { match x { 0 => 1, 1 => 2, _ => 3 } }";
1059        assert_eq!(cx("rust", three_arms), 3);
1060        // A guard is a branch of its own on top of the arm it guards.
1061        let guarded = "fn f(x: i32, ok: bool) -> i32 { match x { 0 if ok => 1, 0 => 2, _ => 3 } }";
1062        assert_eq!(cx("rust", guarded), 4);
1063        // A match with a single arm does not branch at all.
1064        assert_eq!(cx("rust", "fn f(x: i32) -> i32 { match x { _ => 0 } }"), 1);
1065        // `=>` opens an arrow function in JavaScript, so it is a declaration
1066        // there rather than a branch.
1067        assert_eq!(cx("javascript", "const f = (x) => x + 1;"), 1);
1068    }
1069
1070    #[test]
1071    fn complexity_counts_one_path_per_function() {
1072        // Four functions, three of them with a single branch.
1073        let code = "def a(n):\n if n: pass\ndef b(n):\n if n: pass\ndef c(n):\n if n: pass\ndef d(n):\n pass\n";
1074        assert_eq!(cx("python", code), 7);
1075        // A language with no marker the scan can trust keeps the per-file
1076        // baseline of one rather than guessing.
1077        assert_eq!(cx("cobol", "IF x THEN y"), 2);
1078    }
1079
1080    #[test]
1081    fn guard_and_select_are_branches_where_they_exist() {
1082        assert_eq!(
1083            cx("swift", "func f(x: Int) { guard x > 0 else { return } }"),
1084            2
1085        );
1086        assert_eq!(cx("go", "func f() { select { } }"), 2);
1087        // `guard` is an ordinary word elsewhere.
1088        assert_eq!(cx("c", "int f() { int guard = 1; return guard; }"), 1);
1089    }
1090
1091    #[test]
1092    fn elvis_branches_only_where_the_language_has_one() {
1093        // Kotlin's `?:` is the elvis operator.
1094        assert_eq!(
1095            cx("kotlin", "fun f(a: Int?, b: Int): Int { return a ?: b }"),
1096            2
1097        );
1098        // TypeScript has no elvis; `a?: T` marks an optional property.
1099        assert_eq!(cx("typescript", "function f(a?: string) { return a; }"), 1);
1100    }
1101
1102    #[test]
1103    fn a_docstring_is_not_a_pile_of_branches() {
1104        // The real tokenizer closes every string at the end of its line, so a
1105        // docstring leaves a literal holding just its opening quote and its
1106        // body arrives as ordinary words.
1107        let quote = "\u{22}";
1108        let tokens = vec![
1109            lit_token("def", 0, 3, TokenKind::Identifier),
1110            lit_token("f", 4, 5, TokenKind::Identifier),
1111            lit_token(&quote.repeat(2), 6, 8, TokenKind::Literal),
1112            lit_token(quote, 8, 9, TokenKind::Literal),
1113            lit_token("if", 10, 12, TokenKind::Identifier),
1114            lit_token("for", 13, 16, TokenKind::Identifier),
1115            lit_token("while", 17, 22, TokenKind::Identifier),
1116            lit_token(&quote.repeat(2), 23, 25, TokenKind::Literal),
1117            lit_token(quote, 25, 26, TokenKind::Literal),
1118            lit_token("if", 27, 29, TokenKind::Identifier),
1119        ];
1120        let sources = vec![SourceFile {
1121            id: "a".into(),
1122            format: "python".into(),
1123            tokens,
1124            bytes: 30,
1125        }];
1126        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1127        // One `def`, and only the `if` outside the docstring.
1128        assert_eq!(summary.files[0].complexity, 2);
1129    }
1130
1131    #[test]
1132    fn a_docstring_with_its_summary_on_the_opening_line_closes_where_it_ends() {
1133        // PEP 257 style, as the tokenizer really hands it over: the opener is
1134        // `""` + `"Summary.` rather than a bare quote, so a reader that looked
1135        // for `""` + `"` alone took the *closing* quotes for an opener and
1136        // swallowed every branch after them.
1137        let q = "\u{22}";
1138        let tokens = vec![
1139            lit_token("def", 0, 3, TokenKind::Identifier),
1140            lit_token("f", 4, 5, TokenKind::Identifier),
1141            lit_token(&q.repeat(2), 10, 12, TokenKind::Literal),
1142            lit_token(&format!("{q}Summary."), 12, 21, TokenKind::Literal),
1143            lit_token("if", 30, 32, TokenKind::Identifier),
1144            lit_token("and", 33, 36, TokenKind::Identifier),
1145            lit_token(&q.repeat(2), 40, 42, TokenKind::Literal),
1146            lit_token(q, 42, 43, TokenKind::Literal),
1147            lit_token("if", 50, 52, TokenKind::Identifier),
1148            lit_token("x", 53, 54, TokenKind::Identifier),
1149            // A one-line docstring is balanced on its own line.
1150            lit_token(&q.repeat(2), 60, 62, TokenKind::Literal),
1151            lit_token(&format!("{q}One.{q}"), 62, 68, TokenKind::Literal),
1152            lit_token(&q.repeat(2), 68, 70, TokenKind::Literal),
1153            lit_token("for", 75, 78, TokenKind::Identifier),
1154        ];
1155        let sources = vec![SourceFile {
1156            id: "a".into(),
1157            format: "python".into(),
1158            tokens,
1159            bytes: 80,
1160        }];
1161        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1162        // One `def`, plus the `if` and the `for` written as code.
1163        assert_eq!(summary.files[0].complexity, 3);
1164    }
1165
1166    #[test]
1167    fn a_doubled_quote_escape_is_not_a_triple_quoted_string() {
1168        // C# verbatim strings escape a quote by doubling it, so a run spelling
1169        // `"""x` is a quote followed by `x`, not a string that runs on. Read
1170        // as one, it swallowed the rest of a 413-branch file.
1171        let q = "\u{22}";
1172        let file = |format: &str| SourceFile {
1173            id: "a".into(),
1174            format: format.into(),
1175            tokens: vec![
1176                lit_token("x", 0, 1, TokenKind::Identifier),
1177                lit_token(&q.repeat(2), 4, 6, TokenKind::Literal),
1178                lit_token(&format!("{q}x"), 6, 8, TokenKind::Literal),
1179                lit_token("if", 10, 12, TokenKind::Identifier),
1180                lit_token("y", 13, 14, TokenKind::Identifier),
1181            ],
1182            bytes: 20,
1183        };
1184        let cx_of = |format: &str| {
1185            compute_summary(
1186                &[file(format)],
1187                &[],
1188                10,
1189                SummaryMetric::Complexity,
1190                identity,
1191            )
1192            .files[0]
1193                .complexity
1194        };
1195        assert_eq!(cx_of("csharp"), 2, "the `if` after the escape is code");
1196        // The same run in Python really does open a string.
1197        assert_eq!(cx_of("python"), 1);
1198    }
1199
1200    fn lit_token(value: &str, start: u32, end: u32, kind: TokenKind) -> Token {
1201        Token {
1202            kind,
1203            value: value.to_string(),
1204            start: Location {
1205                line: 1,
1206                column: start,
1207                offset: start,
1208            },
1209            end: Location {
1210                line: 1,
1211                column: end,
1212                offset: end,
1213            },
1214        }
1215    }
1216
1217    #[test]
1218    fn a_name_the_file_binds_is_not_a_keyword() {
1219        // `case` and `when` are branch keywords somewhere and ordinary
1220        // variables elsewhere; the tokenizer tags both `Identifier`. A file
1221        // that assigns to the word has settled the question for itself.
1222        let code = "def run(c):\n case = 3\n when = 4\n return case + when\n";
1223        assert_eq!(cx("python", code), 1);
1224        // Reading it off an object is the same evidence.
1225        assert_eq!(cx("python", "def f(o):\n return o.case\n"), 1);
1226        // So is listing it as a parameter or an argument.
1227        let listed = "def headline(case, when):\n return fmt(case, when)\n";
1228        assert_eq!(cx("python", listed), 1);
1229        // Rust's `?` before a closing paren is still a branch: only words are
1230        // taken as names.
1231        let question = "fn f(s: &str) -> Result<u8, E> { Ok(parse(s)?) }";
1232        assert_eq!(cx("rust", question), 2);
1233        // Without that evidence the keyword still counts: one `def`, plus
1234        // `case` and `when`.
1235        let ruby = "def f(x)\n case x\n when 1 then 2\n end\nend";
1236        assert_eq!(cx("ruby", ruby), 3);
1237    }
1238
1239    #[test]
1240    fn the_c_family_declares_functions_without_a_keyword() {
1241        // `head(args) {` is the only marker C, C++, Java and C# give, and it
1242        // has to be told apart from the `) {` of a control statement.
1243        let code =
1244            "int add(int a, int b) { if (a > b) { return a; } return b; }\nvoid noop(void) { }\n";
1245        assert_eq!(cx("c", code), 3, "two functions plus one if");
1246        let java = "class T { int f(int a) { if (a > 0 && a < 10) { return a; } return 0; } }";
1247        assert_eq!(cx("java", java), 3, "one function, one if, one &&");
1248    }
1249
1250    #[test]
1251    fn a_parenthesised_statement_is_not_a_function() {
1252        // `switch (x) {` and `for (…) {` end in `) {` just like a signature.
1253        // One function plus the single `case`; the `switch` head adds neither.
1254        assert_eq!(
1255            cx(
1256                "c",
1257                "int f(int x) { switch (x) { case 1: return 1; } return 0; }"
1258            ),
1259            2
1260        );
1261        assert_eq!(
1262            cx("c", "void f(void) { for (int i = 0; i < 3; i++) { } }"),
1263            2
1264        );
1265        assert_eq!(cx("c", "void f(void) { while (x) { } }"), 2);
1266        // A language that declares functions by keyword is unaffected by this.
1267        assert_eq!(cx("go", "func f() { if x { } }"), 2);
1268    }
1269
1270    #[test]
1271    fn complexity_counts_decision_tokens() {
1272        let sources = vec![source(
1273            "a.js",
1274            "javascript",
1275            &["if", "x", "&&", "y", "for", "z", "else"],
1276            10,
1277        )];
1278        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1279        // 1 + (if, &&, for) = 4; "else" is not a decision point.
1280        assert_eq!(summary.files[0].complexity, 4);
1281    }
1282
1283    #[test]
1284    fn complexity_is_case_insensitive() {
1285        // SQL / PL/SQL / Fortran style uppercase keywords.
1286        let sources = vec![source(
1287            "a.sql",
1288            "sql",
1289            &["IF", "x", "OR", "y", "WHEN", "THEN", "If"],
1290            10,
1291        )];
1292        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1293        // 1 + (IF, OR, WHEN, If) = 5; THEN is not a decision point.
1294        assert_eq!(summary.files[0].complexity, 5);
1295    }
1296
1297    #[test]
1298    fn decision_token_edge_cases() {
1299        assert!(is_decision_token("unless"));
1300        assert!(is_decision_token("ELSEIF"));
1301        assert!(is_decision_token("andalso"));
1302        assert!(!is_decision_token(""));
1303        assert!(!is_decision_token("iffy"));
1304        assert!(!is_decision_token("conditionally"), "length-capped");
1305        assert!(!is_decision_token("форматирование"), "non-ASCII ignored");
1306    }
1307
1308    #[test]
1309    fn folder_rollup_uses_direct_parent() {
1310        let sources = vec![
1311            source("src/app/a.js", "javascript", &["x"], 5),
1312            source("src/app/b.js", "javascript", &["x", "y"], 5),
1313            source("src/c.js", "javascript", &["x"], 5),
1314            source("root.js", "javascript", &["x"], 5),
1315        ];
1316        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
1317        assert_eq!(summary.total_folders, 3);
1318        let app = summary
1319            .folders
1320            .iter()
1321            .find(|f| f.path == "src/app")
1322            .expect("src/app folder");
1323        assert_eq!(app.files, 2);
1324        assert_eq!(app.tokens, 3);
1325        let root = summary.folders.iter().find(|f| f.path == ".");
1326        assert!(root.is_some(), "root files grouped under '.'");
1327    }
1328
1329    #[test]
1330    fn duplication_attributed_to_both_fragments() {
1331        let sources = vec![
1332            source("a.js", "javascript", &["x", "y", "z"], 5),
1333            source("b.js", "javascript", &["x", "y", "z"], 5),
1334        ];
1335        let clones = vec![clone_between("javascript", "a.js", "b.js", 9, 30)];
1336        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, identity);
1337        for path in ["a.js", "b.js"] {
1338            let file = summary.files.iter().find(|f| f.path == path).unwrap();
1339            assert_eq!(file.duplicated_lines, 9, "{path} duplicated lines");
1340            assert_eq!(file.duplicated_tokens, 30, "{path} duplicated tokens");
1341        }
1342    }
1343
1344    #[test]
1345    fn synthetic_sub_format_sources_are_skipped() {
1346        let sources = vec![
1347            source("doc.md", "markdown", &["x", "y"], 100),
1348            source("doc.md:javascript", "javascript", &["x"], 0),
1349        ];
1350        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
1351        assert_eq!(summary.total_files, 1);
1352        assert_eq!(summary.files[0].path, "doc.md");
1353    }
1354
1355    #[test]
1356    fn sub_format_clone_folds_into_parent_file() {
1357        let sources = vec![source("doc.md", "markdown", &["x", "y"], 100)];
1358        let clones = vec![clone_between(
1359            "javascript",
1360            "doc.md:javascript",
1361            "doc.md:javascript",
1362            4,
1363            20,
1364        )];
1365        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, identity);
1366        assert_eq!(
1367            summary.files[0].duplicated_lines, 8,
1368            "both fragments fold in"
1369        );
1370    }
1371
1372    #[test]
1373    fn gap_lines_of_a_merged_clone_stay_out_of_file_duplication() {
1374        let sources = vec![
1375            source("a.js", "javascript", &["x"; 20], 10),
1376            source("b.js", "javascript", &["x"; 20], 10),
1377        ];
1378        let mut merged = clone_between("javascript", "a.js", "b.js", 10, 60);
1379        merged.unmatched_lines = [0, 3];
1380        let summary = compute_summary(&sources, &[merged], 10, SummaryMetric::Tokens, identity);
1381        let dup = |path: &str| {
1382            summary
1383                .files
1384                .iter()
1385                .find(|f| f.path == path)
1386                .unwrap()
1387                .duplicated_lines
1388        };
1389        assert_eq!(dup("a.js"), 10);
1390        assert_eq!(dup("b.js"), 7, "three gap lines in b are not duplicated");
1391    }
1392
1393    #[test]
1394    fn display_path_applied_before_dup_matching() {
1395        let sources = vec![source("/abs/root/a.js", "javascript", &["x"], 5)];
1396        let clones = vec![clone_between("javascript", "a.js", "a.js", 2, 10)];
1397        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, |p| {
1398            p.strip_prefix("/abs/root/").unwrap_or(p).to_string()
1399        });
1400        assert_eq!(summary.files[0].path, "a.js");
1401        assert_eq!(summary.files[0].duplicated_lines, 4);
1402    }
1403
1404    #[test]
1405    fn folders_truncated_to_top_n_but_total_reported() {
1406        let sources: Vec<SourceFile> = (0..5)
1407            .map(|i| source(&format!("dir{i}/f.js"), "javascript", &["x"], 1))
1408            .collect();
1409        let summary = compute_summary(&sources, &[], 2, SummaryMetric::Tokens, identity);
1410        assert_eq!(summary.folders.len(), 2);
1411        assert_eq!(summary.total_folders, 5);
1412    }
1413
1414    #[test]
1415    fn metric_parses_from_str() {
1416        assert_eq!(
1417            "complexity".parse::<SummaryMetric>().unwrap(),
1418            SummaryMetric::Complexity
1419        );
1420        assert!("bogus".parse::<SummaryMetric>().is_err());
1421    }
1422
1423    #[test]
1424    fn summary_serializes_camel_case() {
1425        let sources = vec![source("a.js", "javascript", &["x"], 5)];
1426        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Size, identity);
1427        let json = serde_json::to_string(&summary).unwrap();
1428        assert!(json.contains("\"totalFiles\""));
1429        assert!(json.contains("\"duplicatedLines\""));
1430        assert!(json.contains("\"by\":\"size\""));
1431        assert!(!json.contains("total_files"));
1432    }
1433}