Skip to main content

cpd_core/
summary.rs

1// summary.rs — opt-in codebase summary: per-file metrics, folder rollup, top-N lists.
2//
3// Everything in this module runs only when `--summary` is enabled, after
4// detection has finished, over data already held in memory (SourceFile tokens
5// and detected clones). Nothing in the detection hot path calls into it.
6
7use crate::models::{CpdClone, SourceFile, Token, TokenKind};
8use serde::{Deserialize, Serialize};
9use std::collections::HashMap;
10
11/// Metric used to rank files and folders in the summary.
12#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)]
13#[serde(rename_all = "lowercase")]
14pub enum SummaryMetric {
15    #[default]
16    Tokens,
17    Lines,
18    Size,
19    Complexity,
20}
21
22impl std::str::FromStr for SummaryMetric {
23    type Err = String;
24
25    fn from_str(s: &str) -> Result<Self, Self::Err> {
26        match s {
27            "tokens" => Ok(Self::Tokens),
28            "lines" => Ok(Self::Lines),
29            "size" => Ok(Self::Size),
30            "complexity" => Ok(Self::Complexity),
31            other => Err(format!(
32                "invalid summary metric '{other}': must be one of: tokens, lines, size, complexity"
33            )),
34        }
35    }
36}
37
38impl std::fmt::Display for SummaryMetric {
39    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
40        let s = match self {
41            Self::Tokens => "tokens",
42            Self::Lines => "lines",
43            Self::Size => "size",
44            Self::Complexity => "complexity",
45        };
46        f.write_str(s)
47    }
48}
49
50/// Per-file summary row.
51#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
52#[serde(rename_all = "camelCase")]
53pub struct FileSummary {
54    pub path: String,
55    pub format: String,
56    pub lines: u64,
57    pub tokens: u64,
58    pub bytes: u64,
59    pub duplicated_lines: u64,
60    pub duplicated_tokens: u64,
61    /// Cyclomatic-complexity estimate: 1 + count of decision-point tokens
62    /// (`if`, `for`, `while`, `case`, `catch`, `&&`, `||`, `?`, …).
63    pub complexity: u64,
64}
65
66/// Per-folder rollup. Files are counted in their direct parent directory only
67/// (no cumulative ancestor totals), so every file contributes to exactly one
68/// folder row and rows are directly comparable.
69#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
70#[serde(rename_all = "camelCase")]
71pub struct FolderSummary {
72    pub path: String,
73    pub files: u64,
74    pub lines: u64,
75    pub tokens: u64,
76    pub bytes: u64,
77    pub duplicated_lines: u64,
78    /// Sum of per-file complexity estimates (divide by `files` for the mean).
79    pub complexity: u64,
80}
81
82/// Codebase summary: top files and folder rollup.
83#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
84#[serde(rename_all = "camelCase")]
85pub struct Summary {
86    /// Primary sort metric.
87    pub by: SummaryMetric,
88    /// Top-N files by `by`, descending. Every row carries all metrics
89    /// (tokens, lines, bytes, complexity, duplication) so one list serves
90    /// every lens; re-run with a different `--summary-by` to re-rank.
91    pub files: Vec<FileSummary>,
92    /// Top-N folders by `by`, direct-parent aggregation.
93    pub folders: Vec<FolderSummary>,
94    /// Total number of files analyzed (before top-N truncation).
95    pub total_files: u64,
96    /// Total number of folders (before top-N truncation).
97    pub total_folders: u64,
98}
99
100/// Decision-point tokens counted by the complexity estimate. Conservative,
101/// language-agnostic list: branch/loop keywords and short-circuit operators
102/// that appear as standalone tokens across supported languages.
103///
104/// Matching is ASCII-case-insensitive so case-insensitive and
105/// uppercase-keyword languages (SQL, PL/SQL, Fortran, COBOL, BASIC, Pascal)
106/// count too. The occasional identifier spelled like a keyword slightly
107/// inflates an estimate that is only used for ranking.
108///
109/// Operators reach this function already joined — see [`joined_token`]. Only
110/// the JavaScript tokenizer emits `&&` as one token; the generic one splits
111/// every punctuation run into single characters, so without that joining the
112/// short-circuit arms here would be unreachable for every other format.
113fn is_decision_token(value: &str) -> bool {
114    let bytes = value.as_bytes();
115    if bytes.is_empty() || bytes.len() > 7 {
116        return false;
117    }
118    let mut lower = [0u8; 7];
119    for (dst, b) in lower.iter_mut().zip(bytes) {
120        *dst = b.to_ascii_lowercase();
121    }
122    matches!(
123        &lower[..bytes.len()],
124        b"if"
125            | b"elif"
126            | b"elsif"
127            | b"elseif"
128            | b"unless"
129            | b"for"
130            | b"foreach"
131            | b"while"
132            | b"until"
133            | b"case"
134            | b"cond"
135            | b"when"
136            | b"catch"
137            | b"rescue"
138            | b"except"
139            | b"andalso"
140            | b"orelse"
141            | b"&&"
142            | b"||"
143            | b"and"
144            | b"or"
145            | b"?"
146            | b"??"
147    )
148}
149
150/// What a bare `?` means in a language.
151#[derive(Clone, Copy, PartialEq, Eq)]
152enum QuestionMark {
153    /// It only ever opens a ternary: C, Java, PHP, Go templates, and most else.
154    Ternary,
155    /// It also marks an optional, so `String?` and `x?.y` are types and
156    /// accesses rather than branches. Neither TypeScript nor Swift has an
157    /// Elvis operator, so `?:` there is an optional property, not a branch.
158    Optional,
159    /// As above, but `?:` *is* the Elvis operator and does branch: Kotlin,
160    /// Groovy.
161    OptionalWithElvis,
162}
163
164/// How one language spells its branches, beyond the shared keyword set.
165///
166/// The shared set is right for most of the formats jscpd knows. An entry here
167/// exists only where a language branches on something the set has no word for
168/// (`match` arms in Rust, `guard` in Swift, `select` in Go) or spells
169/// something in the set so differently that counting it is simply wrong.
170#[derive(Clone, Copy)]
171struct DecisionRules {
172    /// Branch tokens beyond the shared set, matched after joining.
173    extra: &'static [&'static str],
174    question: QuestionMark,
175    /// Tokens that open a function body. Cyclomatic complexity is one path per
176    /// function; where a language has no single reliable marker this stays
177    /// empty and the estimate keeps its per-file baseline of one.
178    declarations: &'static [&'static str],
179    /// The C family declares a function as `head(args) {` with no keyword at
180    /// all, so its functions are counted from that shape instead.
181    braced_declarations: bool,
182    /// Whether `"""` and `'''` delimit a string that can span lines; see
183    /// [`has_triple_quoted_strings`].
184    triple_quoted_strings: bool,
185    /// Tokens that open a group of arms counted one by one through `extra`.
186    /// Each cancels one arm, because N arms are N paths and so N - 1 branches,
187    /// the same way a `switch` counts its `case` labels but not its `default`.
188    arm_groups: &'static [&'static str],
189}
190
191impl Default for DecisionRules {
192    fn default() -> Self {
193        Self {
194            extra: &[],
195            question: QuestionMark::Ternary,
196            declarations: &[],
197            braced_declarations: false,
198            triple_quoted_strings: false,
199            arm_groups: &[],
200        }
201    }
202}
203
204/// Formats whose strings can span lines between `"""` or `'''` delimiters.
205///
206/// Where a doubled quote is how a quote is escaped — C# verbatim strings,
207/// VB.NET, SQL, Pascal — three quotes in a row are ordinary string content: the
208/// regex `@"""((?:\\.|[^""\\])*)"""` is one line of C#. Reading such a run as
209/// a delimiter opens a string that never closes and swallows the rest of the
210/// file, which is why this is a list rather than a default.
211fn has_triple_quoted_strings(format: &str) -> bool {
212    matches!(
213        format,
214        "python" | "kotlin" | "scala" | "groovy" | "swift" | "java" | "julia" | "elixir" | "dart"
215    )
216}
217
218/// False for prose and data formats, whose "if" and `||` are words and
219/// version ranges rather than branches. One half of [`is_code`]: a format
220/// that can never have a branch can never have complexity above zero.
221pub fn has_control_flow(format: &str) -> bool {
222    !matches!(
223        format,
224        "markdown"
225            | "asciidoc"
226            | "rest"
227            | "textile"
228            | "wiki"
229            | "txt"
230            | "log"
231            | "csv"
232            | "json"
233            | "json5"
234            | "yaml"
235            | "toml"
236            | "ini"
237            | "properties"
238            | "editorconfig"
239            | "ignore"
240            | "diff"
241            | "gettext"
242    )
243}
244
245/// Markup, stylesheets, declarative schemas and the templating languages
246/// built on top of markup: a duplicated template or style rule repeating is
247/// not the maintenance problem duplicated programming logic is, so it does
248/// not count toward the health score's duplication share at all — the same
249/// treatment prose and data files get, just decided per clone rather than
250/// per file, since a `.svelte` or `.vue` file's markup and style blocks are
251/// tokenized separately from its script block. The other half of
252/// [`is_code`]: whole files in these formats have no complexity either — the
253/// "if" in an HTML attribute and the "and" in a media query are words, not
254/// branches.
255///
256/// These are the tokenizer's own format *names*
257/// (`cpd-tokenizer/src/formats.rs`), not file extensions: html/htm/xml/svg
258/// all tokenize as `markup` (`html` is the name a component file's markup
259/// block and a Markdown html snippet carry), `.puml`/`.plantuml` as
260/// `plant-uml`, `.tpl` as `smarty`, `.jade` as `pug`, and `.vtl` as
261/// `velocity` — matching on the extension instead of the name a clone's
262/// `format` field actually carries would silently never exclude anything.
263pub fn is_markup(format: &str) -> bool {
264    matches!(
265        format,
266        "markup"
267            | "html"
268            | "css"
269            | "scss"
270            | "sass"
271            | "less"
272            | "stylus"
273            | "razor"
274            | "haml"
275            | "pug"
276            | "handlebars"
277            | "erb"
278            | "liquid"
279            | "twig"
280            | "velocity"
281            | "ftl"
282            | "soy"
283            | "smarty"
284            | "tt2"
285            | "protobuf"
286            | "plant-uml"
287            | "mermaid"
288            | "django"
289            | "aspnet"
290    )
291}
292
293/// Whether a format counts as the project's code: prose and data
294/// ([`has_control_flow`]) and markup ([`is_markup`]) do not. This one test
295/// decides where complexity can be above zero, and through that which files
296/// are `health::compute`'s "code files"; a format-level duplication
297/// breakdown leaves the same formats out.
298pub fn is_code(format: &str) -> bool {
299    has_control_flow(format) && !is_markup(format)
300}
301
302fn rules_for(format: &str) -> DecisionRules {
303    DecisionRules {
304        triple_quoted_strings: has_triple_quoted_strings(format),
305        ..language_rules(format)
306    }
307}
308
309fn language_rules(format: &str) -> DecisionRules {
310    match format {
311        // A `match` has no keyword per arm, only the `=>` each one is written
312        // with. Every arm is counted and the `match` itself takes one back: a
313        // three-arm match is three paths, which is two branches.
314        "rust" => DecisionRules {
315            extra: &["=>"],
316            declarations: &["fn"],
317            arm_groups: &["match"],
318            ..Default::default()
319        },
320        "swift" => DecisionRules {
321            extra: &["guard"],
322            question: QuestionMark::Optional,
323            declarations: &["func"],
324            ..Default::default()
325        },
326        // `=>` opens an arrow function here rather than a branch, so it counts
327        // toward the function tally instead.
328        "typescript" | "tsx" | "flow" | "javascript" | "jsx" => DecisionRules {
329            question: QuestionMark::Optional,
330            declarations: &["function", "=>"],
331            ..Default::default()
332        },
333        "kotlin" => DecisionRules {
334            question: QuestionMark::OptionalWithElvis,
335            declarations: &["fun"],
336            ..Default::default()
337        },
338        "groovy" => DecisionRules {
339            question: QuestionMark::OptionalWithElvis,
340            declarations: &["def"],
341            ..Default::default()
342        },
343        "csharp" => DecisionRules {
344            question: QuestionMark::Optional,
345            braced_declarations: true,
346            ..Default::default()
347        },
348        "c" | "c-header" | "cpp" | "cpp-header" | "java" | "objectivec" | "clike" => {
349            DecisionRules {
350                braced_declarations: true,
351                ..Default::default()
352            }
353        }
354        "go" => DecisionRules {
355            extra: &["select"],
356            declarations: &["func"],
357            ..Default::default()
358        },
359        "scala" | "python" | "ruby" | "crystal" => DecisionRules {
360            declarations: &["def"],
361            ..Default::default()
362        },
363        "php" | "lua" => DecisionRules {
364            declarations: &["function"],
365            ..Default::default()
366        },
367        "erlang" => DecisionRules {
368            extra: &["receive"],
369            ..Default::default()
370        },
371        _ => DecisionRules::default(),
372    }
373}
374
375/// Two-character operators the generic tokenizer hands over as two tokens.
376const JOINED_OPERATORS: &[&str] = &["&&", "||", "??", "?.", "?:", "=>"];
377
378/// One token's text, or the two-character operator that two adjacent tokens
379/// spell between them.
380///
381/// The pair cannot borrow: each token owns its own `String`, so two adjacent
382/// characters of the source are not adjacent in memory. Two bytes on the stack
383/// avoid an allocation per operator.
384enum Scanned<'a> {
385    Single(&'a str),
386    Pair([u8; 2]),
387}
388
389impl Scanned<'_> {
390    fn text(&self) -> &str {
391        match self {
392            Self::Single(text) => text,
393            Self::Pair(bytes) => std::str::from_utf8(bytes).unwrap_or(""),
394        }
395    }
396}
397
398/// The token at `at`, joined with the next one when the two are adjacent
399/// single-character punctuation spelling a two-character operator.
400///
401/// The generic tokenizer splits `&&` into two `&` tokens, so a scan that looks
402/// at one token at a time can never see a short-circuit operator at all.
403/// Returns what to classify and the index to continue from.
404fn joined_token(tokens: &[Token], at: usize) -> (Scanned<'_>, usize) {
405    let current = &tokens[at];
406    let single = (Scanned::Single(current.value.as_str()), at + 1);
407    let Some(next) = tokens.get(at + 1) else {
408        return single;
409    };
410    let (Some(&left), Some(&right)) = (
411        current.value.as_bytes().first(),
412        next.value.as_bytes().first(),
413    ) else {
414        return single;
415    };
416    // Adjacent in the source, and both exactly one punctuation character.
417    if current.end.offset != next.start.offset
418        || current.value.len() != 1
419        || next.value.len() != 1
420        || !left.is_ascii_punctuation()
421        || !right.is_ascii_punctuation()
422    {
423        return single;
424    }
425    let pair = [left, right];
426    match std::str::from_utf8(&pair).is_ok_and(|text| JOINED_OPERATORS.contains(&text)) {
427        true => (Scanned::Pair(pair), at + 2),
428        false => single,
429    }
430}
431
432/// Which tokens lie inside a triple-quoted string.
433///
434/// The tokenizer is line-based: it closes every string at the end of the line
435/// it started on, so the body of a `"""` string arrives as ordinary words, and
436/// a docstring saying "if the value is big" lands three branches on the file.
437///
438/// The quotes themselves survive as literal tokens, and a run of adjacent
439/// literals spells exactly the quote characters its line holds: a PEP 257
440/// opener `"""Summary.` arrives as `""` and `"Summary.`, a closer on its own
441/// line as `""` and `"`, and a one-line docstring as `""`, `"One."`, `""`. So
442/// an odd number of triple quotes in a run opens or closes the string, and an
443/// even number leaves it as it was.
444///
445/// Reading a pair of tokens alone is not enough, and was the first version of
446/// this: with the summary written on the opening line the opener is not a bare
447/// quote, so the *closing* `"""` looked like an opener and swallowed every
448/// branch that followed it.
449fn triple_quoted(tokens: &[Token]) -> Vec<bool> {
450    const DELIMITERS: [&str; 2] = ["\"\"\"", "'''"];
451    let mut inside = vec![false; tokens.len()];
452    let mut open: Option<&str> = None;
453    let mut at = 0usize;
454    while at < tokens.len() {
455        if tokens[at].kind != TokenKind::Literal {
456            inside[at] = open.is_some();
457            at += 1;
458            continue;
459        }
460        let start = at;
461        let mut text = tokens[at].value.clone();
462        at += 1;
463        while at < tokens.len()
464            && tokens[at].kind == TokenKind::Literal
465            && tokens[at - 1].end.offset == tokens[at].start.offset
466        {
467            text.push_str(&tokens[at].value);
468            at += 1;
469        }
470        let was_open = open.is_some();
471        for delimiter in DELIMITERS {
472            let toggles = text.matches(delimiter).count() % 2 == 1;
473            match open {
474                Some(current) if current == delimiter && toggles => open = None,
475                None if toggles => open = Some(delimiter),
476                _ => {}
477            }
478        }
479        // The run is string text whichever way it turned the state.
480        inside[start..at].fill(was_open || open.is_some());
481    }
482    inside
483}
484
485/// Words a file binds as names of its own.
486///
487/// `case`, `when` and `cond` are branch keywords in some languages and
488/// perfectly ordinary variable names in others — the shared list cannot tell
489/// them apart, and the tokenizer is no help: its `classify_word` returns
490/// `Identifier` for every word, so `if` in Python is tagged exactly like a
491/// variable called `case`.
492///
493/// What a file *does* reveal is which words it assigns to, reads off an
494/// object, or lists as a parameter or argument. A word this file writes
495/// `case = 3`, `x.case` or `f(case, when)` for is that file's own name,
496/// whatever the language reserves, and counting it as a branch is what
497/// inflated the estimate elevenfold on a file of five such assignments.
498fn locally_bound(tokens: &[Token]) -> Vec<&str> {
499    let mut bound = Vec::new();
500    for (at, token) in tokens.iter().enumerate() {
501        if !is_decision_token(&token.value) {
502            continue;
503        }
504        // `obj.case` — a member, never the keyword.
505        let after_dot = at
506            .checked_sub(1)
507            .is_some_and(|previous| tokens[previous].value == ".");
508        // `case = 3`, but not `case == 3` or `case => 3`.
509        let assigned = tokens.get(at + 1).is_some_and(|next| next.value == "=")
510            && tokens
511                .get(at + 2)
512                .is_none_or(|after| !matches!(after.value.as_str(), "=" | ">"));
513        // `def headline(case, when)` or `headline(case, when)`: a keyword is
514        // never written directly before a comma or a closing paren, a name in
515        // a parameter or argument list always is. Words only — `Ok(parse()?)`
516        // is Rust's `?` doing its job, not a name.
517        let listed = token.value.starts_with(|c: char| c.is_ascii_alphabetic())
518            && tokens
519                .get(at + 1)
520                .is_some_and(|next| matches!(next.value.as_str(), "," | ")"));
521        if after_dot || assigned || listed {
522            bound.push(token.value.as_str());
523        }
524    }
525    bound.sort_unstable();
526    bound.dedup();
527    bound
528}
529
530/// Keywords that take a parenthesised head and a block, so that `) {` after
531/// one of them opens a branch rather than a function body.
532const PARENTHESISED_STATEMENTS: &[&str] = &[
533    "switch",
534    "using",
535    "lock",
536    "synchronized",
537    "with",
538    "do",
539    "try",
540    "return",
541    "sizeof",
542    "typeof",
543    "new",
544    "throw",
545    "await",
546    "yield",
547    "fixed",
548    "unsafe",
549];
550
551/// Decision points and function count for one file.
552///
553/// Cyclomatic complexity is one path per function plus one per branch. Where a
554/// language has no reliable function marker the tally falls back to the
555/// per-file baseline of one, which is what this estimate has always used.
556fn scan_complexity(tokens: &[Token], rules: &DecisionRules) -> (u64, u64) {
557    let (mut decisions, mut functions, mut groups) = (0u64, 0u64, 0u64);
558    let mut at = 0usize;
559    let in_string = match rules.triple_quoted_strings {
560        true => triple_quoted(tokens),
561        false => vec![false; tokens.len()],
562    };
563    let bound = locally_bound(tokens);
564    // What the token before each open paren was, so a `) {` can be told from
565    // the head it closes: a function signature, or `if (…) {`.
566    let mut heads: Vec<&str> = Vec::new();
567    let mut last_head: Option<&str> = None;
568    while at < tokens.len() {
569        // Skip the body of a string that spans lines; see `triple_quoted`.
570        if in_string[at] {
571            at += 1;
572            continue;
573        }
574        let (scanned, next) = joined_token(tokens, at);
575        let text = scanned.text();
576        // `String?` writes the optional against the type it belongs to;
577        // `cond ? a : b` puts the ternary in the open. Nothing else separates
578        // the two without parsing, and the convention is near-universal.
579        let attached = at
580            .checked_sub(1)
581            .and_then(|previous| tokens.get(previous))
582            .is_some_and(|previous| previous.end.offset == tokens[at].start.offset);
583        at = next;
584        // C, C++, Java and C# open a function with `) {` and no keyword of
585        // their own. Tracking what preceded each `(` is enough to tell that
586        // from the `) {` of an `if` or a `switch`, and it is the only reason
587        // those languages kept a per-file baseline.
588        if rules.braced_declarations {
589            match text {
590                "(" => {
591                    let head = at
592                        .checked_sub(2)
593                        .and_then(|before| tokens.get(before))
594                        .map(|token| token.value.as_str())
595                        .unwrap_or("");
596                    heads.push(head);
597                }
598                ")" => last_head = heads.pop(),
599                "{" => {
600                    if let Some(head) = last_head.take()
601                        && !head.is_empty()
602                        && head.chars().all(|c| c.is_alphanumeric() || c == '_')
603                        && !is_decision_token(head)
604                        && !PARENTHESISED_STATEMENTS
605                            .iter()
606                            .any(|s| s.eq_ignore_ascii_case(head))
607                    {
608                        functions += 1;
609                    }
610                }
611                _ => last_head = None,
612            }
613        }
614        if rules
615            .declarations
616            .iter()
617            .any(|d| d.eq_ignore_ascii_case(text))
618        {
619            functions += 1;
620            continue;
621        }
622        if rules
623            .arm_groups
624            .iter()
625            .any(|g| g.eq_ignore_ascii_case(text))
626        {
627            groups += 1;
628            continue;
629        }
630        if rules.extra.iter().any(|e| e.eq_ignore_ascii_case(text)) {
631            decisions += 1;
632            continue;
633        }
634        // A word this file binds as a name is that file's name, not a keyword.
635        if bound.binary_search(&text).is_ok() {
636            continue;
637        }
638        let counts = match text {
639            // `?.` never branches, and `?:` branches only where it is the
640            // Elvis operator rather than an optional property.
641            "?." => false,
642            "?:" => rules.question != QuestionMark::Optional,
643            "?" => rules.question == QuestionMark::Ternary || !attached,
644            other => is_decision_token(other),
645        };
646        if counts {
647            decisions += 1;
648        }
649    }
650    (decisions.saturating_sub(groups), functions)
651}
652
653/// A synthetic source is the per-sub-format shadow of a multi-format file
654/// (markdown/vue/svelte embedded code); its id is `<parent-id>:<format>` and
655/// its metrics are already covered by the parent entry.
656fn is_synthetic(source: &SourceFile) -> bool {
657    source
658        .id
659        .strip_suffix(source.format.as_str())
660        .is_some_and(|prefix| prefix.ends_with(':'))
661}
662
663fn metric_of(file: &FileSummary, by: SummaryMetric) -> u64 {
664    match by {
665        SummaryMetric::Tokens => file.tokens,
666        SummaryMetric::Lines => file.lines,
667        SummaryMetric::Size => file.bytes,
668        SummaryMetric::Complexity => file.complexity,
669    }
670}
671
672fn folder_metric_of(folder: &FolderSummary, by: SummaryMetric) -> u64 {
673    match by {
674        SummaryMetric::Tokens => folder.tokens,
675        SummaryMetric::Lines => folder.lines,
676        SummaryMetric::Size => folder.bytes,
677        SummaryMetric::Complexity => folder.complexity,
678    }
679}
680
681/// Parent directory of a path, with separators normalized to `/`.
682/// Files at the scan root map to `"."`.
683fn parent_dir(path: &str) -> String {
684    let normalized = path.replace('\\', "/");
685    match normalized.rfind('/') {
686        Some(0) => "/".to_string(),
687        Some(idx) => normalized[..idx].to_string(),
688        None => ".".to_string(),
689    }
690}
691
692/// Compute the summary from detection results.
693///
694/// `display_path` maps a source id (canonical absolute path) to the path shown
695/// in reports — the same relativization applied to clone fragments, so
696/// per-file duplication matching works on identical strings.
697pub fn compute_summary(
698    sources: &[SourceFile],
699    clones: &[CpdClone],
700    top: usize,
701    by: SummaryMetric,
702    display_path: impl Fn(&str) -> String,
703) -> Summary {
704    // Per-file duplication, keyed by display path. Both fragments of a clone
705    // count toward their file: the question here is "where does duplicated
706    // code live", not the de-duplicated total that Statistics reports.
707    let mut dup: HashMap<String, (u64, u64)> = HashMap::new();
708    for clone in clones {
709        for (index, fragment) in [&clone.fragment_a, &clone.fragment_b]
710            .into_iter()
711            .enumerate()
712        {
713            // Sub-format fragments carry a `<path>:<format>` id; fold them
714            // into the parent file.
715            let path = fragment
716                .source_id
717                .strip_suffix(&format!(":{}", clone.format))
718                .unwrap_or(&fragment.source_id);
719            let entry = dup.entry(path.to_string()).or_default();
720            entry.0 += clone.fragment_lines(index);
721            entry.1 += clone.token_count as u64;
722        }
723    }
724
725    let mut files: Vec<FileSummary> = sources
726        .iter()
727        .filter(|s| !is_synthetic(s))
728        .map(|source| {
729            let path = display_path(&source.id);
730            // Same line metric as Statistics: max token start line.
731            let lines = source
732                .tokens
733                .iter()
734                .map(|t| t.start.line)
735                .max()
736                .unwrap_or(0) as u64;
737            let (decisions, functions) =
738                scan_complexity(&source.tokens, &rules_for(&source.format));
739            let (duplicated_lines, duplicated_tokens) = dup.get(&path).copied().unwrap_or_default();
740            FileSummary {
741                lines,
742                tokens: source.tokens.len() as u64,
743                bytes: source.bytes,
744                duplicated_lines,
745                duplicated_tokens,
746                // One path per function, or the per-file baseline where the
747                // language has no marker the scan can trust. Prose, data and
748                // markup have no paths: an "if" in a README or an HTML
749                // attribute is a word, and a lock file full of `||` version
750                // ranges is not code.
751                complexity: match is_code(&source.format) {
752                    true => functions.max(1) + decisions,
753                    false => 0,
754                },
755                format: source.format.clone(),
756                path,
757            }
758        })
759        .collect();
760
761    let total_files = files.len() as u64;
762
763    // Folder rollup over ALL files (before top-N truncation).
764    let mut folder_map: HashMap<String, FolderSummary> = HashMap::new();
765    for file in &files {
766        let dir = parent_dir(&file.path);
767        let entry = folder_map
768            .entry(dir.clone())
769            .or_insert_with(|| FolderSummary {
770                path: dir,
771                files: 0,
772                lines: 0,
773                tokens: 0,
774                bytes: 0,
775                duplicated_lines: 0,
776                complexity: 0,
777            });
778        entry.files += 1;
779        entry.lines += file.lines;
780        entry.tokens += file.tokens;
781        entry.bytes += file.bytes;
782        entry.duplicated_lines += file.duplicated_lines;
783        entry.complexity += file.complexity;
784    }
785    let total_folders = folder_map.len() as u64;
786
787    // Top-N files by the primary metric: `--summary-top N` always yields at
788    // most N rows (least surprise). Other lenses are one `--summary-by` away;
789    // every row still carries all metrics.
790    files.sort_by(|a, b| {
791        metric_of(b, by)
792            .cmp(&metric_of(a, by))
793            .then_with(|| a.path.cmp(&b.path))
794    });
795    files.truncate(top);
796
797    let mut folders: Vec<FolderSummary> = folder_map.into_values().collect();
798    folders.sort_by(|a, b| {
799        folder_metric_of(b, by)
800            .cmp(&folder_metric_of(a, by))
801            .then_with(|| a.path.cmp(&b.path))
802    });
803    folders.truncate(top);
804
805    Summary {
806        by,
807        files,
808        folders,
809        total_files,
810        total_folders,
811    }
812}
813
814#[cfg(test)]
815mod tests {
816    use super::*;
817    use crate::models::{CpdClone, Fragment, Location, Token, TokenKind};
818
819    fn loc(line: u32) -> Location {
820        Location {
821            line,
822            column: 0,
823            offset: 0,
824        }
825    }
826
827    fn token(value: &str, line: u32) -> Token {
828        Token {
829            kind: TokenKind::Keyword,
830            value: value.to_string(),
831            start: loc(line),
832            end: loc(line),
833        }
834    }
835
836    fn source(id: &str, format: &str, values: &[&str], bytes: u64) -> SourceFile {
837        SourceFile {
838            id: id.to_string(),
839            format: format.to_string(),
840            tokens: values
841                .iter()
842                .enumerate()
843                .map(|(i, v)| token(v, i as u32 + 1))
844                .collect(),
845            bytes,
846        }
847    }
848
849    /// A clone whose fragments both cover `lines` lines, counted the way the
850    /// statistics count them: line 1 through line `lines`, both ends included.
851    fn clone_between(format: &str, a: &str, b: &str, lines: u32, tokens: u32) -> CpdClone {
852        let fragment = |id: &str| Fragment {
853            source_id: id.to_string(),
854            source_root: None,
855            start: loc(1),
856            end: loc(lines),
857            range: [0, tokens],
858            blame: None,
859        };
860        CpdClone {
861            format: format.to_string(),
862            fragment_a: fragment(a),
863            fragment_b: fragment(b),
864            token_count: tokens,
865            is_new: false,
866            kind: Default::default(),
867            similarity: None,
868            similarity_method: None,
869            unmatched_lines: [0, 0],
870        }
871    }
872
873    fn identity(path: &str) -> String {
874        path.to_string()
875    }
876
877    #[test]
878    fn empty_input_produces_empty_summary() {
879        let summary = compute_summary(&[], &[], 10, SummaryMetric::Tokens, identity);
880        assert!(summary.files.is_empty());
881        assert!(summary.folders.is_empty());
882        assert_eq!(summary.total_files, 0);
883        assert_eq!(summary.total_folders, 0);
884    }
885
886    #[test]
887    fn prose_and_data_have_no_complexity() {
888        let words = [
889            "If", "you", "need", "it", "or", "while", "waiting", "for", "a", "case",
890        ];
891        let sources = vec![
892            source("README.md", "markdown", &words, 10),
893            source(
894                "pnpm-lock.yaml",
895                "yaml",
896                &["version", ":", "^1", "||", "^2"],
897                10,
898            ),
899            source("guide.rst", "rest", &words, 10),
900            source("notes.py", "python", &words, 10),
901        ];
902        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
903        let cx = |path: &str| {
904            summary
905                .files
906                .iter()
907                .find(|f| f.path == path)
908                .unwrap()
909                .complexity
910        };
911        assert_eq!(cx("README.md"), 0, "a word is not a branch");
912        assert_eq!(cx("pnpm-lock.yaml"), 0, "a version range is not a branch");
913        assert_eq!(cx("guide.rst"), 0, "reStructuredText is prose too");
914        assert!(cx("notes.py") > 1, "the same words in code still count");
915    }
916
917    #[test]
918    fn markup_styles_and_templates_have_no_complexity() {
919        // The same exclusion the duplication share applies (`is_markup`):
920        // an "if" in markup text or a media query's "and" is a word, not a
921        // branch. `markup` is what html/xml files tokenize as; `html` is the
922        // name a component file's markup block carries.
923        let words = [
924            "<", "a", ">", "If", "you", "click", "or", "wait", "<", "/", "a", ">",
925        ];
926        let sources = vec![
927            source("index.html", "markup", &words, 10),
928            source("snippet.html", "html", &words, 10),
929            source(
930                "site.css",
931                "css",
932                &["@", "media", "screen", "and", "(", "print", ")"],
933                10,
934            ),
935            source("card.twig", "twig", &["{", "%", "if", "user", "%", "}"], 10),
936        ];
937        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
938        assert!(
939            summary.files.iter().all(|f| f.complexity == 0),
940            "markup formats must have complexity 0: {:?}",
941            summary.files
942        );
943    }
944
945    #[test]
946    fn files_sorted_by_primary_metric() {
947        let sources = vec![
948            source("src/small.js", "javascript", &["a", "b"], 10),
949            source("src/big.js", "javascript", &["a", "b", "c", "d"], 20),
950        ];
951        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
952        assert_eq!(summary.files[0].path, "src/big.js");
953        assert_eq!(summary.files[0].tokens, 4);
954        assert_eq!(summary.total_files, 2);
955    }
956
957    #[test]
958    fn top_n_is_exact_row_count_by_primary_metric() {
959        // huge.js wins on tokens, fat.js wins on size — top=1 by tokens must
960        // yield exactly one row: huge.js. `--summary-top N` never surprises
961        // with more than N rows; other metrics are served by --summary-by.
962        let sources = vec![
963            source("huge.js", "javascript", &["a", "b", "c", "d", "e"], 1),
964            source("fat.js", "javascript", &["a"], 9999),
965        ];
966        let summary = compute_summary(&sources, &[], 1, SummaryMetric::Tokens, identity);
967        assert_eq!(summary.files.len(), 1);
968        assert_eq!(summary.files[0].path, "huge.js");
969        assert_eq!(summary.total_files, 2, "truncation stays visible");
970
971        let by_size = compute_summary(&sources, &[], 1, SummaryMetric::Size, identity);
972        assert_eq!(by_size.files[0].path, "fat.js");
973    }
974
975    /// Tokens laid out over a real source string the way the generic
976    /// tokenizer hands them over: one per word, one per punctuation
977    /// character, with the offsets that make adjacency visible.
978    fn lay_out(id: &str, format: &str, code: &str) -> SourceFile {
979        let mut tokens = Vec::new();
980        let bytes = code.as_bytes();
981        let mut at = 0usize;
982        while at < bytes.len() {
983            let byte = bytes[at];
984            if byte.is_ascii_whitespace() {
985                at += 1;
986                continue;
987            }
988            let start = at;
989            if byte.is_ascii_alphanumeric() || byte == b'_' {
990                while at < bytes.len() && (bytes[at].is_ascii_alphanumeric() || bytes[at] == b'_') {
991                    at += 1;
992                }
993            } else {
994                at += 1;
995            }
996            let at32 = |offset: usize| Location {
997                line: 1,
998                column: offset as u32,
999                offset: offset as u32,
1000            };
1001            tokens.push(Token {
1002                kind: TokenKind::Identifier,
1003                value: code[start..at].to_string(),
1004                start: at32(start),
1005                end: at32(at),
1006            });
1007        }
1008        SourceFile {
1009            id: id.to_string(),
1010            format: format.to_string(),
1011            tokens,
1012            bytes: code.len() as u64,
1013        }
1014    }
1015
1016    fn cx(format: &str, code: &str) -> u64 {
1017        let sources = vec![lay_out("a", format, code)];
1018        compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity).files[0].complexity
1019    }
1020
1021    #[test]
1022    fn short_circuit_operators_count_when_split_into_characters() {
1023        // The generic tokenizer hands `&&` over as two `&` tokens, so without
1024        // joining them the short-circuit arms of the shared list never match
1025        // for any format but JavaScript.
1026        assert_eq!(cx("c", "int f() { return a && b; }"), 2);
1027        assert_eq!(cx("c", "int f() { return a || b; }"), 2);
1028        assert_eq!(cx("c", "int f() { return a && b || c; }"), 3);
1029        // A single `&` is a bitwise and, not a branch.
1030        assert_eq!(cx("c", "int f() { return a & b; }"), 1);
1031    }
1032
1033    #[test]
1034    fn an_optional_is_not_a_ternary() {
1035        // `String?` writes the question mark against its type; a ternary puts
1036        // it in the open. Nothing else tells them apart without parsing.
1037        assert_eq!(cx("swift", "func f(x: String?) -> Int { return 1 }"), 1);
1038        assert_eq!(cx("swift", "func f(x: Foo) { let y = x?.bar }"), 1);
1039        assert_eq!(
1040            cx("swift", "func f(x: Int) -> Int { return x > 0 ? 1 : 2 }"),
1041            2
1042        );
1043        // Nil-coalescing is a branch in any spelling.
1044        assert_eq!(
1045            cx("swift", "func f(a: Int?, b: Int) -> Int { return a ?? b }"),
1046            2
1047        );
1048        // A language where `?` only ever opens a ternary is unaffected.
1049        assert_eq!(cx("c", "int f(int x) { return x ? 1 : 2; }"), 2);
1050    }
1051
1052    #[test]
1053    fn a_match_arm_is_the_branch_not_the_match() {
1054        // Three arms are three paths: one for the function, two branches.
1055        let three_arms = "fn f(x: i32) -> i32 { match x { 0 => 1, 1 => 2, _ => 3 } }";
1056        assert_eq!(cx("rust", three_arms), 3);
1057        // A guard is a branch of its own on top of the arm it guards.
1058        let guarded = "fn f(x: i32, ok: bool) -> i32 { match x { 0 if ok => 1, 0 => 2, _ => 3 } }";
1059        assert_eq!(cx("rust", guarded), 4);
1060        // A match with a single arm does not branch at all.
1061        assert_eq!(cx("rust", "fn f(x: i32) -> i32 { match x { _ => 0 } }"), 1);
1062        // `=>` opens an arrow function in JavaScript, so it is a declaration
1063        // there rather than a branch.
1064        assert_eq!(cx("javascript", "const f = (x) => x + 1;"), 1);
1065    }
1066
1067    #[test]
1068    fn complexity_counts_one_path_per_function() {
1069        // Four functions, three of them with a single branch.
1070        let code = "def a(n):\n if n: pass\ndef b(n):\n if n: pass\ndef c(n):\n if n: pass\ndef d(n):\n pass\n";
1071        assert_eq!(cx("python", code), 7);
1072        // A language with no marker the scan can trust keeps the per-file
1073        // baseline of one rather than guessing.
1074        assert_eq!(cx("cobol", "IF x THEN y"), 2);
1075    }
1076
1077    #[test]
1078    fn guard_and_select_are_branches_where_they_exist() {
1079        assert_eq!(
1080            cx("swift", "func f(x: Int) { guard x > 0 else { return } }"),
1081            2
1082        );
1083        assert_eq!(cx("go", "func f() { select { } }"), 2);
1084        // `guard` is an ordinary word elsewhere.
1085        assert_eq!(cx("c", "int f() { int guard = 1; return guard; }"), 1);
1086    }
1087
1088    #[test]
1089    fn elvis_branches_only_where_the_language_has_one() {
1090        // Kotlin's `?:` is the elvis operator.
1091        assert_eq!(
1092            cx("kotlin", "fun f(a: Int?, b: Int): Int { return a ?: b }"),
1093            2
1094        );
1095        // TypeScript has no elvis; `a?: T` marks an optional property.
1096        assert_eq!(cx("typescript", "function f(a?: string) { return a; }"), 1);
1097    }
1098
1099    #[test]
1100    fn a_docstring_is_not_a_pile_of_branches() {
1101        // The real tokenizer closes every string at the end of its line, so a
1102        // docstring leaves a literal holding just its opening quote and its
1103        // body arrives as ordinary words.
1104        let quote = "\u{22}";
1105        let tokens = vec![
1106            lit_token("def", 0, 3, TokenKind::Identifier),
1107            lit_token("f", 4, 5, TokenKind::Identifier),
1108            lit_token(&quote.repeat(2), 6, 8, TokenKind::Literal),
1109            lit_token(quote, 8, 9, TokenKind::Literal),
1110            lit_token("if", 10, 12, TokenKind::Identifier),
1111            lit_token("for", 13, 16, TokenKind::Identifier),
1112            lit_token("while", 17, 22, TokenKind::Identifier),
1113            lit_token(&quote.repeat(2), 23, 25, TokenKind::Literal),
1114            lit_token(quote, 25, 26, TokenKind::Literal),
1115            lit_token("if", 27, 29, TokenKind::Identifier),
1116        ];
1117        let sources = vec![SourceFile {
1118            id: "a".into(),
1119            format: "python".into(),
1120            tokens,
1121            bytes: 30,
1122        }];
1123        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1124        // One `def`, and only the `if` outside the docstring.
1125        assert_eq!(summary.files[0].complexity, 2);
1126    }
1127
1128    #[test]
1129    fn a_docstring_with_its_summary_on_the_opening_line_closes_where_it_ends() {
1130        // PEP 257 style, as the tokenizer really hands it over: the opener is
1131        // `""` + `"Summary.` rather than a bare quote, so a reader that looked
1132        // for `""` + `"` alone took the *closing* quotes for an opener and
1133        // swallowed every branch after them.
1134        let q = "\u{22}";
1135        let tokens = vec![
1136            lit_token("def", 0, 3, TokenKind::Identifier),
1137            lit_token("f", 4, 5, TokenKind::Identifier),
1138            lit_token(&q.repeat(2), 10, 12, TokenKind::Literal),
1139            lit_token(&format!("{q}Summary."), 12, 21, TokenKind::Literal),
1140            lit_token("if", 30, 32, TokenKind::Identifier),
1141            lit_token("and", 33, 36, TokenKind::Identifier),
1142            lit_token(&q.repeat(2), 40, 42, TokenKind::Literal),
1143            lit_token(q, 42, 43, TokenKind::Literal),
1144            lit_token("if", 50, 52, TokenKind::Identifier),
1145            lit_token("x", 53, 54, TokenKind::Identifier),
1146            // A one-line docstring is balanced on its own line.
1147            lit_token(&q.repeat(2), 60, 62, TokenKind::Literal),
1148            lit_token(&format!("{q}One.{q}"), 62, 68, TokenKind::Literal),
1149            lit_token(&q.repeat(2), 68, 70, TokenKind::Literal),
1150            lit_token("for", 75, 78, TokenKind::Identifier),
1151        ];
1152        let sources = vec![SourceFile {
1153            id: "a".into(),
1154            format: "python".into(),
1155            tokens,
1156            bytes: 80,
1157        }];
1158        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1159        // One `def`, plus the `if` and the `for` written as code.
1160        assert_eq!(summary.files[0].complexity, 3);
1161    }
1162
1163    #[test]
1164    fn a_doubled_quote_escape_is_not_a_triple_quoted_string() {
1165        // C# verbatim strings escape a quote by doubling it, so a run spelling
1166        // `"""x` is a quote followed by `x`, not a string that runs on. Read
1167        // as one, it swallowed the rest of a 413-branch file.
1168        let q = "\u{22}";
1169        let file = |format: &str| SourceFile {
1170            id: "a".into(),
1171            format: format.into(),
1172            tokens: vec![
1173                lit_token("x", 0, 1, TokenKind::Identifier),
1174                lit_token(&q.repeat(2), 4, 6, TokenKind::Literal),
1175                lit_token(&format!("{q}x"), 6, 8, TokenKind::Literal),
1176                lit_token("if", 10, 12, TokenKind::Identifier),
1177                lit_token("y", 13, 14, TokenKind::Identifier),
1178            ],
1179            bytes: 20,
1180        };
1181        let cx_of = |format: &str| {
1182            compute_summary(
1183                &[file(format)],
1184                &[],
1185                10,
1186                SummaryMetric::Complexity,
1187                identity,
1188            )
1189            .files[0]
1190                .complexity
1191        };
1192        assert_eq!(cx_of("csharp"), 2, "the `if` after the escape is code");
1193        // The same run in Python really does open a string.
1194        assert_eq!(cx_of("python"), 1);
1195    }
1196
1197    fn lit_token(value: &str, start: u32, end: u32, kind: TokenKind) -> Token {
1198        Token {
1199            kind,
1200            value: value.to_string(),
1201            start: Location {
1202                line: 1,
1203                column: start,
1204                offset: start,
1205            },
1206            end: Location {
1207                line: 1,
1208                column: end,
1209                offset: end,
1210            },
1211        }
1212    }
1213
1214    #[test]
1215    fn a_name_the_file_binds_is_not_a_keyword() {
1216        // `case` and `when` are branch keywords somewhere and ordinary
1217        // variables elsewhere; the tokenizer tags both `Identifier`. A file
1218        // that assigns to the word has settled the question for itself.
1219        let code = "def run(c):\n case = 3\n when = 4\n return case + when\n";
1220        assert_eq!(cx("python", code), 1);
1221        // Reading it off an object is the same evidence.
1222        assert_eq!(cx("python", "def f(o):\n return o.case\n"), 1);
1223        // So is listing it as a parameter or an argument.
1224        let listed = "def headline(case, when):\n return fmt(case, when)\n";
1225        assert_eq!(cx("python", listed), 1);
1226        // Rust's `?` before a closing paren is still a branch: only words are
1227        // taken as names.
1228        let question = "fn f(s: &str) -> Result<u8, E> { Ok(parse(s)?) }";
1229        assert_eq!(cx("rust", question), 2);
1230        // Without that evidence the keyword still counts: one `def`, plus
1231        // `case` and `when`.
1232        let ruby = "def f(x)\n case x\n when 1 then 2\n end\nend";
1233        assert_eq!(cx("ruby", ruby), 3);
1234    }
1235
1236    #[test]
1237    fn the_c_family_declares_functions_without_a_keyword() {
1238        // `head(args) {` is the only marker C, C++, Java and C# give, and it
1239        // has to be told apart from the `) {` of a control statement.
1240        let code =
1241            "int add(int a, int b) { if (a > b) { return a; } return b; }\nvoid noop(void) { }\n";
1242        assert_eq!(cx("c", code), 3, "two functions plus one if");
1243        let java = "class T { int f(int a) { if (a > 0 && a < 10) { return a; } return 0; } }";
1244        assert_eq!(cx("java", java), 3, "one function, one if, one &&");
1245    }
1246
1247    #[test]
1248    fn a_parenthesised_statement_is_not_a_function() {
1249        // `switch (x) {` and `for (…) {` end in `) {` just like a signature.
1250        // One function plus the single `case`; the `switch` head adds neither.
1251        assert_eq!(
1252            cx(
1253                "c",
1254                "int f(int x) { switch (x) { case 1: return 1; } return 0; }"
1255            ),
1256            2
1257        );
1258        assert_eq!(
1259            cx("c", "void f(void) { for (int i = 0; i < 3; i++) { } }"),
1260            2
1261        );
1262        assert_eq!(cx("c", "void f(void) { while (x) { } }"), 2);
1263        // A language that declares functions by keyword is unaffected by this.
1264        assert_eq!(cx("go", "func f() { if x { } }"), 2);
1265    }
1266
1267    #[test]
1268    fn complexity_counts_decision_tokens() {
1269        let sources = vec![source(
1270            "a.js",
1271            "javascript",
1272            &["if", "x", "&&", "y", "for", "z", "else"],
1273            10,
1274        )];
1275        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1276        // 1 + (if, &&, for) = 4; "else" is not a decision point.
1277        assert_eq!(summary.files[0].complexity, 4);
1278    }
1279
1280    #[test]
1281    fn complexity_is_case_insensitive() {
1282        // SQL / PL/SQL / Fortran style uppercase keywords.
1283        let sources = vec![source(
1284            "a.sql",
1285            "sql",
1286            &["IF", "x", "OR", "y", "WHEN", "THEN", "If"],
1287            10,
1288        )];
1289        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1290        // 1 + (IF, OR, WHEN, If) = 5; THEN is not a decision point.
1291        assert_eq!(summary.files[0].complexity, 5);
1292    }
1293
1294    #[test]
1295    fn decision_token_edge_cases() {
1296        assert!(is_decision_token("unless"));
1297        assert!(is_decision_token("ELSEIF"));
1298        assert!(is_decision_token("andalso"));
1299        assert!(!is_decision_token(""));
1300        assert!(!is_decision_token("iffy"));
1301        assert!(!is_decision_token("conditionally"), "length-capped");
1302        assert!(!is_decision_token("форматирование"), "non-ASCII ignored");
1303    }
1304
1305    #[test]
1306    fn folder_rollup_uses_direct_parent() {
1307        let sources = vec![
1308            source("src/app/a.js", "javascript", &["x"], 5),
1309            source("src/app/b.js", "javascript", &["x", "y"], 5),
1310            source("src/c.js", "javascript", &["x"], 5),
1311            source("root.js", "javascript", &["x"], 5),
1312        ];
1313        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
1314        assert_eq!(summary.total_folders, 3);
1315        let app = summary
1316            .folders
1317            .iter()
1318            .find(|f| f.path == "src/app")
1319            .expect("src/app folder");
1320        assert_eq!(app.files, 2);
1321        assert_eq!(app.tokens, 3);
1322        let root = summary.folders.iter().find(|f| f.path == ".");
1323        assert!(root.is_some(), "root files grouped under '.'");
1324    }
1325
1326    #[test]
1327    fn duplication_attributed_to_both_fragments() {
1328        let sources = vec![
1329            source("a.js", "javascript", &["x", "y", "z"], 5),
1330            source("b.js", "javascript", &["x", "y", "z"], 5),
1331        ];
1332        let clones = vec![clone_between("javascript", "a.js", "b.js", 9, 30)];
1333        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, identity);
1334        for path in ["a.js", "b.js"] {
1335            let file = summary.files.iter().find(|f| f.path == path).unwrap();
1336            assert_eq!(file.duplicated_lines, 9, "{path} duplicated lines");
1337            assert_eq!(file.duplicated_tokens, 30, "{path} duplicated tokens");
1338        }
1339    }
1340
1341    #[test]
1342    fn synthetic_sub_format_sources_are_skipped() {
1343        let sources = vec![
1344            source("doc.md", "markdown", &["x", "y"], 100),
1345            source("doc.md:javascript", "javascript", &["x"], 0),
1346        ];
1347        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
1348        assert_eq!(summary.total_files, 1);
1349        assert_eq!(summary.files[0].path, "doc.md");
1350    }
1351
1352    #[test]
1353    fn sub_format_clone_folds_into_parent_file() {
1354        let sources = vec![source("doc.md", "markdown", &["x", "y"], 100)];
1355        let clones = vec![clone_between(
1356            "javascript",
1357            "doc.md:javascript",
1358            "doc.md:javascript",
1359            4,
1360            20,
1361        )];
1362        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, identity);
1363        assert_eq!(
1364            summary.files[0].duplicated_lines, 8,
1365            "both fragments fold in"
1366        );
1367    }
1368
1369    #[test]
1370    fn gap_lines_of_a_merged_clone_stay_out_of_file_duplication() {
1371        let sources = vec![
1372            source("a.js", "javascript", &["x"; 20], 10),
1373            source("b.js", "javascript", &["x"; 20], 10),
1374        ];
1375        let mut merged = clone_between("javascript", "a.js", "b.js", 10, 60);
1376        merged.unmatched_lines = [0, 3];
1377        let summary = compute_summary(&sources, &[merged], 10, SummaryMetric::Tokens, identity);
1378        let dup = |path: &str| {
1379            summary
1380                .files
1381                .iter()
1382                .find(|f| f.path == path)
1383                .unwrap()
1384                .duplicated_lines
1385        };
1386        assert_eq!(dup("a.js"), 10);
1387        assert_eq!(dup("b.js"), 7, "three gap lines in b are not duplicated");
1388    }
1389
1390    #[test]
1391    fn display_path_applied_before_dup_matching() {
1392        let sources = vec![source("/abs/root/a.js", "javascript", &["x"], 5)];
1393        let clones = vec![clone_between("javascript", "a.js", "a.js", 2, 10)];
1394        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, |p| {
1395            p.strip_prefix("/abs/root/").unwrap_or(p).to_string()
1396        });
1397        assert_eq!(summary.files[0].path, "a.js");
1398        assert_eq!(summary.files[0].duplicated_lines, 4);
1399    }
1400
1401    #[test]
1402    fn folders_truncated_to_top_n_but_total_reported() {
1403        let sources: Vec<SourceFile> = (0..5)
1404            .map(|i| source(&format!("dir{i}/f.js"), "javascript", &["x"], 1))
1405            .collect();
1406        let summary = compute_summary(&sources, &[], 2, SummaryMetric::Tokens, identity);
1407        assert_eq!(summary.folders.len(), 2);
1408        assert_eq!(summary.total_folders, 5);
1409    }
1410
1411    #[test]
1412    fn metric_parses_from_str() {
1413        assert_eq!(
1414            "complexity".parse::<SummaryMetric>().unwrap(),
1415            SummaryMetric::Complexity
1416        );
1417        assert!("bogus".parse::<SummaryMetric>().is_err());
1418    }
1419
1420    #[test]
1421    fn summary_serializes_camel_case() {
1422        let sources = vec![source("a.js", "javascript", &["x"], 5)];
1423        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Size, identity);
1424        let json = serde_json::to_string(&summary).unwrap();
1425        assert!(json.contains("\"totalFiles\""));
1426        assert!(json.contains("\"duplicatedLines\""));
1427        assert!(json.contains("\"by\":\"size\""));
1428        assert!(!json.contains("total_files"));
1429    }
1430}