Skip to main content

cpd_core/
summary.rs

1// summary.rs — opt-in codebase summary: per-file metrics, folder rollup, top-N lists.
2//
3// Everything in this module runs only when `--summary` is enabled, after
4// detection has finished, over data already held in memory (SourceFile tokens
5// and detected clones). Nothing in the detection hot path calls into it.
6
7use crate::models::{CpdClone, SourceFile, Token, TokenKind};
8use serde::{Deserialize, Serialize};
9use std::collections::HashMap;
10
11/// Metric used to rank files and folders in the summary.
12#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)]
13#[serde(rename_all = "lowercase")]
14pub enum SummaryMetric {
15    #[default]
16    Tokens,
17    Lines,
18    Size,
19    Complexity,
20}
21
22impl std::str::FromStr for SummaryMetric {
23    type Err = String;
24
25    fn from_str(s: &str) -> Result<Self, Self::Err> {
26        match s {
27            "tokens" => Ok(Self::Tokens),
28            "lines" => Ok(Self::Lines),
29            "size" => Ok(Self::Size),
30            "complexity" => Ok(Self::Complexity),
31            other => Err(format!(
32                "invalid summary metric '{other}': must be one of: tokens, lines, size, complexity"
33            )),
34        }
35    }
36}
37
38impl std::fmt::Display for SummaryMetric {
39    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
40        let s = match self {
41            Self::Tokens => "tokens",
42            Self::Lines => "lines",
43            Self::Size => "size",
44            Self::Complexity => "complexity",
45        };
46        f.write_str(s)
47    }
48}
49
50/// Per-file summary row.
51#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
52#[serde(rename_all = "camelCase")]
53pub struct FileSummary {
54    pub path: String,
55    pub format: String,
56    pub lines: u64,
57    pub tokens: u64,
58    pub bytes: u64,
59    pub duplicated_lines: u64,
60    pub duplicated_tokens: u64,
61    /// Cyclomatic-complexity estimate: 1 + count of decision-point tokens
62    /// (`if`, `for`, `while`, `case`, `catch`, `&&`, `||`, `?`, …).
63    pub complexity: u64,
64}
65
66/// Per-folder rollup. Files are counted in their direct parent directory only
67/// (no cumulative ancestor totals), so every file contributes to exactly one
68/// folder row and rows are directly comparable.
69#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
70#[serde(rename_all = "camelCase")]
71pub struct FolderSummary {
72    pub path: String,
73    pub files: u64,
74    pub lines: u64,
75    pub tokens: u64,
76    pub bytes: u64,
77    pub duplicated_lines: u64,
78    /// Sum of per-file complexity estimates (divide by `files` for the mean).
79    pub complexity: u64,
80}
81
82/// Codebase summary: top files and folder rollup.
83#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
84#[serde(rename_all = "camelCase")]
85pub struct Summary {
86    /// Primary sort metric.
87    pub by: SummaryMetric,
88    /// Top-N files by `by`, descending. Every row carries all metrics
89    /// (tokens, lines, bytes, complexity, duplication) so one list serves
90    /// every lens; re-run with a different `--summary-by` to re-rank.
91    pub files: Vec<FileSummary>,
92    /// Top-N folders by `by`, direct-parent aggregation.
93    pub folders: Vec<FolderSummary>,
94    /// Total number of files analyzed (before top-N truncation).
95    pub total_files: u64,
96    /// Total number of folders (before top-N truncation).
97    pub total_folders: u64,
98}
99
100/// Decision-point tokens counted by the complexity estimate. Conservative,
101/// language-agnostic list: branch/loop keywords and short-circuit operators
102/// that appear as standalone tokens across supported languages.
103///
104/// Matching is ASCII-case-insensitive so case-insensitive and
105/// uppercase-keyword languages (SQL, PL/SQL, Fortran, COBOL, BASIC, Pascal)
106/// count too. The occasional identifier spelled like a keyword slightly
107/// inflates an estimate that is only used for ranking.
108///
109/// Operators reach this function already joined — see [`joined_token`]. Only
110/// the JavaScript tokenizer emits `&&` as one token; the generic one splits
111/// every punctuation run into single characters, so without that joining the
112/// short-circuit arms here would be unreachable for every other format.
113fn is_decision_token(value: &str) -> bool {
114    let bytes = value.as_bytes();
115    if bytes.is_empty() || bytes.len() > 7 {
116        return false;
117    }
118    let mut lower = [0u8; 7];
119    for (dst, b) in lower.iter_mut().zip(bytes) {
120        *dst = b.to_ascii_lowercase();
121    }
122    matches!(
123        &lower[..bytes.len()],
124        b"if"
125            | b"elif"
126            | b"elsif"
127            | b"elseif"
128            | b"unless"
129            | b"for"
130            | b"foreach"
131            | b"while"
132            | b"until"
133            | b"case"
134            | b"cond"
135            | b"when"
136            | b"catch"
137            | b"rescue"
138            | b"except"
139            | b"andalso"
140            | b"orelse"
141            | b"&&"
142            | b"||"
143            | b"and"
144            | b"or"
145            | b"?"
146            | b"??"
147    )
148}
149
150/// What a bare `?` means in a language.
151#[derive(Clone, Copy, PartialEq, Eq)]
152enum QuestionMark {
153    /// It only ever opens a ternary: C, Java, PHP, Go templates, and most else.
154    Ternary,
155    /// It also marks an optional, so `String?` and `x?.y` are types and
156    /// accesses rather than branches. Neither TypeScript nor Swift has an
157    /// Elvis operator, so `?:` there is an optional property, not a branch.
158    Optional,
159    /// As above, but `?:` *is* the Elvis operator and does branch: Kotlin,
160    /// Groovy.
161    OptionalWithElvis,
162}
163
164/// How one language spells its branches, beyond the shared keyword set.
165///
166/// The shared set is right for most of the formats jscpd knows. An entry here
167/// exists only where a language branches on something the set has no word for
168/// (`match` arms in Rust, `guard` in Swift, `select` in Go) or spells
169/// something in the set so differently that counting it is simply wrong.
170#[derive(Clone, Copy)]
171struct DecisionRules {
172    /// Branch tokens beyond the shared set, matched after joining.
173    extra: &'static [&'static str],
174    question: QuestionMark,
175    /// Tokens that open a function body. Cyclomatic complexity is one path per
176    /// function; where a language has no single reliable marker this stays
177    /// empty and the estimate keeps its per-file baseline of one.
178    declarations: &'static [&'static str],
179    /// The C family declares a function as `head(args) {` with no keyword at
180    /// all, so its functions are counted from that shape instead.
181    braced_declarations: bool,
182    /// Whether `"""` and `'''` delimit a string that can span lines; see
183    /// [`has_triple_quoted_strings`].
184    triple_quoted_strings: bool,
185    /// Tokens that open a group of arms counted one by one through `extra`.
186    /// Each cancels one arm, because N arms are N paths and so N - 1 branches,
187    /// the same way a `switch` counts its `case` labels but not its `default`.
188    arm_groups: &'static [&'static str],
189}
190
191impl Default for DecisionRules {
192    fn default() -> Self {
193        Self {
194            extra: &[],
195            question: QuestionMark::Ternary,
196            declarations: &[],
197            braced_declarations: false,
198            triple_quoted_strings: false,
199            arm_groups: &[],
200        }
201    }
202}
203
204/// Formats whose strings can span lines between `"""` or `'''` delimiters.
205///
206/// Where a doubled quote is how a quote is escaped — C# verbatim strings,
207/// VB.NET, SQL, Pascal — three quotes in a row are ordinary string content: the
208/// regex `@"""((?:\\.|[^""\\])*)"""` is one line of C#. Reading such a run as
209/// a delimiter opens a string that never closes and swallows the rest of the
210/// file, which is why this is a list rather than a default.
211fn has_triple_quoted_strings(format: &str) -> bool {
212    matches!(
213        format,
214        "python" | "kotlin" | "scala" | "groovy" | "swift" | "java" | "julia" | "elixir" | "dart"
215    )
216}
217
218/// False for prose and data formats, whose "if" and `||` are words and
219/// version ranges rather than branches. One half of [`is_code`]: a format
220/// that can never have a branch can never have complexity above zero.
221pub fn has_control_flow(format: &str) -> bool {
222    !matches!(
223        format,
224        "markdown"
225            | "asciidoc"
226            | "rest"
227            | "textile"
228            | "wiki"
229            | "txt"
230            | "log"
231            | "csv"
232            | "json"
233            | "json5"
234            | "yaml"
235            | "toml"
236            | "ini"
237            | "properties"
238            | "editorconfig"
239            | "ignore"
240            | "diff"
241            | "gettext"
242    )
243}
244
245/// Markup, stylesheets, declarative schemas and the templating languages
246/// built on top of markup: a duplicated template or style rule repeating is
247/// not the maintenance problem duplicated programming logic is, so it does
248/// not count toward the health score's duplication share at all — the same
249/// treatment prose and data files get, just decided per clone rather than
250/// per file, since a `.svelte` or `.vue` file's markup and style blocks are
251/// tokenized separately from its script block. The other half of
252/// [`is_code`]: whole files in these formats have no complexity either — the
253/// "if" in an HTML attribute and the "and" in a media query are words, not
254/// branches.
255///
256/// These are the tokenizer's own format *names*
257/// (`cpd-tokenizer/src/formats.rs`), not file extensions: html/htm/xml/svg
258/// all tokenize as `markup` (`html` is the name a component file's markup
259/// block and a Markdown html snippet carry), `.puml`/`.plantuml` as
260/// `plant-uml`, `.tpl` as `smarty`, `.jade` as `pug`, and `.vtl` as
261/// `velocity` — matching on the extension instead of the name a clone's
262/// `format` field actually carries would silently never exclude anything.
263pub fn is_markup(format: &str) -> bool {
264    matches!(
265        format,
266        "markup"
267            | "html"
268            | "css"
269            | "scss"
270            | "sass"
271            | "less"
272            | "stylus"
273            | "razor"
274            | "haml"
275            | "pug"
276            | "handlebars"
277            | "erb"
278            | "liquid"
279            | "twig"
280            | "velocity"
281            | "ftl"
282            | "soy"
283            | "smarty"
284            | "tt2"
285            | "protobuf"
286            | "plant-uml"
287            | "mermaid"
288            | "django"
289            | "aspnet"
290    )
291}
292
293/// Whether a format counts as the project's code: prose and data
294/// ([`has_control_flow`]) and markup ([`is_markup`]) do not. This one test
295/// decides where complexity can be above zero, and through that which files
296/// are `health::compute`'s "code files"; a format-level duplication
297/// breakdown leaves the same formats out.
298pub fn is_code(format: &str) -> bool {
299    has_control_flow(format) && !is_markup(format)
300}
301
302fn rules_for(format: &str) -> DecisionRules {
303    DecisionRules {
304        triple_quoted_strings: has_triple_quoted_strings(format),
305        ..language_rules(format)
306    }
307}
308
309fn language_rules(format: &str) -> DecisionRules {
310    match format {
311        // A `match` has no keyword per arm, only the `=>` each one is written
312        // with. Every arm is counted and the `match` itself takes one back: a
313        // three-arm match is three paths, which is two branches.
314        "rust" => DecisionRules {
315            extra: &["=>"],
316            declarations: &["fn"],
317            arm_groups: &["match"],
318            ..Default::default()
319        },
320        "swift" => DecisionRules {
321            extra: &["guard"],
322            question: QuestionMark::Optional,
323            declarations: &["func"],
324            ..Default::default()
325        },
326        // `=>` opens an arrow function here rather than a branch, so it counts
327        // toward the function tally instead.
328        "typescript" | "tsx" | "flow" | "javascript" | "jsx" => DecisionRules {
329            question: QuestionMark::Optional,
330            declarations: &["function", "=>"],
331            ..Default::default()
332        },
333        "kotlin" => DecisionRules {
334            question: QuestionMark::OptionalWithElvis,
335            declarations: &["fun"],
336            ..Default::default()
337        },
338        "groovy" => DecisionRules {
339            question: QuestionMark::OptionalWithElvis,
340            declarations: &["def"],
341            ..Default::default()
342        },
343        "csharp" => DecisionRules {
344            question: QuestionMark::Optional,
345            braced_declarations: true,
346            ..Default::default()
347        },
348        "c" | "c-header" | "cpp" | "cpp-header" | "java" | "objectivec" | "clike" => {
349            DecisionRules {
350                braced_declarations: true,
351                ..Default::default()
352            }
353        }
354        "go" => DecisionRules {
355            extra: &["select"],
356            declarations: &["func"],
357            ..Default::default()
358        },
359        "scala" | "python" | "ruby" | "crystal" => DecisionRules {
360            declarations: &["def"],
361            ..Default::default()
362        },
363        "php" | "lua" => DecisionRules {
364            declarations: &["function"],
365            ..Default::default()
366        },
367        "erlang" => DecisionRules {
368            extra: &["receive"],
369            ..Default::default()
370        },
371        _ => DecisionRules::default(),
372    }
373}
374
375/// Two-character operators the generic tokenizer hands over as two tokens.
376const JOINED_OPERATORS: &[&str] = &["&&", "||", "??", "?.", "?:", "=>"];
377
378/// One token's text, or the two-character operator that two adjacent tokens
379/// spell between them.
380///
381/// The pair cannot borrow: each token owns its own `String`, so two adjacent
382/// characters of the source are not adjacent in memory. Two bytes on the stack
383/// avoid an allocation per operator.
384enum Scanned<'a> {
385    Single(&'a str),
386    Pair([u8; 2]),
387}
388
389impl Scanned<'_> {
390    fn text(&self) -> &str {
391        match self {
392            Self::Single(text) => text,
393            Self::Pair(bytes) => std::str::from_utf8(bytes).unwrap_or(""),
394        }
395    }
396}
397
398/// The token at `at`, joined with the next one when the two are adjacent
399/// single-character punctuation spelling a two-character operator.
400///
401/// The generic tokenizer splits `&&` into two `&` tokens, so a scan that looks
402/// at one token at a time can never see a short-circuit operator at all.
403/// Returns what to classify and the index to continue from.
404fn joined_token(tokens: &[Token], at: usize) -> (Scanned<'_>, usize) {
405    let current = &tokens[at];
406    let single = (Scanned::Single(current.value.as_str()), at + 1);
407    let Some(next) = tokens.get(at + 1) else {
408        return single;
409    };
410    let (Some(&left), Some(&right)) = (
411        current.value.as_bytes().first(),
412        next.value.as_bytes().first(),
413    ) else {
414        return single;
415    };
416    // Adjacent in the source, and both exactly one punctuation character.
417    if current.end.offset != next.start.offset
418        || current.value.len() != 1
419        || next.value.len() != 1
420        || !left.is_ascii_punctuation()
421        || !right.is_ascii_punctuation()
422    {
423        return single;
424    }
425    let pair = [left, right];
426    match std::str::from_utf8(&pair).is_ok_and(|text| JOINED_OPERATORS.contains(&text)) {
427        true => (Scanned::Pair(pair), at + 2),
428        false => single,
429    }
430}
431
432/// Which tokens lie inside a triple-quoted string.
433///
434/// The tokenizer is line-based: it closes every string at the end of the line
435/// it started on, so the body of a `"""` string arrives as ordinary words, and
436/// a docstring saying "if the value is big" lands three branches on the file.
437///
438/// The quotes themselves survive as literal tokens, and a run of adjacent
439/// literals spells exactly the quote characters its line holds: a PEP 257
440/// opener `"""Summary.` arrives as `""` and `"Summary.`, a closer on its own
441/// line as `""` and `"`, and a one-line docstring as `""`, `"One."`, `""`. So
442/// an odd number of triple quotes in a run opens or closes the string, and an
443/// even number leaves it as it was.
444///
445/// Reading a pair of tokens alone is not enough, and was the first version of
446/// this: with the summary written on the opening line the opener is not a bare
447/// quote, so the *closing* `"""` looked like an opener and swallowed every
448/// branch that followed it.
449fn triple_quoted(tokens: &[Token]) -> Vec<bool> {
450    const DELIMITERS: [&str; 2] = ["\"\"\"", "'''"];
451    let mut inside = vec![false; tokens.len()];
452    let mut open: Option<&str> = None;
453    let mut at = 0usize;
454    while at < tokens.len() {
455        if tokens[at].kind != TokenKind::Literal {
456            inside[at] = open.is_some();
457            at += 1;
458            continue;
459        }
460        let start = at;
461        let mut text = tokens[at].value.clone();
462        at += 1;
463        while at < tokens.len()
464            && tokens[at].kind == TokenKind::Literal
465            && tokens[at - 1].end.offset == tokens[at].start.offset
466        {
467            text.push_str(&tokens[at].value);
468            at += 1;
469        }
470        let was_open = open.is_some();
471        for delimiter in DELIMITERS {
472            let toggles = text.matches(delimiter).count() % 2 == 1;
473            match open {
474                Some(current) if current == delimiter && toggles => open = None,
475                None if toggles => open = Some(delimiter),
476                _ => {}
477            }
478        }
479        // The run is string text whichever way it turned the state.
480        inside[start..at].fill(was_open || open.is_some());
481    }
482    inside
483}
484
485/// Words a file binds as names of its own.
486///
487/// `case`, `when` and `cond` are branch keywords in some languages and
488/// perfectly ordinary variable names in others — the shared list cannot tell
489/// them apart, and the tokenizer is no help: its `classify_word` returns
490/// `Identifier` for every word, so `if` in Python is tagged exactly like a
491/// variable called `case`.
492///
493/// What a file *does* reveal is which words it assigns to, reads off an
494/// object, or lists as a parameter or argument. A word this file writes
495/// `case = 3`, `x.case` or `f(case, when)` for is that file's own name,
496/// whatever the language reserves, and counting it as a branch is what
497/// inflated the estimate elevenfold on a file of five such assignments.
498fn locally_bound(tokens: &[Token]) -> Vec<&str> {
499    let mut bound = Vec::new();
500    for (at, token) in tokens.iter().enumerate() {
501        if !is_decision_token(&token.value) {
502            continue;
503        }
504        // `obj.case` — a member, never the keyword.
505        let after_dot = at
506            .checked_sub(1)
507            .is_some_and(|previous| tokens[previous].value == ".");
508        // `case = 3`, but not `case == 3` or `case => 3`.
509        let assigned = tokens.get(at + 1).is_some_and(|next| next.value == "=")
510            && tokens
511                .get(at + 2)
512                .is_none_or(|after| !matches!(after.value.as_str(), "=" | ">"));
513        // `def headline(case, when)` or `headline(case, when)`: a keyword is
514        // never written directly before a comma or a closing paren, a name in
515        // a parameter or argument list always is. Words only — `Ok(parse()?)`
516        // is Rust's `?` doing its job, not a name.
517        let listed = token.value.starts_with(|c: char| c.is_ascii_alphabetic())
518            && tokens
519                .get(at + 1)
520                .is_some_and(|next| matches!(next.value.as_str(), "," | ")"));
521        if after_dot || assigned || listed {
522            bound.push(token.value.as_str());
523        }
524    }
525    bound.sort_unstable();
526    bound.dedup();
527    bound
528}
529
530/// Keywords that take a parenthesised head and a block, so that `) {` after
531/// one of them opens a branch rather than a function body.
532const PARENTHESISED_STATEMENTS: &[&str] = &[
533    "switch",
534    "using",
535    "lock",
536    "synchronized",
537    "with",
538    "do",
539    "try",
540    "return",
541    "sizeof",
542    "typeof",
543    "new",
544    "throw",
545    "await",
546    "yield",
547    "fixed",
548    "unsafe",
549];
550
551/// The complexity of a file's code: one path per function, or one for the
552/// whole file where the language has no function marker the scan can trust,
553/// plus one per branch. Prose, data and markup have no paths: an "if" in a
554/// README or an HTML attribute is a word, and a lock file full of `||`
555/// version ranges is not code.
556pub fn file_complexity(tokens: &[Token], format: &str) -> u64 {
557    if !is_code(format) {
558        return 0;
559    }
560    let (decisions, functions) = scan_complexity(tokens, &rules_for(format));
561    functions.max(1) + decisions
562}
563
564/// The complexity of the code from byte `start` to byte `end` of a file with
565/// these `tokens`, such as one function: one path plus one per branch in it.
566/// The branches of a function nested in the span count too. `0` for prose,
567/// data and markup.
568pub fn span_complexity(tokens: &[Token], format: &str, start: u32, end: u32) -> u64 {
569    if !is_code(format) {
570        return 0;
571    }
572    let first = tokens.partition_point(|t| t.start.offset < start);
573    let last = tokens.partition_point(|t| t.start.offset < end);
574    let (decisions, _) = scan_complexity(&tokens[first..last.max(first)], &rules_for(format));
575    1 + decisions
576}
577
578/// Decision points and function count for one file.
579///
580/// Cyclomatic complexity is one path per function plus one per branch. Where a
581/// language has no reliable function marker the tally falls back to the
582/// per-file baseline of one, which is what this estimate has always used.
583fn scan_complexity(tokens: &[Token], rules: &DecisionRules) -> (u64, u64) {
584    let (mut decisions, mut functions, mut groups) = (0u64, 0u64, 0u64);
585    let mut at = 0usize;
586    let in_string = match rules.triple_quoted_strings {
587        true => triple_quoted(tokens),
588        false => vec![false; tokens.len()],
589    };
590    let bound = locally_bound(tokens);
591    // What the token before each open paren was, so a `) {` can be told from
592    // the head it closes: a function signature, or `if (…) {`.
593    let mut heads: Vec<&str> = Vec::new();
594    let mut last_head: Option<&str> = None;
595    while at < tokens.len() {
596        // Skip the body of a string that spans lines; see `triple_quoted`.
597        if in_string[at] {
598            at += 1;
599            continue;
600        }
601        let (scanned, next) = joined_token(tokens, at);
602        let text = scanned.text();
603        // `String?` writes the optional against the type it belongs to;
604        // `cond ? a : b` puts the ternary in the open. Nothing else separates
605        // the two without parsing, and the convention is near-universal.
606        let attached = at
607            .checked_sub(1)
608            .and_then(|previous| tokens.get(previous))
609            .is_some_and(|previous| previous.end.offset == tokens[at].start.offset);
610        at = next;
611        // C, C++, Java and C# open a function with `) {` and no keyword of
612        // their own. Tracking what preceded each `(` is enough to tell that
613        // from the `) {` of an `if` or a `switch`, and it is the only reason
614        // those languages kept a per-file baseline.
615        if rules.braced_declarations {
616            match text {
617                "(" => {
618                    let head = at
619                        .checked_sub(2)
620                        .and_then(|before| tokens.get(before))
621                        .map(|token| token.value.as_str())
622                        .unwrap_or("");
623                    heads.push(head);
624                }
625                ")" => last_head = heads.pop(),
626                "{" => {
627                    if let Some(head) = last_head.take()
628                        && !head.is_empty()
629                        && head.chars().all(|c| c.is_alphanumeric() || c == '_')
630                        && !is_decision_token(head)
631                        && !PARENTHESISED_STATEMENTS
632                            .iter()
633                            .any(|s| s.eq_ignore_ascii_case(head))
634                    {
635                        functions += 1;
636                    }
637                }
638                _ => last_head = None,
639            }
640        }
641        if rules
642            .declarations
643            .iter()
644            .any(|d| d.eq_ignore_ascii_case(text))
645        {
646            functions += 1;
647            continue;
648        }
649        if rules
650            .arm_groups
651            .iter()
652            .any(|g| g.eq_ignore_ascii_case(text))
653        {
654            groups += 1;
655            continue;
656        }
657        if rules.extra.iter().any(|e| e.eq_ignore_ascii_case(text)) {
658            decisions += 1;
659            continue;
660        }
661        // A word this file binds as a name is that file's name, not a keyword.
662        if bound.binary_search(&text).is_ok() {
663            continue;
664        }
665        let counts = match text {
666            // `?.` never branches, and `?:` branches only where it is the
667            // Elvis operator rather than an optional property.
668            "?." => false,
669            "?:" => rules.question != QuestionMark::Optional,
670            "?" => rules.question == QuestionMark::Ternary || !attached,
671            other => is_decision_token(other),
672        };
673        if counts {
674            decisions += 1;
675        }
676    }
677    (decisions.saturating_sub(groups), functions)
678}
679
680/// A synthetic source is the per-sub-format shadow of a multi-format file
681/// (markdown/vue/svelte embedded code); its id is `<parent-id>:<format>` and
682/// its metrics are already covered by the parent entry.
683fn is_synthetic(source: &SourceFile) -> bool {
684    source
685        .id
686        .strip_suffix(source.format.as_str())
687        .is_some_and(|prefix| prefix.ends_with(':'))
688}
689
690fn metric_of(file: &FileSummary, by: SummaryMetric) -> u64 {
691    match by {
692        SummaryMetric::Tokens => file.tokens,
693        SummaryMetric::Lines => file.lines,
694        SummaryMetric::Size => file.bytes,
695        SummaryMetric::Complexity => file.complexity,
696    }
697}
698
699fn folder_metric_of(folder: &FolderSummary, by: SummaryMetric) -> u64 {
700    match by {
701        SummaryMetric::Tokens => folder.tokens,
702        SummaryMetric::Lines => folder.lines,
703        SummaryMetric::Size => folder.bytes,
704        SummaryMetric::Complexity => folder.complexity,
705    }
706}
707
708/// Parent directory of a path, with separators normalized to `/`.
709/// Files at the scan root map to `"."`.
710fn parent_dir(path: &str) -> String {
711    let normalized = path.replace('\\', "/");
712    match normalized.rfind('/') {
713        Some(0) => "/".to_string(),
714        Some(idx) => normalized[..idx].to_string(),
715        None => ".".to_string(),
716    }
717}
718
719/// Compute the summary from detection results.
720///
721/// `display_path` maps a source id (canonical absolute path) to the path shown
722/// in reports — the same relativization applied to clone fragments, so
723/// per-file duplication matching works on identical strings.
724pub fn compute_summary(
725    sources: &[SourceFile],
726    clones: &[CpdClone],
727    top: usize,
728    by: SummaryMetric,
729    display_path: impl Fn(&str) -> String,
730) -> Summary {
731    // Per-file duplication, keyed by display path. Both fragments of a clone
732    // count toward their file: the question here is "where does duplicated
733    // code live", not the de-duplicated total that Statistics reports.
734    let mut dup: HashMap<String, (u64, u64)> = HashMap::new();
735    for clone in clones {
736        for (index, fragment) in [&clone.fragment_a, &clone.fragment_b]
737            .into_iter()
738            .enumerate()
739        {
740            // Sub-format fragments carry a `<path>:<format>` id; fold them
741            // into the parent file.
742            let path = fragment
743                .source_id
744                .strip_suffix(&format!(":{}", clone.format))
745                .unwrap_or(&fragment.source_id);
746            let entry = dup.entry(path.to_string()).or_default();
747            entry.0 += clone.fragment_lines(index);
748            entry.1 += clone.token_count as u64;
749        }
750    }
751
752    let mut files: Vec<FileSummary> = sources
753        .iter()
754        .filter(|s| !is_synthetic(s))
755        .map(|source| {
756            let path = display_path(&source.id);
757            // Same line metric as Statistics: max token start line.
758            let lines = source
759                .tokens
760                .iter()
761                .map(|t| t.start.line)
762                .max()
763                .unwrap_or(0) as u64;
764            let (duplicated_lines, duplicated_tokens) = dup.get(&path).copied().unwrap_or_default();
765            FileSummary {
766                lines,
767                tokens: source.tokens.len() as u64,
768                bytes: source.bytes,
769                duplicated_lines,
770                duplicated_tokens,
771                complexity: file_complexity(&source.tokens, &source.format),
772                format: source.format.clone(),
773                path,
774            }
775        })
776        .collect();
777
778    let total_files = files.len() as u64;
779
780    // Folder rollup over ALL files (before top-N truncation).
781    let mut folder_map: HashMap<String, FolderSummary> = HashMap::new();
782    for file in &files {
783        let dir = parent_dir(&file.path);
784        let entry = folder_map
785            .entry(dir.clone())
786            .or_insert_with(|| FolderSummary {
787                path: dir,
788                files: 0,
789                lines: 0,
790                tokens: 0,
791                bytes: 0,
792                duplicated_lines: 0,
793                complexity: 0,
794            });
795        entry.files += 1;
796        entry.lines += file.lines;
797        entry.tokens += file.tokens;
798        entry.bytes += file.bytes;
799        entry.duplicated_lines += file.duplicated_lines;
800        entry.complexity += file.complexity;
801    }
802    let total_folders = folder_map.len() as u64;
803
804    // Top-N files by the primary metric: `--summary-top N` always yields at
805    // most N rows (least surprise). Other lenses are one `--summary-by` away;
806    // every row still carries all metrics.
807    files.sort_by(|a, b| {
808        metric_of(b, by)
809            .cmp(&metric_of(a, by))
810            .then_with(|| a.path.cmp(&b.path))
811    });
812    files.truncate(top);
813
814    let mut folders: Vec<FolderSummary> = folder_map.into_values().collect();
815    folders.sort_by(|a, b| {
816        folder_metric_of(b, by)
817            .cmp(&folder_metric_of(a, by))
818            .then_with(|| a.path.cmp(&b.path))
819    });
820    folders.truncate(top);
821
822    Summary {
823        by,
824        files,
825        folders,
826        total_files,
827        total_folders,
828    }
829}
830
831#[cfg(test)]
832mod tests {
833    use super::*;
834    use crate::models::{CpdClone, Fragment, Location, Token, TokenKind};
835
836    fn loc(line: u32) -> Location {
837        Location {
838            line,
839            column: 0,
840            offset: 0,
841        }
842    }
843
844    fn token(value: &str, line: u32) -> Token {
845        Token {
846            kind: TokenKind::Keyword,
847            value: value.to_string(),
848            start: loc(line),
849            end: loc(line),
850        }
851    }
852
853    fn source(id: &str, format: &str, values: &[&str], bytes: u64) -> SourceFile {
854        SourceFile {
855            id: id.to_string(),
856            format: format.to_string(),
857            tokens: values
858                .iter()
859                .enumerate()
860                .map(|(i, v)| token(v, i as u32 + 1))
861                .collect(),
862            bytes,
863        }
864    }
865
866    /// A clone whose fragments both cover `lines` lines, counted the way the
867    /// statistics count them: line 1 through line `lines`, both ends included.
868    fn clone_between(format: &str, a: &str, b: &str, lines: u32, tokens: u32) -> CpdClone {
869        let fragment = |id: &str| Fragment {
870            source_id: id.to_string(),
871            source_root: None,
872            start: loc(1),
873            end: loc(lines),
874            range: [0, tokens],
875            blame: None,
876        };
877        CpdClone {
878            format: format.to_string(),
879            fragment_a: fragment(a),
880            fragment_b: fragment(b),
881            token_count: tokens,
882            is_new: false,
883            kind: Default::default(),
884            similarity: None,
885            similarity_method: None,
886            unmatched_lines: [0, 0],
887        }
888    }
889
890    fn identity(path: &str) -> String {
891        path.to_string()
892    }
893
894    #[test]
895    fn empty_input_produces_empty_summary() {
896        let summary = compute_summary(&[], &[], 10, SummaryMetric::Tokens, identity);
897        assert!(summary.files.is_empty());
898        assert!(summary.folders.is_empty());
899        assert_eq!(summary.total_files, 0);
900        assert_eq!(summary.total_folders, 0);
901    }
902
903    #[test]
904    fn prose_and_data_have_no_complexity() {
905        let words = [
906            "If", "you", "need", "it", "or", "while", "waiting", "for", "a", "case",
907        ];
908        let sources = vec![
909            source("README.md", "markdown", &words, 10),
910            source(
911                "pnpm-lock.yaml",
912                "yaml",
913                &["version", ":", "^1", "||", "^2"],
914                10,
915            ),
916            source("guide.rst", "rest", &words, 10),
917            source("notes.py", "python", &words, 10),
918        ];
919        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
920        let cx = |path: &str| {
921            summary
922                .files
923                .iter()
924                .find(|f| f.path == path)
925                .unwrap()
926                .complexity
927        };
928        assert_eq!(cx("README.md"), 0, "a word is not a branch");
929        assert_eq!(cx("pnpm-lock.yaml"), 0, "a version range is not a branch");
930        assert_eq!(cx("guide.rst"), 0, "reStructuredText is prose too");
931        assert!(cx("notes.py") > 1, "the same words in code still count");
932    }
933
934    #[test]
935    fn markup_styles_and_templates_have_no_complexity() {
936        // The same exclusion the duplication share applies (`is_markup`):
937        // an "if" in markup text or a media query's "and" is a word, not a
938        // branch. `markup` is what html/xml files tokenize as; `html` is the
939        // name a component file's markup block carries.
940        let words = [
941            "<", "a", ">", "If", "you", "click", "or", "wait", "<", "/", "a", ">",
942        ];
943        let sources = vec![
944            source("index.html", "markup", &words, 10),
945            source("snippet.html", "html", &words, 10),
946            source(
947                "site.css",
948                "css",
949                &["@", "media", "screen", "and", "(", "print", ")"],
950                10,
951            ),
952            source("card.twig", "twig", &["{", "%", "if", "user", "%", "}"], 10),
953        ];
954        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
955        assert!(
956            summary.files.iter().all(|f| f.complexity == 0),
957            "markup formats must have complexity 0: {:?}",
958            summary.files
959        );
960    }
961
962    #[test]
963    fn files_sorted_by_primary_metric() {
964        let sources = vec![
965            source("src/small.js", "javascript", &["a", "b"], 10),
966            source("src/big.js", "javascript", &["a", "b", "c", "d"], 20),
967        ];
968        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
969        assert_eq!(summary.files[0].path, "src/big.js");
970        assert_eq!(summary.files[0].tokens, 4);
971        assert_eq!(summary.total_files, 2);
972    }
973
974    #[test]
975    fn top_n_is_exact_row_count_by_primary_metric() {
976        // huge.js wins on tokens, fat.js wins on size — top=1 by tokens must
977        // yield exactly one row: huge.js. `--summary-top N` never surprises
978        // with more than N rows; other metrics are served by --summary-by.
979        let sources = vec![
980            source("huge.js", "javascript", &["a", "b", "c", "d", "e"], 1),
981            source("fat.js", "javascript", &["a"], 9999),
982        ];
983        let summary = compute_summary(&sources, &[], 1, SummaryMetric::Tokens, identity);
984        assert_eq!(summary.files.len(), 1);
985        assert_eq!(summary.files[0].path, "huge.js");
986        assert_eq!(summary.total_files, 2, "truncation stays visible");
987
988        let by_size = compute_summary(&sources, &[], 1, SummaryMetric::Size, identity);
989        assert_eq!(by_size.files[0].path, "fat.js");
990    }
991
992    /// Tokens laid out over a real source string the way the generic
993    /// tokenizer hands them over: one per word, one per punctuation
994    /// character, with the offsets that make adjacency visible.
995    fn lay_out(id: &str, format: &str, code: &str) -> SourceFile {
996        let mut tokens = Vec::new();
997        let bytes = code.as_bytes();
998        let mut at = 0usize;
999        while at < bytes.len() {
1000            let byte = bytes[at];
1001            if byte.is_ascii_whitespace() {
1002                at += 1;
1003                continue;
1004            }
1005            let start = at;
1006            if byte.is_ascii_alphanumeric() || byte == b'_' {
1007                while at < bytes.len() && (bytes[at].is_ascii_alphanumeric() || bytes[at] == b'_') {
1008                    at += 1;
1009                }
1010            } else {
1011                at += 1;
1012            }
1013            let at32 = |offset: usize| Location {
1014                line: 1,
1015                column: offset as u32,
1016                offset: offset as u32,
1017            };
1018            tokens.push(Token {
1019                kind: TokenKind::Identifier,
1020                value: code[start..at].to_string(),
1021                start: at32(start),
1022                end: at32(at),
1023            });
1024        }
1025        SourceFile {
1026            id: id.to_string(),
1027            format: format.to_string(),
1028            tokens,
1029            bytes: code.len() as u64,
1030        }
1031    }
1032
1033    fn cx(format: &str, code: &str) -> u64 {
1034        let sources = vec![lay_out("a", format, code)];
1035        compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity).files[0].complexity
1036    }
1037
1038    #[test]
1039    fn short_circuit_operators_count_when_split_into_characters() {
1040        // The generic tokenizer hands `&&` over as two `&` tokens, so without
1041        // joining them the short-circuit arms of the shared list never match
1042        // for any format but JavaScript.
1043        assert_eq!(cx("c", "int f() { return a && b; }"), 2);
1044        assert_eq!(cx("c", "int f() { return a || b; }"), 2);
1045        assert_eq!(cx("c", "int f() { return a && b || c; }"), 3);
1046        // A single `&` is a bitwise and, not a branch.
1047        assert_eq!(cx("c", "int f() { return a & b; }"), 1);
1048    }
1049
1050    #[test]
1051    fn an_optional_is_not_a_ternary() {
1052        // `String?` writes the question mark against its type; a ternary puts
1053        // it in the open. Nothing else tells them apart without parsing.
1054        assert_eq!(cx("swift", "func f(x: String?) -> Int { return 1 }"), 1);
1055        assert_eq!(cx("swift", "func f(x: Foo) { let y = x?.bar }"), 1);
1056        assert_eq!(
1057            cx("swift", "func f(x: Int) -> Int { return x > 0 ? 1 : 2 }"),
1058            2
1059        );
1060        // Nil-coalescing is a branch in any spelling.
1061        assert_eq!(
1062            cx("swift", "func f(a: Int?, b: Int) -> Int { return a ?? b }"),
1063            2
1064        );
1065        // A language where `?` only ever opens a ternary is unaffected.
1066        assert_eq!(cx("c", "int f(int x) { return x ? 1 : 2; }"), 2);
1067    }
1068
1069    #[test]
1070    fn a_match_arm_is_the_branch_not_the_match() {
1071        // Three arms are three paths: one for the function, two branches.
1072        let three_arms = "fn f(x: i32) -> i32 { match x { 0 => 1, 1 => 2, _ => 3 } }";
1073        assert_eq!(cx("rust", three_arms), 3);
1074        // A guard is a branch of its own on top of the arm it guards.
1075        let guarded = "fn f(x: i32, ok: bool) -> i32 { match x { 0 if ok => 1, 0 => 2, _ => 3 } }";
1076        assert_eq!(cx("rust", guarded), 4);
1077        // A match with a single arm does not branch at all.
1078        assert_eq!(cx("rust", "fn f(x: i32) -> i32 { match x { _ => 0 } }"), 1);
1079        // `=>` opens an arrow function in JavaScript, so it is a declaration
1080        // there rather than a branch.
1081        assert_eq!(cx("javascript", "const f = (x) => x + 1;"), 1);
1082    }
1083
1084    #[test]
1085    fn a_span_counts_one_path_and_the_branches_in_it() {
1086        let code = "def a(n):\n if n: pass\ndef b(n):\n if n and n: pass\n";
1087        let file = lay_out("a", "python", code);
1088        let second = code.find("def b").unwrap() as u32;
1089        assert_eq!(span_complexity(&file.tokens, "python", 0, second), 2);
1090        assert_eq!(
1091            span_complexity(&file.tokens, "python", second, code.len() as u32),
1092            3
1093        );
1094        // Two functions and three branches.
1095        assert_eq!(file_complexity(&file.tokens, "python"), 5);
1096        assert_eq!(span_complexity(&file.tokens, "markdown", 0, 10), 0);
1097    }
1098
1099    #[test]
1100    fn complexity_counts_one_path_per_function() {
1101        // Four functions, three of them with a single branch.
1102        let code = "def a(n):\n if n: pass\ndef b(n):\n if n: pass\ndef c(n):\n if n: pass\ndef d(n):\n pass\n";
1103        assert_eq!(cx("python", code), 7);
1104        // A language with no marker the scan can trust keeps the per-file
1105        // baseline of one rather than guessing.
1106        assert_eq!(cx("cobol", "IF x THEN y"), 2);
1107    }
1108
1109    #[test]
1110    fn guard_and_select_are_branches_where_they_exist() {
1111        assert_eq!(
1112            cx("swift", "func f(x: Int) { guard x > 0 else { return } }"),
1113            2
1114        );
1115        assert_eq!(cx("go", "func f() { select { } }"), 2);
1116        // `guard` is an ordinary word elsewhere.
1117        assert_eq!(cx("c", "int f() { int guard = 1; return guard; }"), 1);
1118    }
1119
1120    #[test]
1121    fn elvis_branches_only_where_the_language_has_one() {
1122        // Kotlin's `?:` is the elvis operator.
1123        assert_eq!(
1124            cx("kotlin", "fun f(a: Int?, b: Int): Int { return a ?: b }"),
1125            2
1126        );
1127        // TypeScript has no elvis; `a?: T` marks an optional property.
1128        assert_eq!(cx("typescript", "function f(a?: string) { return a; }"), 1);
1129    }
1130
1131    #[test]
1132    fn a_docstring_is_not_a_pile_of_branches() {
1133        // The real tokenizer closes every string at the end of its line, so a
1134        // docstring leaves a literal holding just its opening quote and its
1135        // body arrives as ordinary words.
1136        let quote = "\u{22}";
1137        let tokens = vec![
1138            lit_token("def", 0, 3, TokenKind::Identifier),
1139            lit_token("f", 4, 5, TokenKind::Identifier),
1140            lit_token(&quote.repeat(2), 6, 8, TokenKind::Literal),
1141            lit_token(quote, 8, 9, TokenKind::Literal),
1142            lit_token("if", 10, 12, TokenKind::Identifier),
1143            lit_token("for", 13, 16, TokenKind::Identifier),
1144            lit_token("while", 17, 22, TokenKind::Identifier),
1145            lit_token(&quote.repeat(2), 23, 25, TokenKind::Literal),
1146            lit_token(quote, 25, 26, TokenKind::Literal),
1147            lit_token("if", 27, 29, TokenKind::Identifier),
1148        ];
1149        let sources = vec![SourceFile {
1150            id: "a".into(),
1151            format: "python".into(),
1152            tokens,
1153            bytes: 30,
1154        }];
1155        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1156        // One `def`, and only the `if` outside the docstring.
1157        assert_eq!(summary.files[0].complexity, 2);
1158    }
1159
1160    #[test]
1161    fn a_docstring_with_its_summary_on_the_opening_line_closes_where_it_ends() {
1162        // PEP 257 style, as the tokenizer really hands it over: the opener is
1163        // `""` + `"Summary.` rather than a bare quote, so a reader that looked
1164        // for `""` + `"` alone took the *closing* quotes for an opener and
1165        // swallowed every branch after them.
1166        let q = "\u{22}";
1167        let tokens = vec![
1168            lit_token("def", 0, 3, TokenKind::Identifier),
1169            lit_token("f", 4, 5, TokenKind::Identifier),
1170            lit_token(&q.repeat(2), 10, 12, TokenKind::Literal),
1171            lit_token(&format!("{q}Summary."), 12, 21, TokenKind::Literal),
1172            lit_token("if", 30, 32, TokenKind::Identifier),
1173            lit_token("and", 33, 36, TokenKind::Identifier),
1174            lit_token(&q.repeat(2), 40, 42, TokenKind::Literal),
1175            lit_token(q, 42, 43, TokenKind::Literal),
1176            lit_token("if", 50, 52, TokenKind::Identifier),
1177            lit_token("x", 53, 54, TokenKind::Identifier),
1178            // A one-line docstring is balanced on its own line.
1179            lit_token(&q.repeat(2), 60, 62, TokenKind::Literal),
1180            lit_token(&format!("{q}One.{q}"), 62, 68, TokenKind::Literal),
1181            lit_token(&q.repeat(2), 68, 70, TokenKind::Literal),
1182            lit_token("for", 75, 78, TokenKind::Identifier),
1183        ];
1184        let sources = vec![SourceFile {
1185            id: "a".into(),
1186            format: "python".into(),
1187            tokens,
1188            bytes: 80,
1189        }];
1190        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1191        // One `def`, plus the `if` and the `for` written as code.
1192        assert_eq!(summary.files[0].complexity, 3);
1193    }
1194
1195    #[test]
1196    fn a_doubled_quote_escape_is_not_a_triple_quoted_string() {
1197        // C# verbatim strings escape a quote by doubling it, so a run spelling
1198        // `"""x` is a quote followed by `x`, not a string that runs on. Read
1199        // as one, it swallowed the rest of a 413-branch file.
1200        let q = "\u{22}";
1201        let file = |format: &str| SourceFile {
1202            id: "a".into(),
1203            format: format.into(),
1204            tokens: vec![
1205                lit_token("x", 0, 1, TokenKind::Identifier),
1206                lit_token(&q.repeat(2), 4, 6, TokenKind::Literal),
1207                lit_token(&format!("{q}x"), 6, 8, TokenKind::Literal),
1208                lit_token("if", 10, 12, TokenKind::Identifier),
1209                lit_token("y", 13, 14, TokenKind::Identifier),
1210            ],
1211            bytes: 20,
1212        };
1213        let cx_of = |format: &str| {
1214            compute_summary(
1215                &[file(format)],
1216                &[],
1217                10,
1218                SummaryMetric::Complexity,
1219                identity,
1220            )
1221            .files[0]
1222                .complexity
1223        };
1224        assert_eq!(cx_of("csharp"), 2, "the `if` after the escape is code");
1225        // The same run in Python really does open a string.
1226        assert_eq!(cx_of("python"), 1);
1227    }
1228
1229    fn lit_token(value: &str, start: u32, end: u32, kind: TokenKind) -> Token {
1230        Token {
1231            kind,
1232            value: value.to_string(),
1233            start: Location {
1234                line: 1,
1235                column: start,
1236                offset: start,
1237            },
1238            end: Location {
1239                line: 1,
1240                column: end,
1241                offset: end,
1242            },
1243        }
1244    }
1245
1246    #[test]
1247    fn a_name_the_file_binds_is_not_a_keyword() {
1248        // `case` and `when` are branch keywords somewhere and ordinary
1249        // variables elsewhere; the tokenizer tags both `Identifier`. A file
1250        // that assigns to the word has settled the question for itself.
1251        let code = "def run(c):\n case = 3\n when = 4\n return case + when\n";
1252        assert_eq!(cx("python", code), 1);
1253        // Reading it off an object is the same evidence.
1254        assert_eq!(cx("python", "def f(o):\n return o.case\n"), 1);
1255        // So is listing it as a parameter or an argument.
1256        let listed = "def headline(case, when):\n return fmt(case, when)\n";
1257        assert_eq!(cx("python", listed), 1);
1258        // Rust's `?` before a closing paren is still a branch: only words are
1259        // taken as names.
1260        let question = "fn f(s: &str) -> Result<u8, E> { Ok(parse(s)?) }";
1261        assert_eq!(cx("rust", question), 2);
1262        // Without that evidence the keyword still counts: one `def`, plus
1263        // `case` and `when`.
1264        let ruby = "def f(x)\n case x\n when 1 then 2\n end\nend";
1265        assert_eq!(cx("ruby", ruby), 3);
1266    }
1267
1268    #[test]
1269    fn the_c_family_declares_functions_without_a_keyword() {
1270        // `head(args) {` is the only marker C, C++, Java and C# give, and it
1271        // has to be told apart from the `) {` of a control statement.
1272        let code =
1273            "int add(int a, int b) { if (a > b) { return a; } return b; }\nvoid noop(void) { }\n";
1274        assert_eq!(cx("c", code), 3, "two functions plus one if");
1275        let java = "class T { int f(int a) { if (a > 0 && a < 10) { return a; } return 0; } }";
1276        assert_eq!(cx("java", java), 3, "one function, one if, one &&");
1277    }
1278
1279    #[test]
1280    fn a_parenthesised_statement_is_not_a_function() {
1281        // `switch (x) {` and `for (…) {` end in `) {` just like a signature.
1282        // One function plus the single `case`; the `switch` head adds neither.
1283        assert_eq!(
1284            cx(
1285                "c",
1286                "int f(int x) { switch (x) { case 1: return 1; } return 0; }"
1287            ),
1288            2
1289        );
1290        assert_eq!(
1291            cx("c", "void f(void) { for (int i = 0; i < 3; i++) { } }"),
1292            2
1293        );
1294        assert_eq!(cx("c", "void f(void) { while (x) { } }"), 2);
1295        // A language that declares functions by keyword is unaffected by this.
1296        assert_eq!(cx("go", "func f() { if x { } }"), 2);
1297    }
1298
1299    #[test]
1300    fn complexity_counts_decision_tokens() {
1301        let sources = vec![source(
1302            "a.js",
1303            "javascript",
1304            &["if", "x", "&&", "y", "for", "z", "else"],
1305            10,
1306        )];
1307        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1308        // 1 + (if, &&, for) = 4; "else" is not a decision point.
1309        assert_eq!(summary.files[0].complexity, 4);
1310    }
1311
1312    #[test]
1313    fn complexity_is_case_insensitive() {
1314        // SQL / PL/SQL / Fortran style uppercase keywords.
1315        let sources = vec![source(
1316            "a.sql",
1317            "sql",
1318            &["IF", "x", "OR", "y", "WHEN", "THEN", "If"],
1319            10,
1320        )];
1321        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Complexity, identity);
1322        // 1 + (IF, OR, WHEN, If) = 5; THEN is not a decision point.
1323        assert_eq!(summary.files[0].complexity, 5);
1324    }
1325
1326    #[test]
1327    fn decision_token_edge_cases() {
1328        assert!(is_decision_token("unless"));
1329        assert!(is_decision_token("ELSEIF"));
1330        assert!(is_decision_token("andalso"));
1331        assert!(!is_decision_token(""));
1332        assert!(!is_decision_token("iffy"));
1333        assert!(!is_decision_token("conditionally"), "length-capped");
1334        assert!(!is_decision_token("форматирование"), "non-ASCII ignored");
1335    }
1336
1337    #[test]
1338    fn folder_rollup_uses_direct_parent() {
1339        let sources = vec![
1340            source("src/app/a.js", "javascript", &["x"], 5),
1341            source("src/app/b.js", "javascript", &["x", "y"], 5),
1342            source("src/c.js", "javascript", &["x"], 5),
1343            source("root.js", "javascript", &["x"], 5),
1344        ];
1345        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
1346        assert_eq!(summary.total_folders, 3);
1347        let app = summary
1348            .folders
1349            .iter()
1350            .find(|f| f.path == "src/app")
1351            .expect("src/app folder");
1352        assert_eq!(app.files, 2);
1353        assert_eq!(app.tokens, 3);
1354        let root = summary.folders.iter().find(|f| f.path == ".");
1355        assert!(root.is_some(), "root files grouped under '.'");
1356    }
1357
1358    #[test]
1359    fn duplication_attributed_to_both_fragments() {
1360        let sources = vec![
1361            source("a.js", "javascript", &["x", "y", "z"], 5),
1362            source("b.js", "javascript", &["x", "y", "z"], 5),
1363        ];
1364        let clones = vec![clone_between("javascript", "a.js", "b.js", 9, 30)];
1365        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, identity);
1366        for path in ["a.js", "b.js"] {
1367            let file = summary.files.iter().find(|f| f.path == path).unwrap();
1368            assert_eq!(file.duplicated_lines, 9, "{path} duplicated lines");
1369            assert_eq!(file.duplicated_tokens, 30, "{path} duplicated tokens");
1370        }
1371    }
1372
1373    #[test]
1374    fn synthetic_sub_format_sources_are_skipped() {
1375        let sources = vec![
1376            source("doc.md", "markdown", &["x", "y"], 100),
1377            source("doc.md:javascript", "javascript", &["x"], 0),
1378        ];
1379        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Tokens, identity);
1380        assert_eq!(summary.total_files, 1);
1381        assert_eq!(summary.files[0].path, "doc.md");
1382    }
1383
1384    #[test]
1385    fn sub_format_clone_folds_into_parent_file() {
1386        let sources = vec![source("doc.md", "markdown", &["x", "y"], 100)];
1387        let clones = vec![clone_between(
1388            "javascript",
1389            "doc.md:javascript",
1390            "doc.md:javascript",
1391            4,
1392            20,
1393        )];
1394        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, identity);
1395        assert_eq!(
1396            summary.files[0].duplicated_lines, 8,
1397            "both fragments fold in"
1398        );
1399    }
1400
1401    #[test]
1402    fn gap_lines_of_a_merged_clone_stay_out_of_file_duplication() {
1403        let sources = vec![
1404            source("a.js", "javascript", &["x"; 20], 10),
1405            source("b.js", "javascript", &["x"; 20], 10),
1406        ];
1407        let mut merged = clone_between("javascript", "a.js", "b.js", 10, 60);
1408        merged.unmatched_lines = [0, 3];
1409        let summary = compute_summary(&sources, &[merged], 10, SummaryMetric::Tokens, identity);
1410        let dup = |path: &str| {
1411            summary
1412                .files
1413                .iter()
1414                .find(|f| f.path == path)
1415                .unwrap()
1416                .duplicated_lines
1417        };
1418        assert_eq!(dup("a.js"), 10);
1419        assert_eq!(dup("b.js"), 7, "three gap lines in b are not duplicated");
1420    }
1421
1422    #[test]
1423    fn display_path_applied_before_dup_matching() {
1424        let sources = vec![source("/abs/root/a.js", "javascript", &["x"], 5)];
1425        let clones = vec![clone_between("javascript", "a.js", "a.js", 2, 10)];
1426        let summary = compute_summary(&sources, &clones, 10, SummaryMetric::Tokens, |p| {
1427            p.strip_prefix("/abs/root/").unwrap_or(p).to_string()
1428        });
1429        assert_eq!(summary.files[0].path, "a.js");
1430        assert_eq!(summary.files[0].duplicated_lines, 4);
1431    }
1432
1433    #[test]
1434    fn folders_truncated_to_top_n_but_total_reported() {
1435        let sources: Vec<SourceFile> = (0..5)
1436            .map(|i| source(&format!("dir{i}/f.js"), "javascript", &["x"], 1))
1437            .collect();
1438        let summary = compute_summary(&sources, &[], 2, SummaryMetric::Tokens, identity);
1439        assert_eq!(summary.folders.len(), 2);
1440        assert_eq!(summary.total_folders, 5);
1441    }
1442
1443    #[test]
1444    fn metric_parses_from_str() {
1445        assert_eq!(
1446            "complexity".parse::<SummaryMetric>().unwrap(),
1447            SummaryMetric::Complexity
1448        );
1449        assert!("bogus".parse::<SummaryMetric>().is_err());
1450    }
1451
1452    #[test]
1453    fn summary_serializes_camel_case() {
1454        let sources = vec![source("a.js", "javascript", &["x"], 5)];
1455        let summary = compute_summary(&sources, &[], 10, SummaryMetric::Size, identity);
1456        let json = serde_json::to_string(&summary).unwrap();
1457        assert!(json.contains("\"totalFiles\""));
1458        assert!(json.contains("\"duplicatedLines\""));
1459        assert!(json.contains("\"by\":\"size\""));
1460        assert!(!json.contains("total_files"));
1461    }
1462}