use std::collections::{BTreeSet, HashMap, HashSet};
use std::path::{Path, PathBuf};
use std::rc::Rc;
use quote::ToTokens;
use serde::Serialize;
use syn::spanned::Spanned;
use syn::visit::Visit;
use crate::finding::{EvidenceClass, Finding, Location, OneBasedLine, Origin, Severity};
use crate::functions::{read_and_parse_source, walk_functions};
use crate::ingest::SourceFile;
pub const DUPLICATE_RULE: &str = "duplicate-code";
pub const DUPLICATE_RULE_REVISION: u32 = 2;
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum DupeMode {
Strict,
Mild,
Weak,
Semantic,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
enum LiteralKind {
Str,
ByteStr,
Byte,
Char,
Int,
Float,
Verbatim,
}
pub const DEFAULT_MIN_TOKENS: usize = 20;
#[derive(Debug, Clone)]
pub struct CloneMember {
pub qualified_name: String,
pub file: PathBuf,
pub start_line: usize,
pub end_line: usize,
pub start_token: usize,
pub end_token: usize,
pub token_count: usize,
pub mode: DupeMode,
pub identifier_mapping: Vec<(String, String)>,
pub normalized_literal_kinds: Vec<String>,
}
impl CloneMember {
pub fn to_finding(&self) -> Finding {
let evidence_class = match self.mode {
DupeMode::Strict | DupeMode::Mild => EvidenceClass::DerivedFact,
DupeMode::Weak | DupeMode::Semantic => EvidenceClass::Heuristic,
};
let mut evidence = serde_json::json!({ "token_count": self.token_count });
if !self.identifier_mapping.is_empty() {
let mapping: Vec<_> = self
.identifier_mapping
.iter()
.map(|(placeholder, identifier)| {
serde_json::json!({ "placeholder": placeholder, "identifier": identifier })
})
.collect();
evidence["identifier_mapping"] = serde_json::Value::Array(mapping);
}
if !self.normalized_literal_kinds.is_empty() {
evidence["normalized_literal_kinds"] = serde_json::Value::Array(
self.normalized_literal_kinds
.iter()
.map(|kind| serde_json::Value::String(kind.clone()))
.collect(),
);
}
Finding {
id: format!(
"{DUPLICATE_RULE}:{}:{}:{}-{}",
self.file.display(),
self.qualified_name,
self.start_token,
self.end_token
)
.into(),
rule: DUPLICATE_RULE.into(),
severity: Severity::Warn,
location: Location {
file: self.file.clone(),
line: OneBasedLine::new(self.start_line)
.expect("proc-macro2 span lines are 1-based"),
item_path: self.qualified_name.clone(),
},
evidence_class,
origin: Origin::Code,
evidence: Some(evidence),
limitations: None,
caused_by: Vec::new(),
causes: Vec::new(),
}
}
}
#[derive(Debug)]
pub struct CloneFamily {
pub members: Vec<CloneMember>,
}
#[derive(Debug, Serialize, PartialEq, Eq)]
pub struct RefactoringSummary {
pub schema_version: u32,
pub clone_families: usize,
pub clone_members: usize,
pub top_families: Vec<CloneFamilySummary>,
}
#[derive(Debug, Serialize, PartialEq, Eq)]
pub struct CloneFamilySummary {
pub rank: usize,
pub members: usize,
pub tokens_per_member: usize,
pub duplicated_token_mass: usize,
pub files: Vec<PathBuf>,
pub representative_items: Vec<String>,
}
#[derive(Debug)]
pub enum DuplicationError {
Io(PathBuf, std::io::Error),
Parse(PathBuf, syn::Error),
MissingSuppressionReason(PathBuf, usize),
DanglingItemSuppression(PathBuf, usize),
}
impl std::fmt::Display for DuplicationError {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
match self {
Self::Io(path, err) => write!(f, "{}: failed to read file: {err}", path.display()),
Self::Parse(path, err) => write!(f, "{}: failed to parse: {err}", path.display()),
Self::MissingSuppressionReason(path, line) => write!(
f,
"{}:{line}: duplication suppression requires a reason (`judge-dupe-ignore: <why>`)",
path.display()
),
Self::DanglingItemSuppression(path, line) => write!(
f,
"{}:{line}: `judge-dupe-ignore` must immediately precede a function item",
path.display()
),
}
}
}
impl std::error::Error for DuplicationError {
fn source(&self) -> Option<&(dyn std::error::Error + 'static)> {
match self {
Self::Io(_, err) => Some(err),
Self::Parse(_, err) => Some(err),
Self::MissingSuppressionReason(_, _) | Self::DanglingItemSuppression(_, _) => None,
}
}
}
#[derive(Debug, Default)]
pub struct WorkspaceDuplication {
pub families: Vec<CloneFamily>,
pub errors: Vec<DuplicationError>,
pub excluded_generated: usize,
}
impl WorkspaceDuplication {
pub fn to_findings(&self) -> Vec<Finding> {
self.families
.iter()
.flat_map(|family| family.members.iter().map(CloneMember::to_finding))
.collect()
}
pub fn refactoring_summary(&self, workspace_root: &Path, limit: usize) -> RefactoringSummary {
let mut families: Vec<CloneFamilySummary> = self
.families
.iter()
.map(|family| family_summary(family, workspace_root))
.collect();
families.sort_by(compare_refactoring_priority);
for (index, family) in families.iter_mut().enumerate() {
family.rank = index + 1;
}
let clone_members = self
.families
.iter()
.map(|family| family.members.len())
.sum();
families.truncate(limit);
RefactoringSummary {
schema_version: 1,
clone_families: self.families.len(),
clone_members,
top_families: families,
}
}
pub fn refactoring_order(&self, workspace_root: &Path) -> Vec<&CloneFamily> {
let mut families: Vec<&CloneFamily> = self.families.iter().collect();
families.sort_by(|left, right| {
compare_refactoring_priority(
&family_summary(left, workspace_root),
&family_summary(right, workspace_root),
)
});
families
}
}
fn family_summary(family: &CloneFamily, workspace_root: &Path) -> CloneFamilySummary {
let tokens_per_member = family
.members
.first()
.map_or(0, |member| member.token_count);
let files = family
.members
.iter()
.map(|member| relative_path(workspace_root, &member.file))
.collect::<BTreeSet<_>>()
.into_iter()
.collect();
let representative_items = family
.members
.iter()
.take(3)
.map(|member| member.qualified_name.clone())
.collect();
CloneFamilySummary {
rank: 0,
members: family.members.len(),
tokens_per_member,
duplicated_token_mass: family.members.len() * tokens_per_member,
files,
representative_items,
}
}
fn compare_refactoring_priority(
left: &CloneFamilySummary,
right: &CloneFamilySummary,
) -> std::cmp::Ordering {
right
.members
.cmp(&left.members)
.then_with(|| right.duplicated_token_mass.cmp(&left.duplicated_token_mass))
.then_with(|| left.files.cmp(&right.files))
.then_with(|| left.representative_items.cmp(&right.representative_items))
}
fn relative_path(root: &Path, path: &Path) -> PathBuf {
path.strip_prefix(root).unwrap_or(path).to_path_buf()
}
struct TokenUnit {
byte_start: usize,
byte_end: usize,
mild_text: String,
weak_text: String,
semantic_text: String,
literal_kind: Option<LiteralKind>,
is_ident: bool,
semantic_identifier_mapping: Option<(String, String)>,
start_line: usize,
end_line: usize,
}
impl TokenUnit {
fn new(span: proc_macro2::Span, mild_text: String) -> Self {
let range = span.byte_range();
Self {
byte_start: range.start,
byte_end: range.end,
weak_text: mild_text.clone(),
semantic_text: mild_text.clone(),
mild_text,
literal_kind: None,
is_ident: false,
semantic_identifier_mapping: None,
start_line: span.start().line,
end_line: span.end().line,
}
}
}
struct FuncTokens {
qualified_name: String,
file: PathBuf,
source: Rc<str>,
tokens: Vec<TokenUnit>,
suppressed: Rc<Vec<(usize, usize)>>,
item_suppressed: bool,
}
pub fn analyze_workspace<'a>(
source_files: impl IntoIterator<Item = &'a SourceFile>,
mode: DupeMode,
min_tokens: usize,
include_generated: bool,
) -> WorkspaceDuplication {
analyze_workspace_with_options(source_files, mode, min_tokens, include_generated, true)
}
pub fn analyze_workspace_with_options<'a>(
source_files: impl IntoIterator<Item = &'a SourceFile>,
mode: DupeMode,
min_tokens: usize,
include_generated: bool,
include_tests: bool,
) -> WorkspaceDuplication {
let min_tokens = min_tokens.max(1);
let mut functions = Vec::new();
let mut errors = Vec::new();
let mut excluded_generated = 0;
for file in source_files {
if !include_generated && !file.kind.is_locally_reportable() {
excluded_generated += 1;
continue;
}
if !include_tests && is_test_source_path(&file.path) {
continue;
}
match collect_function_tokens(&file.path, min_tokens, include_tests) {
Ok(mut found) => functions.append(&mut found),
Err(err) => errors.push(err),
}
}
let families = find_clone_families(&functions, mode, min_tokens);
WorkspaceDuplication {
families,
errors,
excluded_generated,
}
}
fn collect_function_tokens(
path: &Path,
min_tokens: usize,
include_tests: bool,
) -> Result<Vec<FuncTokens>, DuplicationError> {
let (source, ast) = read_and_parse_source(
path,
|err| DuplicationError::Io(path.to_path_buf(), err),
|err| DuplicationError::Parse(path.to_path_buf(), err),
)?;
let suppressed = Rc::new(suppressed_ranges(path, &source)?);
let mut item_suppressions = item_suppression_lines(path, &source)?;
let source: Rc<str> = Rc::from(source.into_boxed_str());
let mut functions = Vec::new();
walk_functions(&ast, |site| {
let item_suppressed = item_suppressions.remove(&site.span.start().line).is_some();
if !include_tests && site.is_test_context {
return;
}
let mut nested_functions = NestedFunctionRanges::default();
nested_functions.visit_block(site.block);
let mut tokens = Vec::new();
flatten_tokens(
site.block.to_token_stream(),
&mut tokens,
&nested_functions.ranges,
);
assign_semantic_text(&mut tokens);
if tokens.len() < min_tokens {
return;
}
functions.push(FuncTokens {
qualified_name: site.qualified_name,
file: path.to_path_buf(),
source: Rc::clone(&source),
tokens,
suppressed: Rc::clone(&suppressed),
item_suppressed,
});
});
if let Some((&_, &directive_line)) = item_suppressions.iter().next() {
return Err(DuplicationError::DanglingItemSuppression(
path.to_path_buf(),
directive_line,
));
}
Ok(functions)
}
fn is_test_source_path(path: &Path) -> bool {
path.components().any(|component| {
matches!(component, std::path::Component::Normal(name) if name == "tests" || name == "benches")
})
}
#[derive(Default)]
struct NestedFunctionRanges {
ranges: Vec<std::ops::Range<usize>>,
}
impl<'ast> Visit<'ast> for NestedFunctionRanges {
fn visit_item_fn(&mut self, node: &'ast syn::ItemFn) {
self.ranges.push(node.span().byte_range());
}
fn visit_impl_item_fn(&mut self, node: &'ast syn::ImplItemFn) {
self.ranges.push(node.span().byte_range());
}
fn visit_trait_item_fn(&mut self, node: &'ast syn::TraitItemFn) {
if node.default.is_some() {
self.ranges.push(node.span().byte_range());
}
}
}
fn flatten_tokens(
stream: proc_macro2::TokenStream,
out: &mut Vec<TokenUnit>,
excluded_ranges: &[std::ops::Range<usize>],
) {
let tokens: Vec<_> = stream.into_iter().collect();
let mut index = 0;
while index < tokens.len() {
if named_macro_argument_count(&tokens, index).is_some() {
index += 3;
continue;
}
let tt = tokens[index].clone();
index += 1;
let token_range = tt.span().byte_range();
if excluded_ranges
.iter()
.any(|excluded| excluded.start <= token_range.start && token_range.end <= excluded.end)
{
continue;
}
match tt {
proc_macro2::TokenTree::Group(group) => match group.delimiter() {
proc_macro2::Delimiter::None => {
flatten_tokens(group.stream(), out, excluded_ranges)
}
delimiter => {
let (open, close) = delimiter_chars(delimiter);
out.push(TokenUnit::new(group.span_open(), open.to_string()));
flatten_tokens(group.stream(), out, excluded_ranges);
out.push(TokenUnit::new(group.span_close(), close.to_string()));
}
},
proc_macro2::TokenTree::Ident(ident) => {
let mild_text = ident.to_string();
let mut token = TokenUnit::new(ident.span(), mild_text.clone());
if mild_text == "true" || mild_text == "false" {
token.weak_text = BOOL_LIT_PLACEHOLDER.to_string();
token.semantic_text = BOOL_LIT_PLACEHOLDER.to_string();
} else {
token.is_ident = true;
}
out.push(token);
}
proc_macro2::TokenTree::Punct(punct) => {
out.push(TokenUnit::new(punct.span(), punct.to_string()));
}
proc_macro2::TokenTree::Literal(lit) => {
let mild_text = lit.to_string();
let mut token = TokenUnit::new(lit.span(), mild_text);
let kind = literal_kind(&lit);
let placeholder = literal_placeholder(kind);
token.weak_text = placeholder.to_string();
token.semantic_text = placeholder.to_string();
token.literal_kind = Some(kind);
out.push(token);
}
}
}
}
fn named_macro_argument_count(tokens: &[proc_macro2::TokenTree], index: usize) -> Option<usize> {
let [
proc_macro2::TokenTree::Ident(_),
proc_macro2::TokenTree::Punct(bang),
proc_macro2::TokenTree::Group(group),
] = tokens.get(index..index + 3)?
else {
return None;
};
if bang.as_char() != '!' || group.delimiter() != proc_macro2::Delimiter::Brace {
return None;
}
let mut input = group.stream().into_iter();
matches!(
(input.next(), input.next()),
(
Some(proc_macro2::TokenTree::Ident(_)),
Some(proc_macro2::TokenTree::Punct(colon))
) if colon.as_char() == ':'
)
.then_some(3)
}
fn delimiter_chars(delimiter: proc_macro2::Delimiter) -> (&'static str, &'static str) {
match delimiter {
proc_macro2::Delimiter::Parenthesis => ("(", ")"),
proc_macro2::Delimiter::Brace => ("{", "}"),
proc_macro2::Delimiter::Bracket => ("[", "]"),
proc_macro2::Delimiter::None => ("", ""),
}
}
const BOOL_LIT_PLACEHOLDER: &str = "__BOOL_LIT__";
fn literal_kind(lit: &proc_macro2::Literal) -> LiteralKind {
match syn::Lit::new(lit.clone()) {
syn::Lit::Str(_) => LiteralKind::Str,
syn::Lit::ByteStr(_) => LiteralKind::ByteStr,
syn::Lit::Byte(_) => LiteralKind::Byte,
syn::Lit::Char(_) => LiteralKind::Char,
syn::Lit::Int(_) => LiteralKind::Int,
syn::Lit::Float(_) => LiteralKind::Float,
_ => LiteralKind::Verbatim,
}
}
fn literal_placeholder(kind: LiteralKind) -> &'static str {
match kind {
LiteralKind::Str => "__STR_LIT__",
LiteralKind::ByteStr => "__BYTESTR_LIT__",
LiteralKind::Byte => "__BYTE_LIT__",
LiteralKind::Char => "__CHAR_LIT__",
LiteralKind::Int => "__INT_LIT__",
LiteralKind::Float => "__FLOAT_LIT__",
LiteralKind::Verbatim => "__LIT__",
}
}
fn literal_kind_name(kind: LiteralKind) -> &'static str {
match kind {
LiteralKind::Str => "str",
LiteralKind::ByteStr => "bytestr",
LiteralKind::Byte => "byte",
LiteralKind::Char => "char",
LiteralKind::Int => "int",
LiteralKind::Float => "float",
LiteralKind::Verbatim => "verbatim",
}
}
const RUST_KEYWORDS: &[&str] = &[
"as", "async", "await", "break", "const", "continue", "dyn", "else", "enum", "extern", "fn",
"for", "if", "impl", "in", "let", "loop", "match", "mod", "move", "mut", "pub", "ref",
"return", "static", "struct", "trait", "type", "unsafe", "use", "where", "while", "abstract",
"become", "box", "do", "final", "macro", "override", "priv", "try", "typeof", "unsized",
"virtual", "yield", "union",
];
fn is_rust_keyword(name: &str) -> bool {
RUST_KEYWORDS.contains(&name)
}
fn assign_semantic_text(tokens: &mut [TokenUnit]) {
let mut numbering: HashMap<String, usize> = HashMap::new();
for index in 0..tokens.len() {
if !tokens[index].is_ident {
continue;
}
let next = tokens.get(index + 1).map(|t| t.mild_text.as_str());
let next2 = tokens.get(index + 2).map(|t| t.mild_text.as_str());
let prev = index
.checked_sub(1)
.and_then(|i| tokens.get(i))
.map(|t| t.mild_text.as_str());
let prev2 = index
.checked_sub(2)
.and_then(|i| tokens.get(i))
.map(|t| t.mild_text.as_str());
let name = tokens[index].mild_text.as_str();
let keep_literal = next == Some("(")
|| next == Some("!")
|| (prev == Some(":") && prev2 == Some(":"))
|| (next == Some(":") && next2 == Some(":"))
|| prev == Some(".")
|| matches!(name, "self" | "Self" | "crate" | "super")
|| name.chars().next().is_some_and(char::is_uppercase)
|| is_rust_keyword(name);
if keep_literal {
continue;
}
let name = tokens[index].mild_text.clone();
let next_n = numbering.len();
let n = *numbering.entry(name.clone()).or_insert(next_n);
let placeholder = format!("__ID_{n}__");
tokens[index].semantic_text = placeholder.clone();
tokens[index].semantic_identifier_mapping = Some((placeholder, name));
}
}
fn suppressed_ranges(path: &Path, source: &str) -> Result<Vec<(usize, usize)>, DuplicationError> {
const OFF: &str = "// judge-dupe-off:";
const ON: &str = "// judge-dupe-on";
let mut ranges = Vec::new();
let mut open: Option<usize> = None;
for (index, line) in source.lines().enumerate() {
let line_number = index + 1;
if let Some(at) = line.find(OFF) {
require_suppression_reason(path, line, at + OFF.len(), line_number)?;
open.get_or_insert(line_number);
} else if line.contains(ON)
&& let Some(start) = open.take()
{
ranges.push((start, line_number));
}
}
if let Some(start) = open {
ranges.push((start, usize::MAX));
}
Ok(ranges)
}
fn require_suppression_reason(
path: &Path,
line: &str,
marker_end: usize,
line_number: usize,
) -> Result<(), DuplicationError> {
if line[marker_end..].trim().is_empty() {
return Err(DuplicationError::MissingSuppressionReason(
path.to_path_buf(),
line_number,
));
}
Ok(())
}
fn item_suppression_lines(
path: &Path,
source: &str,
) -> Result<HashMap<usize, usize>, DuplicationError> {
let ignore = ["// judge", "-dupe-ignore:"].concat();
let mut targets = HashMap::new();
for (index, line) in source.lines().enumerate() {
let line_number = index + 1;
let Some(at) = line.find(&ignore) else {
continue;
};
require_suppression_reason(path, line, at + ignore.len(), line_number)?;
targets.insert(line_number + 1, line_number);
}
Ok(targets)
}
const FINGERPRINT_BASE: u64 = 0x0000_0100_0000_01b3;
struct FuncFingerprint {
ids: Vec<u64>,
prefix: Vec<u64>,
byte_base: usize,
}
struct Fingerprints {
per_func: Vec<FuncFingerprint>,
pow: Vec<u64>,
}
fn build_fingerprints(functions: &[FuncTokens], mode: DupeMode) -> Fingerprints {
fn prefix_hashes(units: impl Iterator<Item = u64>, capacity: usize) -> Vec<u64> {
let mut prefix = Vec::with_capacity(capacity + 1);
let mut hash = 0u64;
prefix.push(hash);
for unit in units {
hash = hash.wrapping_mul(FINGERPRINT_BASE).wrapping_add(unit);
prefix.push(hash);
}
prefix
}
let mut interner: HashMap<String, u64> = HashMap::new();
let mut per_func = Vec::with_capacity(functions.len());
for func in functions {
let fingerprint = if mode == DupeMode::Strict {
let byte_base = func.tokens[0].byte_start;
let byte_end = func.tokens[func.tokens.len() - 1].byte_end;
let bytes = func.source[byte_base..byte_end].as_bytes();
FuncFingerprint {
ids: Vec::new(),
prefix: prefix_hashes(bytes.iter().map(|&b| u64::from(b) + 1), bytes.len()),
byte_base,
}
} else {
let mut ids = Vec::with_capacity(func.tokens.len());
for token in &func.tokens {
let text = match mode {
DupeMode::Strict => unreachable!(),
DupeMode::Mild => &token.mild_text,
DupeMode::Weak => &token.weak_text,
DupeMode::Semantic => &token.semantic_text,
};
let id = match interner.get(text.as_str()) {
Some(&id) => id,
None => {
let id = interner.len() as u64 + 1;
interner.insert(text.clone(), id);
id
}
};
ids.push(id);
}
let prefix = prefix_hashes(ids.iter().copied(), ids.len());
FuncFingerprint {
ids,
prefix,
byte_base: 0,
}
};
per_func.push(fingerprint);
}
let max_units = per_func
.iter()
.map(|fingerprint| fingerprint.prefix.len())
.max()
.unwrap_or(1);
let mut pow = Vec::with_capacity(max_units);
pow.push(1u64);
for k in 1..max_units {
pow.push(pow[k - 1].wrapping_mul(FINGERPRINT_BASE));
}
Fingerprints { per_func, pow }
}
impl Fingerprints {
fn window_hash(
&self,
functions: &[FuncTokens],
mode: DupeMode,
func: usize,
start: usize,
end: usize,
) -> u64 {
let fingerprint = &self.per_func[func];
let (from, to) = match mode {
DupeMode::Strict => {
let tokens = &functions[func].tokens;
(
tokens[start].byte_start - fingerprint.byte_base,
tokens[end - 1].byte_end - fingerprint.byte_base,
)
}
DupeMode::Mild | DupeMode::Weak | DupeMode::Semantic => (start, end),
};
fingerprint.prefix[to]
.wrapping_sub(fingerprint.prefix[from].wrapping_mul(self.pow[to - from]))
}
#[allow(clippy::too_many_arguments)]
fn spans_equal(
&self,
functions: &[FuncTokens],
mode: DupeMode,
func_a: usize,
start_a: usize,
end_a: usize,
func_b: usize,
start_b: usize,
end_b: usize,
) -> bool {
match mode {
DupeMode::Strict => {
let a = &functions[func_a];
let b = &functions[func_b];
a.source[a.tokens[start_a].byte_start..a.tokens[end_a - 1].byte_end]
== b.source[b.tokens[start_b].byte_start..b.tokens[end_b - 1].byte_end]
}
DupeMode::Mild | DupeMode::Weak | DupeMode::Semantic => {
end_a - start_a == end_b - start_b
&& self.per_func[func_a].ids[start_a..end_a]
== self.per_func[func_b].ids[start_b..end_b]
}
}
}
}
fn find_clone_families(
functions: &[FuncTokens],
mode: DupeMode,
min_tokens: usize,
) -> Vec<CloneFamily> {
let fingerprints = build_fingerprints(functions, mode);
let mut seeds: HashMap<u64, Vec<(usize, usize)>> = HashMap::new();
for (func_index, func) in functions.iter().enumerate() {
let n = func.tokens.len();
if n < min_tokens {
continue;
}
for start in 0..=(n - min_tokens) {
let hash =
fingerprints.window_hash(functions, mode, func_index, start, start + min_tokens);
seeds.entry(hash).or_default().push((func_index, start));
}
}
let mut matches: HashSet<(usize, usize, usize, usize, usize, usize)> = HashSet::new();
for occurrences in seeds.values() {
if occurrences.len() < 2 {
continue;
}
for i in 0..occurrences.len() {
let (func_a, start_a) = occurrences[i];
for &(func_b, start_b) in &occurrences[i + 1..] {
if func_a == func_b {
continue;
}
if start_a > 0
&& start_b > 0
&& backward_step_matches(
&functions[func_a],
&functions[func_b],
start_a,
start_b,
0,
mode,
)
{
continue;
}
if !fingerprints.spans_equal(
functions,
mode,
func_a,
start_a,
start_a + min_tokens,
func_b,
start_b,
start_b + min_tokens,
) {
continue;
}
let (sa, ea, sb, eb) = extend_match(
functions, func_a, start_a, func_b, start_b, min_tokens, mode,
);
matches.insert((func_a, sa, ea, func_b, sb, eb));
}
}
}
let mut group_members: Vec<Vec<CloneMember>> = Vec::new();
let mut group_reps: Vec<(usize, usize, usize)> = Vec::new();
let mut group_index: HashMap<(u64, usize), Vec<usize>> = HashMap::new();
for (func_a, sa, ea, func_b, sb, eb) in matches {
let a = &functions[func_a];
let b = &functions[func_b];
if is_suppressed(a, sa, ea) || is_suppressed(b, sb, eb) {
continue;
}
let hash = fingerprints.window_hash(functions, mode, func_a, sa, ea);
let candidates = group_index.entry((hash, ea - sa)).or_default();
let group = candidates
.iter()
.copied()
.find(|&group| {
let (rep_func, rep_start, rep_end) = group_reps[group];
fingerprints.spans_equal(
functions, mode, rep_func, rep_start, rep_end, func_a, sa, ea,
)
})
.unwrap_or_else(|| {
let group = group_members.len();
group_members.push(Vec::new());
group_reps.push((func_a, sa, ea));
candidates.push(group);
group
});
let members = &mut group_members[group];
push_unique(members, member_from(a, sa, ea, mode));
push_unique(members, member_from(b, sb, eb, mode));
}
let mut families: Vec<CloneFamily> = group_members
.into_iter()
.filter(|members| members.len() > 1)
.map(|mut members| {
members.sort_by(|x, y| x.file.cmp(&y.file).then(x.start_line.cmp(&y.start_line)));
CloneFamily { members }
})
.collect();
dedupe_contained_spans(&mut families);
families.retain(|family| family.members.len() > 1);
families.sort_by_key(|family| std::cmp::Reverse(family.members.len()));
families
}
fn extend_match(
functions: &[FuncTokens],
func_a: usize,
start_a: usize,
func_b: usize,
start_b: usize,
seed_len: usize,
mode: DupeMode,
) -> (usize, usize, usize, usize) {
let a = &functions[func_a];
let b = &functions[func_b];
let mut back = 0;
while start_a > back
&& start_b > back
&& backward_step_matches(a, b, start_a, start_b, back, mode)
{
back += 1;
}
let mut fwd = seed_len;
while start_a + fwd < a.tokens.len()
&& start_b + fwd < b.tokens.len()
&& forward_step_matches(a, b, start_a, start_b, fwd, mode)
{
fwd += 1;
}
(start_a - back, start_a + fwd, start_b - back, start_b + fwd)
}
fn forward_step_matches(
a: &FuncTokens,
b: &FuncTokens,
start_a: usize,
start_b: usize,
fwd: usize,
mode: DupeMode,
) -> bool {
match mode {
DupeMode::Strict => {
let prev_a = a.tokens[start_a + fwd - 1].byte_end;
let prev_b = b.tokens[start_b + fwd - 1].byte_end;
let next_a = a.tokens[start_a + fwd].byte_end;
let next_b = b.tokens[start_b + fwd].byte_end;
a.source[prev_a..next_a] == b.source[prev_b..next_b]
}
DupeMode::Mild => a.tokens[start_a + fwd].mild_text == b.tokens[start_b + fwd].mild_text,
DupeMode::Weak => a.tokens[start_a + fwd].weak_text == b.tokens[start_b + fwd].weak_text,
DupeMode::Semantic => {
a.tokens[start_a + fwd].semantic_text == b.tokens[start_b + fwd].semantic_text
}
}
}
fn backward_step_matches(
a: &FuncTokens,
b: &FuncTokens,
start_a: usize,
start_b: usize,
back: usize,
mode: DupeMode,
) -> bool {
match mode {
DupeMode::Strict => {
let new_a = a.tokens[start_a - back - 1].byte_start;
let new_b = b.tokens[start_b - back - 1].byte_start;
let old_a = a.tokens[start_a - back].byte_start;
let old_b = b.tokens[start_b - back].byte_start;
a.source[new_a..old_a] == b.source[new_b..old_b]
}
DupeMode::Mild => {
a.tokens[start_a - back - 1].mild_text == b.tokens[start_b - back - 1].mild_text
}
DupeMode::Weak => {
a.tokens[start_a - back - 1].weak_text == b.tokens[start_b - back - 1].weak_text
}
DupeMode::Semantic => {
a.tokens[start_a - back - 1].semantic_text == b.tokens[start_b - back - 1].semantic_text
}
}
}
fn is_suppressed(func: &FuncTokens, start: usize, end: usize) -> bool {
if func.item_suppressed {
return true;
}
let start_line = func.tokens[start].start_line;
let end_line = func.tokens[end - 1].end_line;
func.suppressed
.iter()
.any(|&(off, on)| off <= start_line && end_line <= on)
}
fn member_from(func: &FuncTokens, start: usize, end: usize, mode: DupeMode) -> CloneMember {
let mut identifier_mapping = Vec::new();
let mut normalized_literal_kinds = Vec::new();
if matches!(mode, DupeMode::Weak | DupeMode::Semantic) {
for token in &func.tokens[start..end] {
if let Some(kind) = token.literal_kind {
let name = literal_kind_name(kind).to_string();
if !normalized_literal_kinds.contains(&name) {
normalized_literal_kinds.push(name);
}
}
}
}
if mode == DupeMode::Semantic {
for token in &func.tokens[start..end] {
if let Some(mapping) = &token.semantic_identifier_mapping
&& !identifier_mapping.contains(mapping)
{
identifier_mapping.push(mapping.clone());
}
}
}
CloneMember {
qualified_name: func.qualified_name.clone(),
file: func.file.clone(),
start_line: func.tokens[start].start_line,
end_line: func.tokens[end - 1].end_line,
start_token: start,
end_token: end - 1,
token_count: end - start,
mode,
identifier_mapping,
normalized_literal_kinds,
}
}
fn push_unique(members: &mut Vec<CloneMember>, member: CloneMember) {
let already_present = members.iter().any(|existing| {
existing.file == member.file
&& existing.qualified_name == member.qualified_name
&& existing.start_token == member.start_token
&& existing.end_token == member.end_token
});
if !already_present {
members.push(member);
}
}
fn dedupe_contained_spans(families: &mut [CloneFamily]) {
let mut locations = Vec::new();
for (family_index, family) in families.iter().enumerate() {
for (member_index, member) in family.members.iter().enumerate() {
locations.push((
family_index,
member_index,
member.file.clone(),
member.qualified_name.clone(),
member.start_token,
member.end_token,
));
}
}
let mut drop: HashSet<(usize, usize)> = HashSet::new();
for (fi_a, mi_a, file_a, name_a, start_a, end_a) in &locations {
for (fi_b, mi_b, file_b, name_b, start_b, end_b) in &locations {
if (fi_a, mi_a) == (fi_b, mi_b) || file_a != file_b || name_a != name_b {
continue;
}
let contained = start_a >= start_b && end_a <= end_b;
if contained {
drop.insert((*fi_a, *mi_a));
}
}
}
for (family_index, family) in families.iter_mut().enumerate() {
let mut member_index = 0;
family.members.retain(|_| {
let keep = !drop.contains(&(family_index, member_index));
member_index += 1;
keep
});
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::ingest::SourceKind;
use crate::test_util::TempDir;
fn authored(paths: impl IntoIterator<Item = PathBuf>) -> Vec<SourceFile> {
paths
.into_iter()
.map(|path| SourceFile {
path,
kind: SourceKind::Authored,
})
.collect()
}
fn write_duplicate_fixtures(dir: &TempDir) -> (PathBuf, PathBuf) {
let file_a = dir.join("a.rs");
let file_b = dir.join("b.rs");
std::fs::write(
&file_a,
r#"
fn dup_one(x: i32) -> i32 {
let mut total = 0;
for i in 0..x {
total += i;
}
total
}
fn unique_one() -> i32 {
let mut total = 0;
for i in 0..3 {
total += i * 2;
}
total
}
"#,
)
.unwrap();
std::fs::write(
&file_b,
r#"
fn dup_two(x: i32) -> i32 {
// reformatted duplicate of dup_one
let mut total = 0;
for i in 0..x {
total += i;
}
total
}
"#,
)
.unwrap();
(file_a, file_b)
}
#[test]
fn mild_mode_ignores_whitespace_and_comments() {
let dir = TempDir::new("dup-mild");
let (file_a, file_b) = write_duplicate_fixtures(&dir);
let files = authored([file_a, file_b]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
assert_eq!(report.families.len(), 1);
let members = &report.families[0].members;
let names: Vec<_> = members.iter().map(|m| m.qualified_name.as_str()).collect();
assert_eq!(names, ["dup_one", "dup_two"]);
}
#[test]
fn refactoring_summary_is_compact_and_workspace_relative() {
let dir = TempDir::new("dup-summary");
let (file_a, file_b) = write_duplicate_fixtures(&dir);
let files = authored([file_a, file_b]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
let summary = report.refactoring_summary(&dir, 5);
assert_eq!(summary.schema_version, 1);
assert_eq!(summary.clone_families, 1);
assert_eq!(summary.clone_members, 2);
assert_eq!(summary.top_families.len(), 1);
assert_eq!(summary.top_families[0].rank, 1);
assert_eq!(
summary.top_families[0].files,
vec![PathBuf::from("a.rs"), PathBuf::from("b.rs")]
);
assert_eq!(
summary.top_families[0].representative_items,
vec!["dup_one", "dup_two"]
);
}
#[test]
fn test_only_functions_are_excluded_unless_requested() {
let dir = TempDir::new("dup-test-context");
let file = dir.join("lib.rs");
std::fs::write(
&file,
r#"
fn production() {}
#[cfg(test)]
mod tests {
fn first() {
let mut total = 0;
for value in 0..10 { total += value; }
if total > 0 { total -= 1; }
let _ = total;
}
fn second() {
let mut total = 0;
for value in 0..10 { total += value; }
if total > 0 { total -= 1; }
let _ = total;
}
}
"#,
)
.unwrap();
let files = authored([file]);
let excluded = analyze_workspace_with_options(
files.iter(),
DupeMode::Mild,
DEFAULT_MIN_TOKENS,
false,
false,
);
let included = analyze_workspace_with_options(
files.iter(),
DupeMode::Mild,
DEFAULT_MIN_TOKENS,
false,
true,
);
assert!(excluded.families.is_empty());
assert_eq!(included.families.len(), 1);
}
#[test]
fn strict_mode_requires_byte_identical_spans() {
let dir = TempDir::new("dup-strict");
let (file_a, file_b) = write_duplicate_fixtures(&dir);
let files = authored([file_a, file_b]);
let report = analyze_workspace(files.iter(), DupeMode::Strict, DEFAULT_MIN_TOKENS, false);
assert_eq!(report.families.len(), 1);
let members = &report.families[0].members;
let names: Vec<_> = members.iter().map(|m| m.qualified_name.as_str()).collect();
assert_eq!(names, ["dup_one", "dup_two"]);
for member in members {
assert!(member.token_count < DEFAULT_MIN_TOKENS + 5);
}
}
#[test]
fn spans_shorter_than_the_minimum_are_excluded() {
let dir = TempDir::new("dup-too-short");
let file = dir.join("short.rs");
std::fs::write(
&file,
"fn short_one() -> i32 { 1 }\nfn short_two() -> i32 { 1 }\n",
)
.unwrap();
let files = authored([file]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
assert!(report.families.is_empty());
}
#[test]
fn analyze_workspace_reports_parse_errors() {
let dir = TempDir::new("dup-parse-error");
let file = dir.join("broken.rs");
std::fs::write(&file, "fn broken( {").unwrap();
let files = authored([file]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
assert_eq!(report.errors.len(), 1);
}
#[test]
fn finds_a_duplicated_span_nested_inside_a_larger_unique_function() {
let dir = TempDir::new("dup-nested-span");
let file = dir.join("nested.rs");
std::fs::write(
&file,
r#"
fn contains_dup_block(n: i32) -> i32 {
let a = 1;
let b = 2;
let mut total = 0;
for i in 0..n {
total += i;
}
let c = a + b;
c + total
}
fn other_contains_dup_block(n: i32) -> i32 {
let unrelated = 7;
let mut total = 0;
for i in 0..n {
total += i;
}
unrelated * total
}
"#,
)
.unwrap();
let files = authored([file]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, 10, false);
assert_eq!(report.families.len(), 1);
let members = &report.families[0].members;
assert_eq!(members.len(), 2);
let names: Vec<_> = members.iter().map(|m| m.qualified_name.as_str()).collect();
assert_eq!(names, ["contains_dup_block", "other_contains_dup_block"]);
let contains = members
.iter()
.find(|m| m.qualified_name == "contains_dup_block")
.unwrap();
assert!(contains.start_token > 0);
}
#[test]
fn nested_function_is_not_reported_as_a_duplicate_of_its_parent() {
let dir = TempDir::new("dup-nested-function");
let file = dir.join("nested.rs");
std::fs::write(
&file,
r#"
fn outer() {
fn inner() {
let mut total = 0;
total += 1;
total += 2;
total += 3;
total += 4;
println!("{total}");
}
inner();
}
"#,
)
.unwrap();
let files = authored([file]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, 8, false);
assert!(report.families.is_empty());
}
#[test]
fn contained_spans_are_deduped_to_the_maximal_match() {
let dir = TempDir::new("dup-contained-span");
let file = dir.join("contained.rs");
std::fs::write(
&file,
r#"
fn fn_big(x: i32) -> i32 {
let mut total = 0;
total += x;
total += x * 2;
total
}
fn fn_partner_one(x: i32) -> i32 {
let mut total = 0;
total += x;
total += x * 2;
total
}
fn fn_partner_two(x: i32) -> i32 {
let mut total = 0;
total += x;
}
"#,
)
.unwrap();
let files = authored([file]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, 10, false);
assert_eq!(report.families.len(), 1);
let members = &report.families[0].members;
let names: Vec<_> = members.iter().map(|m| m.qualified_name.as_str()).collect();
assert_eq!(names, ["fn_big", "fn_partner_one"]);
}
#[test]
fn judge_dupe_off_suppresses_a_fully_contained_span() {
let dir = TempDir::new("dup-suppressed");
let file_a = dir.join("a.rs");
let file_b = dir.join("b.rs");
std::fs::write(
&file_a,
r#"
// judge-dupe-off: intentional protocol table duplication
fn dup_one(x: i32) -> i32 {
let mut total = 0;
for i in 0..x {
total += i;
}
total
}
// judge-dupe-on
"#,
)
.unwrap();
std::fs::write(
&file_b,
r#"
fn dup_two(x: i32) -> i32 {
let mut total = 0;
for i in 0..x {
total += i;
}
total
}
"#,
)
.unwrap();
let files = authored([file_a, file_b]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
assert!(report.families.is_empty());
}
#[test]
fn judge_dupe_off_without_a_reason_is_a_hard_error() {
let dir = TempDir::new("dup-missing-reason");
let file = dir.join("bad_suppression.rs");
let missing_reason_marker = ["// judge", "-dupe-off:"].concat();
std::fs::write(
&file,
(r#"
fn dup_one(x: i32) -> i32 {
MISSING_REASON_MARKER
let mut total = 0;
for i in 0..x {
total += i;
}
total
// judge-dupe-on
}
"#)
.replace("MISSING_REASON_MARKER", &missing_reason_marker),
)
.unwrap();
let files = authored([file]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
assert_eq!(report.errors.len(), 1);
match &report.errors[0] {
DuplicationError::MissingSuppressionReason(_, line) => assert_eq!(*line, 3),
other => panic!("expected a missing-reason error, got {other:?}"),
}
}
#[test]
fn judge_dupe_ignore_suppresses_only_the_immediately_following_function() {
let dir = TempDir::new("dup-item-suppressed");
let file_a = dir.join("a.rs");
let file_b = dir.join("b.rs");
let item_suppression_marker =
["// judge", "-dupe-ignore: required external protocol shape"].concat();
std::fs::write(
&file_a,
r#"
ITEM_SUPPRESSION_MARKER
fn protocol_one(x: i32) -> i32 {
let mut total = 0;
for i in 0..x {
total += i;
}
total
}
"#
.replace("ITEM_SUPPRESSION_MARKER", &item_suppression_marker),
)
.unwrap();
std::fs::write(
&file_b,
r#"
fn protocol_two(x: i32) -> i32 {
let mut total = 0;
for i in 0..x {
total += i;
}
total
}
"#,
)
.unwrap();
let files = authored([file_a, file_b]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
assert!(report.errors.is_empty(), "{:?}", report.errors);
assert!(report.families.is_empty());
}
#[test]
fn judge_dupe_ignore_without_a_reason_is_a_hard_error() {
let dir = TempDir::new("dup-item-missing-reason");
let file = dir.join("bad_item_suppression.rs");
let missing_reason_marker = ["// judge", "-dupe-ignore:"].concat();
std::fs::write(
&file,
(r#"
MISSING_REASON_MARKER
fn dup_one() {}
"#)
.replace("MISSING_REASON_MARKER", &missing_reason_marker),
)
.unwrap();
let files = authored([file]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
assert!(matches!(
report.errors.as_slice(),
[DuplicationError::MissingSuppressionReason(_, 2)]
));
}
#[test]
fn judge_dupe_ignore_requires_an_immediately_following_function() {
let dir = TempDir::new("dup-item-dangling");
let file = dir.join("dangling_item_suppression.rs");
let item_suppression_marker = ["// judge", "-dupe-ignore: protocol shape"].concat();
std::fs::write(
&file,
r#"
ITEM_SUPPRESSION_MARKER
fn dup_one() {}
"#
.replace("ITEM_SUPPRESSION_MARKER", &item_suppression_marker),
)
.unwrap();
let files = authored([file]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
assert!(matches!(
report.errors.as_slice(),
[DuplicationError::DanglingItemSuppression(_, 2)]
));
}
#[test]
fn judge_dupe_ignore_before_a_test_function_is_valid_when_tests_are_excluded() {
let dir = TempDir::new("dup-item-test-suppression");
let file = dir.join("test_suppression.rs");
let item_suppression_marker = ["// judge", "-dupe-ignore: fixture protocol shape"].concat();
std::fs::write(
&file,
r#"
#[cfg(test)]
mod tests {
ITEM_SUPPRESSION_MARKER
fn fixture_helper() {}
}
"#
.replace("ITEM_SUPPRESSION_MARKER", &item_suppression_marker),
)
.unwrap();
let files = authored([file]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
assert!(report.errors.is_empty(), "{:?}", report.errors);
}
#[test]
fn to_findings_emits_one_warn_finding_per_member() {
let dir = TempDir::new("dup-findings");
let (file_a, file_b) = write_duplicate_fixtures(&dir);
let files = authored([file_a, file_b]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
let findings = report.to_findings();
assert_eq!(findings.len(), 2);
for finding in &findings {
assert_eq!(finding.rule, DUPLICATE_RULE);
assert_eq!(finding.severity, Severity::Warn);
assert_eq!(finding.origin, Origin::Code);
}
let ids: HashSet<_> = findings.iter().map(|f| f.id.as_str()).collect();
assert_eq!(ids.len(), 2, "each member must get a distinct id");
}
#[test]
fn duplicate_code_registry_example_still_triggers_the_rule() {
let example = crate::rule_registry::lookup(DUPLICATE_RULE)
.expect("duplicate-code has a registry entry")
.example
.expect("duplicate-code has a curated example")
.before;
let dir = TempDir::new("dup-registry-example");
let file = dir.join("shipping.rs");
std::fs::write(&file, example).unwrap();
let files = authored([file]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
let findings = report.to_findings();
assert_eq!(
findings.iter().filter(|f| f.rule == DUPLICATE_RULE).count(),
2
);
}
#[test]
fn generated_files_are_excluded_unless_included() {
let dir = TempDir::new("dup-generated");
let (file_a, file_b) = write_duplicate_fixtures(&dir);
let files = [
SourceFile {
path: file_a,
kind: SourceKind::Authored,
},
SourceFile {
path: file_b,
kind: SourceKind::Generated,
},
];
let excluded = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
assert!(excluded.families.is_empty());
assert_eq!(excluded.excluded_generated, 1);
let included = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, true);
assert_eq!(included.families.len(), 1);
assert_eq!(included.excluded_generated, 0);
}
#[test]
fn weak_mode_normalizes_literals() {
let dir = TempDir::new("dup-weak-literal");
let file = dir.join("lit.rs");
std::fs::write(
&file,
r#"
fn lit_one(x: i32) -> i32 {
let a = 1;
let b = 2;
let c = 3;
a + b + c + x
}
fn lit_two(x: i32) -> i32 {
let a = 1;
let b = 2;
let c = 99;
a + b + c + x
}
"#,
)
.unwrap();
let files = authored([file]);
let mild = analyze_workspace(files.iter(), DupeMode::Mild, 15, false);
assert!(mild.families.is_empty());
let weak = analyze_workspace(files.iter(), DupeMode::Weak, 15, false);
assert_eq!(weak.families.len(), 1);
let names: Vec<_> = weak.families[0]
.members
.iter()
.map(|m| m.qualified_name.as_str())
.collect();
assert_eq!(names, ["lit_one", "lit_two"]);
}
#[test]
fn semantic_mode_matches_renamed_clone() {
let dir = TempDir::new("dup-semantic-rename");
let file = dir.join("rename.rs");
std::fs::write(
&file,
r#"
fn semantic_one(x: i32) -> i32 {
let mut total = 0;
for i in 0..x {
total += i;
}
total
}
fn semantic_two(x: i32) -> i32 {
let mut sum = 0;
for idx in 0..x {
sum += idx;
}
sum
}
"#,
)
.unwrap();
let files = authored([file]);
let mild = analyze_workspace(files.iter(), DupeMode::Mild, 15, false);
assert!(mild.families.is_empty());
let semantic = analyze_workspace(files.iter(), DupeMode::Semantic, 15, false);
assert_eq!(semantic.families.len(), 1);
let members = &semantic.families[0].members;
let names: Vec<_> = members.iter().map(|m| m.qualified_name.as_str()).collect();
assert_eq!(names, ["semantic_one", "semantic_two"]);
let findings = WorkspaceDuplication {
families: semantic.families,
errors: Vec::new(),
excluded_generated: 0,
}
.to_findings();
let one_evidence = findings
.iter()
.find(|f| f.location.item_path == "semantic_one")
.unwrap()
.evidence
.clone()
.unwrap();
let one_mapping = one_evidence["identifier_mapping"].as_array().unwrap();
assert!(one_mapping.contains(&serde_json::json!({
"placeholder": "__ID_0__",
"identifier": "total"
})));
assert!(one_mapping.contains(&serde_json::json!({
"placeholder": "__ID_1__",
"identifier": "i"
})));
let two_evidence = findings
.iter()
.find(|f| f.location.item_path == "semantic_two")
.unwrap()
.evidence
.clone()
.unwrap();
let two_mapping = two_evidence["identifier_mapping"].as_array().unwrap();
assert!(two_mapping.contains(&serde_json::json!({
"placeholder": "__ID_0__",
"identifier": "sum"
})));
assert!(two_mapping.contains(&serde_json::json!({
"placeholder": "__ID_1__",
"identifier": "idx"
})));
}
#[test]
fn semantic_mode_does_not_collapse_different_identifier_reuse_patterns() {
let dir = TempDir::new("dup-semantic-reuse-pattern");
let file = dir.join("reuse.rs");
std::fs::write(
&file,
r#"
fn distinct_params(a: i32, b: i32) -> i32 {
let x = 1;
let y = 2;
let z = 3;
a + b + x + y + z
}
fn reused_param(a: i32) -> i32 {
let x = 1;
let y = 2;
let z = 3;
a + a + x + y + z
}
"#,
)
.unwrap();
let files = authored([file]);
let report = analyze_workspace(files.iter(), DupeMode::Semantic, 19, false);
assert!(
report.families.is_empty(),
"a distinct a+b vs reused a+a pattern must not collapse into one family, got: {:?}",
report.families
);
}
#[test]
fn semantic_mode_keeps_call_names_literal() {
let dir = TempDir::new("dup-semantic-call-names");
let file = dir.join("calls.rs");
std::fs::write(
&file,
r#"
fn calls_helper_one(x: i32) -> i32 {
let y = 1;
let z = 2;
helper_one(x) + y + z
}
fn calls_helper_two(x: i32) -> i32 {
let y = 1;
let z = 2;
helper_two(x) + y + z
}
"#,
)
.unwrap();
let files = authored([file]);
let report = analyze_workspace(files.iter(), DupeMode::Semantic, 12, false);
assert!(
report.families.is_empty(),
"calls to different helper functions must not collapse under Semantic mode, got: {:?}",
report.families
);
}
fn write_repetitive_fixture(dir: &TempDir, reps: usize) -> PathBuf {
let file = dir.join("repetitive.rs");
let mut source = String::new();
for name in ["rep_one", "rep_two"] {
source.push_str(&format!("fn {name}(x: i32) -> i32 {{\n"));
source.push_str(" let mut total = 0;\n");
for _ in 0..reps {
source.push_str(" total += x * 3 + 1;\n");
source.push_str(" total -= x / 2;\n");
}
source.push_str(" total\n}\n");
}
std::fs::write(&file, source).unwrap();
file
}
#[test]
fn strongly_repetitive_file_terminates_in_reasonable_time() {
let dir = TempDir::new("dup-repetitive");
let file = write_repetitive_fixture(&dir, 1000);
let files = authored([file]);
let started = std::time::Instant::now();
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
let elapsed = started.elapsed();
assert_eq!(report.families.len(), 1);
let members = &report.families[0].members;
let names: Vec<_> = members.iter().map(|m| m.qualified_name.as_str()).collect();
assert_eq!(names, ["rep_one", "rep_two"]);
assert!(members.iter().all(|m| m.token_count > 16_000));
assert!(
elapsed < std::time::Duration::from_secs(30),
"repetitive file took {elapsed:?}"
);
}
#[test]
fn weak_and_semantic_are_heuristic_while_strict_and_mild_are_derived_facts() {
let dir = TempDir::new("dup-evidence-class");
let (file_a, file_b) = write_duplicate_fixtures(&dir);
let files = authored([file_a, file_b]);
for (mode, expected) in [
(DupeMode::Strict, EvidenceClass::DerivedFact),
(DupeMode::Mild, EvidenceClass::DerivedFact),
(DupeMode::Weak, EvidenceClass::Heuristic),
(DupeMode::Semantic, EvidenceClass::Heuristic),
] {
let report = analyze_workspace(files.iter(), mode, DEFAULT_MIN_TOKENS, false);
let findings = report.to_findings();
assert!(
!findings.is_empty(),
"{mode:?} should find the fixture duplicate"
);
for finding in &findings {
assert_eq!(finding.evidence_class, expected, "{mode:?}");
}
}
}
#[test]
fn duplicate_behind_an_inactive_cfg_is_still_reported() {
let dir = TempDir::new("dup-cfg-blind");
let file = dir.join("cfg.rs");
std::fs::write(
&file,
r#"
fn dup_active(x: i32) -> i32 {
let mut total = 0;
for i in 0..x {
total += i;
}
total
}
#[cfg(feature = "off")]
fn dup_inactive(x: i32) -> i32 {
let mut total = 0;
for i in 0..x {
total += i;
}
total
}
"#,
)
.unwrap();
let files = authored([file]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
assert_eq!(report.families.len(), 1, "{:?}", report.families);
let names: Vec<_> = report.families[0]
.members
.iter()
.map(|m| m.qualified_name.as_str())
.collect();
assert_eq!(names, ["dup_active", "dup_inactive"]);
}
#[test]
fn semantic_mode_matches_different_macro_call_arguments_with_identical_shape() {
let dir = TempDir::new("dup-macro-blind");
let file = dir.join("macro_calls.rs");
std::fs::write(
&file,
r#"
macro_rules! foo {
($a:expr, $b:expr) => { $a + $b };
}
fn calls_foo_one(a: i32, b: i32) -> i32 {
foo!(a, b) + a + b + a + b + a + b
}
fn calls_foo_two(c: i32, d: i32) -> i32 {
foo!(c, d) + c + d + c + d + c + d
}
"#,
)
.unwrap();
let files = authored([file]);
let mild = analyze_workspace(files.iter(), DupeMode::Mild, 10, false);
assert!(mild.families.is_empty(), "{:?}", mild.families);
let semantic = analyze_workspace(files.iter(), DupeMode::Semantic, 10, false);
assert_eq!(semantic.families.len(), 1, "{:?}", semantic.families);
let names: Vec<_> = semantic.families[0]
.members
.iter()
.map(|m| m.qualified_name.as_str())
.collect();
assert_eq!(names, ["calls_foo_one", "calls_foo_two"]);
}
#[test]
fn named_brace_macro_arguments_are_treated_as_declarative_configuration() {
let dir = TempDir::new("dup-named-macro-arguments");
let file = dir.join("macro_calls.rs");
std::fs::write(
&file,
r#"
macro_rules! candidate {
($($input:tt)*) => { () };
}
fn first() {
candidate! {
evidence: make_evidence(alpha, beta, gamma),
migration: vec![one, two, three],
}
}
fn second() {
candidate! {
evidence: make_evidence(delta, epsilon, zeta),
migration: vec![four, five, six],
}
}
"#,
)
.unwrap();
let files = authored([file]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, 10, false);
assert!(report.families.is_empty(), "{:?}", report.families);
}
#[test]
fn suppression_markers_inside_an_inactive_cfg_block_still_apply() {
let dir = TempDir::new("dup-suppress-cfg");
let file_a = dir.join("a.rs");
let file_b = dir.join("b.rs");
std::fs::write(
&file_a,
r#"
#[cfg(feature = "off")]
// judge-dupe-off: intentional protocol table duplication
fn dup_one(x: i32) -> i32 {
let mut total = 0;
for i in 0..x {
total += i;
}
total
}
// judge-dupe-on
"#,
)
.unwrap();
std::fs::write(
&file_b,
r#"
fn dup_two(x: i32) -> i32 {
let mut total = 0;
for i in 0..x {
total += i;
}
total
}
"#,
)
.unwrap();
let files = authored([file_a, file_b]);
let report = analyze_workspace(files.iter(), DupeMode::Mild, DEFAULT_MIN_TOKENS, false);
assert!(report.families.is_empty(), "{:?}", report.families);
}
#[test]
fn duplication_error_source_preserves_the_underlying_error() {
let err = DuplicationError::Io(PathBuf::from("src/lib.rs"), std::io::Error::other("boom"));
let source = std::error::Error::source(&err).expect("Io must carry a source");
assert!(source.downcast_ref::<std::io::Error>().is_some());
assert_eq!(err.to_string(), "src/lib.rs: failed to read file: boom");
let no_reason = DuplicationError::MissingSuppressionReason(PathBuf::from("a.rs"), 1);
assert!(std::error::Error::source(&no_reason).is_none());
}
}