use std::collections::{BTreeMap, HashMap};
use rusqlite::{params, Connection};
use serde::Serialize;
use crate::collect::identity::suggest::{detect_from_authors, Suggestion, HIGH_CONFIDENCE_CUTOFF};
use crate::core::errors::Result;
pub const AUTHORSHIP_SCHEMA_VERSION: &str = "v0";
#[derive(Debug, Clone, Default, PartialEq, Serialize)]
#[non_exhaustive]
pub struct AuthorshipSummary {
pub schema_version: String,
pub repository: String,
pub distinct_authors: u64,
pub bus_factor: u64,
pub top_author_share_pct: f64,
pub single_author_subsystems: Vec<String>,
pub monthly_trajectory: Vec<MonthlyActivity>,
pub unresolved_authors: u64,
pub identity_merge_risk: Option<IdentityMergeRisk>,
pub caveats: Vec<String>,
}
#[derive(Debug, Clone, Default, PartialEq, Serialize)]
#[non_exhaustive]
pub struct IdentityMergeRisk {
pub suggested_unmerged: u64,
pub affected_metrics: Vec<String>,
pub resolve_command: String,
}
const TOP_N_AUTHORS: usize = 10;
const RISK_AFFECTED_METRICS: &[&str] = &["bus_factor", "top_author_share_pct"];
const RISK_RESOLVE_COMMAND: &str = "tga aliases suggest";
#[derive(Debug, Clone, Default, PartialEq, Serialize)]
pub struct MonthlyActivity {
pub month: String,
pub active_authors: u64,
pub commits: u64,
}
const CAVEATS: &[&str] = &[
"Squash-merge attribution: a GitHub squash-merge preserves the PR author, but a local \
`git merge --squash` by a human does not — this run cannot distinguish the two.",
"Identity aliases are merged only as far as the collection pass resolved them: authors \
are grouped by `authors.canonical_email` where the resolver linked the commit, and by the \
raw commit email where it did not.",
"No vendored-path exclusion: a checked-in vendor/dependency directory can make its \
committer look like the sole owner of thousands of paths.",
];
fn unresolved_caveat(count: u64) -> String {
format!(
"{count} commit identity/identities in this repository were never linked to a resolved \
author, so their aliases stay split — concentration and bus factor read LOWER than \
reality by that much."
)
}
fn suggested_merge_caveat(risk: &IdentityMergeRisk) -> String {
format!(
"{count} identity/identities are suggested for merge but not confirmed, and they touch \
the authors behind {metrics} — both read LOWER than reality until the merges are \
accepted. Run `{command}` to review them; nothing is merged automatically.",
count = risk.suggested_unmerged,
metrics = risk.affected_metrics.join(" and "),
command = risk.resolve_command,
)
}
fn confirmed_alias_map(conn: &Connection) -> Result<HashMap<String, String>> {
let mut stmt = conn.prepare("SELECT canonical_email, aliases FROM authors")?;
let rows = stmt.query_map([], |row| {
Ok((
row.get::<_, String>(0)?,
row.get::<_, Option<String>>(1)?.unwrap_or_default(),
))
})?;
let mut map: HashMap<String, String> = HashMap::new();
for row in rows {
let (canonical, aliases_json) = row?;
if canonical.is_empty() {
continue;
}
let aliases: Vec<String> = match serde_json::from_str(&aliases_json) {
Ok(aliases) => aliases,
Err(e) => {
if !aliases_json.trim().is_empty() {
tracing::warn!(
author = %canonical,
error = %e,
"authors.aliases is not a JSON array of strings; this author's confirmed \
merges are not applied to the authorship figures"
);
}
Vec::new()
}
};
for alias in aliases {
let key = alias.to_lowercase();
if key.is_empty() || key == canonical.to_lowercase() {
continue;
}
map.insert(key, canonical.clone());
}
}
Ok(map)
}
pub fn merge_suggestions(
conn: &Connection,
canonical_domain: Option<&str>,
) -> Result<Vec<Suggestion>> {
detect_from_authors(conn, canonical_domain, HIGH_CONFIDENCE_CUTOFF)
}
fn identity_merge_risk(
suggestions: &[Suggestion],
alias_map: &HashMap<String, String>,
ranked: &[(&String, &u64)],
) -> Option<IdentityMergeRisk> {
let top: std::collections::BTreeSet<String> = ranked
.iter()
.take(TOP_N_AUTHORS)
.map(|(email, _)| email.to_lowercase())
.collect();
if top.is_empty() {
return None;
}
let touching: std::collections::BTreeSet<String> = suggestions
.iter()
.filter(|s| !alias_map.contains_key(&s.src.to_lowercase()))
.filter(|s| top.contains(&s.src.to_lowercase()) || top.contains(&s.dst.to_lowercase()))
.map(|s| s.src.to_lowercase())
.collect();
if touching.is_empty() {
return None;
}
Some(IdentityMergeRisk {
suggested_unmerged: touching.len() as u64,
affected_metrics: RISK_AFFECTED_METRICS
.iter()
.map(|s| s.to_string())
.collect(),
resolve_command: RISK_RESOLVE_COMMAND.to_string(),
})
}
const BOT_MARKERS: &[&str] = &[
"[bot]",
"dependabot",
"renovate",
"github-actions",
"gitlab-ci",
"greenkeeper",
];
fn is_bot(name: &str, email: &str) -> bool {
let name = name.to_ascii_lowercase();
let email = email.to_ascii_lowercase();
BOT_MARKERS
.iter()
.any(|m| name.contains(m) || email.contains(m))
}
fn subsystem_of(path: &str) -> String {
path.split('/').next().unwrap_or(path).to_string()
}
fn month_of(timestamp: &str) -> Option<String> {
timestamp.get(0..7).map(str::to_string)
}
pub fn build_authorship_summary(conn: &Connection, repository: &str) -> Result<AuthorshipSummary> {
let suggestions = merge_suggestions(conn, None)?;
build_authorship_summary_with(conn, repository, &suggestions)
}
pub fn build_authorship_summary_with(
conn: &Connection,
repository: &str,
suggestions: &[Suggestion],
) -> Result<AuthorshipSummary> {
let alias_map = confirmed_alias_map(conn)?;
let mut stmt = conn.prepare(
"SELECT c.id, c.author_name, c.author_email, a.canonical_email, c.timestamp, f.path \
FROM commits c \
JOIN files f ON f.commit_id = c.id \
LEFT JOIN authors a ON a.id = c.author_id \
WHERE c.repository = ?1 AND c.is_merge = 0",
)?;
let rows = stmt.query_map(params![repository], |row| {
Ok((
row.get::<_, i64>(0)?,
row.get::<_, String>(1)?,
row.get::<_, String>(2)?,
row.get::<_, Option<String>>(3)?,
row.get::<_, String>(4)?,
row.get::<_, String>(5)?,
))
})?;
let mut touches_by_author: BTreeMap<String, u64> = BTreeMap::new();
let mut authors_by_subsystem: BTreeMap<String, std::collections::BTreeSet<String>> =
BTreeMap::new();
let mut months: BTreeMap<String, std::collections::BTreeSet<String>> = BTreeMap::new();
let mut month_commits: BTreeMap<String, std::collections::BTreeSet<i64>> = BTreeMap::new();
let mut unresolved: std::collections::BTreeSet<String> = std::collections::BTreeSet::new();
for row in rows {
let (commit_id, name, email, canonical_email, timestamp, path) = row?;
if is_bot(&name, &email) {
continue;
}
let raw_key = if email.is_empty() {
name.clone()
} else {
email.clone()
};
let resolved = canonical_email.filter(|e| !e.is_empty());
let author_key = match resolved {
Some(canonical) => match alias_map.get(&canonical.to_lowercase()) {
Some(merged) => merged.clone(),
None => canonical,
},
None => match alias_map.get(&raw_key.to_lowercase()) {
Some(canonical) => canonical.clone(),
None => {
unresolved.insert(raw_key.clone());
raw_key
}
},
};
*touches_by_author.entry(author_key.clone()).or_insert(0) += 1;
authors_by_subsystem
.entry(subsystem_of(&path))
.or_default()
.insert(author_key.clone());
if let Some(month) = month_of(×tamp) {
months.entry(month.clone()).or_default().insert(author_key);
month_commits.entry(month).or_default().insert(commit_id);
}
}
let total_touches: u64 = touches_by_author.values().sum();
let mut ranked: Vec<(&String, &u64)> = touches_by_author.iter().collect();
ranked.sort_by_key(|(_, n)| std::cmp::Reverse(**n));
let top_author_share_pct = match ranked.first() {
Some((_, n)) if total_touches > 0 => (**n as f64 / total_touches as f64) * 100.0,
_ => 0.0,
};
let bus_factor = bus_factor_of(&ranked, total_touches);
let mut single_author_subsystems: Vec<String> = authors_by_subsystem
.into_iter()
.filter(|(_, authors)| authors.len() == 1)
.map(|(subsystem, _)| subsystem)
.collect();
single_author_subsystems.sort();
let mut month_keys: Vec<String> = months.keys().cloned().collect();
month_keys.sort();
let recent: Vec<String> = month_keys.into_iter().rev().take(12).rev().collect();
let monthly_trajectory = recent
.into_iter()
.map(|month| MonthlyActivity {
active_authors: months.get(&month).map(|s| s.len() as u64).unwrap_or(0),
commits: month_commits
.get(&month)
.map(|s| s.len() as u64)
.unwrap_or(0),
month,
})
.collect();
let unresolved_authors = unresolved.len() as u64;
let mut caveats: Vec<String> = CAVEATS.iter().map(|s| s.to_string()).collect();
if unresolved_authors > 0 {
caveats.push(unresolved_caveat(unresolved_authors));
}
let merge_risk = identity_merge_risk(suggestions, &alias_map, &ranked);
if let Some(risk) = &merge_risk {
caveats.push(suggested_merge_caveat(risk));
}
Ok(AuthorshipSummary {
schema_version: AUTHORSHIP_SCHEMA_VERSION.to_string(),
repository: repository.to_string(),
distinct_authors: touches_by_author.len() as u64,
bus_factor,
top_author_share_pct,
single_author_subsystems,
monthly_trajectory,
unresolved_authors,
identity_merge_risk: merge_risk,
caveats,
})
}
pub fn repository_has_commits(conn: &Connection, repository: &str) -> Result<bool> {
let mut stmt = conn.prepare("SELECT 1 FROM commits WHERE repository = ?1 LIMIT 1")?;
Ok(stmt.exists(params![repository])?)
}
pub fn recorded_repository_names(conn: &Connection) -> Result<Vec<String>> {
let mut stmt = conn.prepare("SELECT DISTINCT repository FROM commits ORDER BY repository")?;
let rows = stmt.query_map([], |row| row.get::<_, String>(0))?;
let mut names = Vec::new();
for row in rows {
names.push(row?);
}
Ok(names)
}
fn bus_factor_of(ranked: &[(&String, &u64)], total: u64) -> u64 {
if total == 0 {
return 0;
}
let half = total as f64 / 2.0;
let mut running = 0u64;
for (i, (_, n)) in ranked.iter().enumerate() {
running += **n;
if running as f64 >= half {
return (i + 1) as u64;
}
}
ranked.len() as u64
}
impl AuthorshipSummary {
pub fn to_json(&self) -> Result<String> {
Ok(serde_json::to_string_pretty(self)?)
}
}
#[cfg(test)]
#[path = "authorship_tests.rs"]
mod authorship_tests;