use std::path::PathBuf;
use crate::collect::ai_attribution::AgenticMode;
use crate::collect::ai_markers::{detect, CommitSignals};
use crate::core::config::{Config, OutputConfig, RepositoryConfig};
use crate::core::db::Database;
use super::aggregator::Aggregator;
use super::formatters::{csv as csv_fmt, json as json_fmt, markdown as md_fmt};
use super::pipeline::ReportPipeline;
fn seed_db() -> Database {
let db = Database::open_in_memory().expect("open db");
let conn = db.connection();
conn.execute(
"INSERT INTO classifications (id, category, subcategory, ticket_id, confidence, method) \
VALUES (1, 'feature', NULL, NULL, 0.9, 'exact_rule')",
[],
)
.expect("insert classification");
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, repository, \
files_changed, insertions, deletions, classification_id, confidence, is_merge) \
VALUES ('aaa111', 'Alice', 'alice@example.com', '2024-01-15T10:00:00+00:00', \
'feat: add login', 'repo-a', 3, 50, 5, 1, 0.9, 0)",
[],
)
.expect("insert commit 1");
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, repository, \
files_changed, insertions, deletions, classification_id, confidence, is_merge) \
VALUES ('bbb222', 'Bob', 'bob@example.com', '2024-01-22T11:00:00+00:00', \
'fix: edge case', 'repo-a', 1, 10, 2, NULL, NULL, 0)",
[],
)
.expect("insert commit 2");
db
}
fn baseline_config() -> Config {
Config {
repositories: vec![RepositoryConfig {
path: PathBuf::from("/tmp/repo-a"),
name: Some("repo-a".into()),
..Default::default()
}],
..Default::default()
}
}
#[test]
fn aggregator_builds_report_data() {
let db = seed_db();
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
assert_eq!(data.total_commits, 2);
assert_eq!(data.total_authors, 2);
assert!(data.period_start.is_some());
assert!(data.period_end.is_some());
assert_eq!(data.repositories.len(), 1);
assert_eq!(data.repositories[0].name, "repo-a");
assert_eq!(data.repositories[0].author_count, 2);
assert_eq!(data.category_breakdown.get("feature").copied(), Some(1));
assert_eq!(data.weekly_activity.len(), 2);
}
#[test]
fn aggregator_handles_empty_db() {
let db = Database::open_in_memory().expect("open db");
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
assert_eq!(data.total_commits, 0);
assert_eq!(data.total_authors, 0);
assert!(data.period_start.is_none());
}
fn tmp_dir(label: &str) -> PathBuf {
let mut path = std::env::temp_dir();
let unique = format!(
"tga-report-{label}-{}-{}",
std::process::id(),
std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map(|d| d.as_nanos())
.unwrap_or(0)
);
path.push(unique);
std::fs::create_dir_all(&path).expect("mkdir");
path
}
#[test]
fn csv_formatter_writes_files_with_headers() {
let db = seed_db();
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
let dir = tmp_dir("csv");
let authors_path = csv_fmt::write_author_csv(&data, &dir).expect("write authors");
let weekly_path = csv_fmt::write_weekly_csv(&data, &dir).expect("write weekly");
let authors_text = std::fs::read_to_string(&authors_path).expect("read");
assert!(authors_text.starts_with("name,email,commit_count"));
assert!(authors_text.contains("Alice"));
assert!(authors_text.contains("Bob"));
let weekly_text = std::fs::read_to_string(&weekly_path).expect("read");
assert!(weekly_text.starts_with("week,author,repository"));
std::fs::remove_dir_all(&dir).ok();
}
#[test]
fn json_formatter_writes_valid_json() {
let db = seed_db();
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
let dir = tmp_dir("json");
let path = json_fmt::write_json(&data, &dir).expect("write json");
let text = std::fs::read_to_string(&path).expect("read");
let parsed: serde_json::Value = serde_json::from_str(&text).expect("valid json");
assert_eq!(parsed["total_commits"], 2);
assert_eq!(parsed["total_authors"], 2);
std::fs::remove_dir_all(&dir).ok();
}
#[test]
fn markdown_formatter_emits_report_header() {
let db = seed_db();
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
let dir = tmp_dir("md");
let path = md_fmt::write_markdown(&data, &dir).expect("write md");
let text = std::fs::read_to_string(&path).expect("read");
assert!(text.contains("# Git Activity Report"));
assert!(text.contains("Alice"));
std::fs::remove_dir_all(&dir).ok();
}
#[test]
fn pipeline_constructs_without_panic() {
let cfg = baseline_config();
let _pipeline = ReportPipeline::new(cfg);
}
#[test]
fn aggregator_computes_summary_and_dora_and_quality() {
let db = seed_db();
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
let summary = data.summary.as_ref().expect("summary present");
assert_eq!(summary.total_commits, 2);
assert_eq!(summary.total_developers, 2);
assert!(summary.total_weeks >= 1);
assert!((summary.classification_coverage_pct - 50.0).abs() < 1e-6);
let dora = data.dora.as_ref().expect("dora present");
let lvl = dora.performance_level.as_str();
assert!(
matches!(lvl, "elite" | "high" | "medium" | "low"),
"unexpected performance_level: {lvl}"
);
let quality = data.quality.as_ref().expect("quality present");
assert!(quality.quality_score >= 0.0 && quality.quality_score <= 1.0);
let velocity = data.velocity.as_ref().expect("velocity present");
assert_eq!(velocity.pr_count, 0);
}
#[test]
fn aggregator_produces_developer_activity_with_score_ordering() {
let db = Database::open_in_memory().expect("open db");
let conn = db.connection();
for i in 0..5 {
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, repository, \
files_changed, insertions, deletions, is_merge) \
VALUES (?1, 'Alice', 'alice@example.com', '2024-01-15T10:00:00+00:00', \
'feat: change', 'repo-a', 1, 10, 1, 0)",
[format!("a{i}")],
)
.expect("seed alice");
}
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, repository, \
files_changed, insertions, deletions, is_merge) \
VALUES ('b1', 'Bob', 'bob@example.com', '2024-01-22T10:00:00+00:00', \
'feat: y', 'repo-a', 1, 1, 1, 0)",
[],
)
.expect("seed bob");
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
let alice = data
.developer_activity
.iter()
.find(|d| d.developer_id == "alice@example.com")
.expect("alice present");
let bob = data
.developer_activity
.iter()
.find(|d| d.developer_id == "bob@example.com")
.expect("bob present");
assert_eq!(alice.total_commits, 5);
assert_eq!(bob.total_commits, 1);
assert!(
alice.activity_score > bob.activity_score,
"alice ({:.4}) should outrank bob ({:.4})",
alice.activity_score,
bob.activity_score
);
}
#[test]
fn csv_formatter_writes_new_report_files() {
let db = seed_db();
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
let dir = tmp_dir("csv-new");
let summary = csv_fmt::write_summary_csv(&data, &dir).expect("write summary");
let weekly_metrics =
csv_fmt::write_weekly_metrics_csv(&data, &dir).expect("write weekly metrics");
let dev_activity =
csv_fmt::write_developer_activity_csv(&data, &dir).expect("write dev activity");
let untracked = csv_fmt::write_untracked_csv(&data, &dir).expect("write untracked");
let weekly_cat =
csv_fmt::write_weekly_categorization_csv(&data, &dir).expect("write weekly categorization");
let weekly_vel =
csv_fmt::write_weekly_velocity_csv(&data, &dir).expect("write weekly velocity");
let dora_csv = csv_fmt::write_weekly_dora_csv(&data, &dir).expect("write dora csv");
for p in [
&summary,
&weekly_metrics,
&dev_activity,
&untracked,
&weekly_cat,
&weekly_vel,
&dora_csv,
] {
assert!(p.exists(), "{} should exist", p.display());
}
let summary_text = std::fs::read_to_string(&summary).expect("read summary");
assert!(summary_text.starts_with("date_range,total_commits"));
let dev_text = std::fs::read_to_string(&dev_activity).expect("read dev activity");
assert!(dev_text.contains("activity_score"));
assert!(dev_text.contains("Alice"));
std::fs::remove_dir_all(&dir).ok();
}
#[test]
fn json_formatter_writes_velocity_quality_dora() {
let db = seed_db();
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
let dir = tmp_dir("json-new");
let velocity = json_fmt::write_velocity_json(&data, &dir).expect("write velocity");
let quality = json_fmt::write_quality_json(&data, &dir).expect("write quality");
let dora = json_fmt::write_dora_json(&data, &dir).expect("write dora");
let velocity_v: serde_json::Value =
serde_json::from_str(&std::fs::read_to_string(&velocity).expect("read"))
.expect("velocity json");
assert!(velocity_v.is_object());
assert!(velocity_v["pr_count"].is_number());
let quality_v: serde_json::Value =
serde_json::from_str(&std::fs::read_to_string(&quality).expect("read"))
.expect("quality json");
assert!(quality_v["quality_score"].as_f64().unwrap() >= 0.0);
let dora_v: serde_json::Value =
serde_json::from_str(&std::fs::read_to_string(&dora).expect("read")).expect("dora json");
assert!(dora_v["performance_level"].is_string());
std::fs::remove_dir_all(&dir).ok();
}
#[test]
fn pipeline_runs_all_formats_when_unspecified() {
let db = seed_db();
let dir = tmp_dir("pipeline");
let cfg = Config {
repositories: vec![RepositoryConfig {
path: PathBuf::from("/tmp/repo-a"),
name: Some("repo-a".into()),
..Default::default()
}],
output: Some(OutputConfig {
directory: Some(dir.clone()),
..Default::default()
}),
..Default::default()
};
let pipeline = ReportPipeline::new(cfg);
let stats = pipeline.run(&db).expect("run");
assert_eq!(stats.total_commits, 2);
assert_eq!(stats.total_authors, 2);
assert_eq!(stats.files_written.len(), 14);
for f in &stats.files_written {
assert!(f.exists(), "{} should exist", f.display());
}
std::fs::remove_dir_all(&dir).ok();
}
fn seed_quality_db() -> Database {
let db = Database::open_in_memory().expect("open db");
let conn = db.connection();
conn.execute(
"INSERT INTO classifications (id, category, subcategory, ticket_id, confidence, method) \
VALUES (2, 'bugfix', NULL, NULL, 0.9, 'exact_rule')",
[],
)
.expect("insert bugfix classification");
let rows = [
("c1", "ENG-1 add feature", 1_i64, None::<i64>),
("c2", "ENG-2 more feature", 1, None),
("c3", "Revert \"ENG-3 bad change\"", 0, None),
("c4", "patch up edge case", 0, Some(2)),
];
for (sha, msg, ticketed, cls) in rows {
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, \
repository, files_changed, insertions, deletions, is_merge, ticketed, \
classification_id) \
VALUES (?1, 'Carol', 'carol@example.com', '2024-01-15T10:00:00+00:00', ?2, \
'repo-a', 1, 5, 1, 0, ?3, ?4)",
rusqlite::params![sha, msg, ticketed, cls],
)
.expect("insert quality commit");
}
db
}
#[test]
fn weekly_activity_carries_quality_columns() {
let db = seed_quality_db();
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
assert_eq!(data.weekly_activity.len(), 1);
let wa = &data.weekly_activity[0];
assert_eq!(wa.commit_count, 4);
assert_eq!(wa.revert_count, 1, "one revert commit");
assert_eq!(wa.bugfix_count, 1, "one classified bugfix");
assert_eq!(wa.ticketed_count, 2, "two ticketed commits");
assert!(
(wa.quality_score - 0.6875).abs() < 1e-9,
"quality_score = {}",
wa.quality_score
);
assert_eq!(wa.quality_tshirt, "4");
}
#[test]
fn weekly_activity_commit_count_net_excludes_reverts() {
let db = seed_quality_db();
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
assert_eq!(data.weekly_activity.len(), 1);
let wa = &data.weekly_activity[0];
assert_eq!(wa.commit_count, 4, "gross commit_count must stay unchanged");
assert_eq!(wa.revert_count, 1);
assert_eq!(
wa.commit_count_net, 3,
"commit_count_net must equal commit_count - revert_count"
);
}
#[test]
fn weekly_activity_splits_ai_count_by_detection_method() {
let db = Database::open_in_memory().expect("open db");
let conn = db.connection();
let rows = [
("m1", 1_i64, Some("trailer")),
("m2", 1, Some("message")),
("m3", 1, Some("email")),
("m4", 0, None),
];
for (sha, is_ai, method) in rows {
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, \
repository, files_changed, insertions, deletions, is_merge, ticketed, \
is_ai_assisted, ai_detection_method) \
VALUES (?1, 'Erin', 'erin@example.com', '2024-01-15T10:00:00+00:00', \
'ENG-7 do work', 'repo-a', 1, 5, 1, 0, 1, ?2, ?3)",
rusqlite::params![sha, is_ai, method],
)
.expect("insert commit");
}
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
assert_eq!(data.weekly_activity.len(), 1);
let wa = &data.weekly_activity[0];
assert_eq!(wa.ai_assisted_count, 3, "the existing total is unchanged");
assert_eq!(wa.ai_trailer_count, 1);
assert_eq!(wa.ai_message_count, 1);
assert_eq!(wa.ai_email_count, 1);
assert_eq!(
wa.ai_trailer_count + wa.ai_message_count + wa.ai_email_count,
wa.ai_assisted_count,
"the split must partition the total on a corpus the current detector \
wrote, or the trailer count is not a floor of anything"
);
}
#[test]
fn weekly_activity_leaves_an_unrecorded_method_unattributed() {
let db = Database::open_in_memory().expect("open db");
let conn = db.connection();
for i in 0..2 {
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, \
repository, files_changed, insertions, deletions, is_merge, ticketed, \
is_ai_assisted, ai_detection_method) \
VALUES (?1, 'Frank', 'frank@example.com', '2024-01-15T10:00:00+00:00', \
'ENG-8 legacy row', 'repo-a', 1, 5, 1, 0, 1, 1, NULL)",
[format!("legacy{i}")],
)
.expect("insert pre-v29 commit");
}
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
let wa = &data.weekly_activity[0];
assert_eq!(wa.ai_assisted_count, 2);
assert_eq!(
(wa.ai_trailer_count, wa.ai_message_count, wa.ai_email_count),
(0, 0, 0),
"an unrecorded method is never assigned to a family"
);
}
#[test]
fn weekly_quality_perfect_when_clean_and_ticketed() {
let db = Database::open_in_memory().expect("open db");
let conn = db.connection();
for i in 0..3 {
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, \
repository, files_changed, insertions, deletions, is_merge, ticketed) \
VALUES (?1, 'Dave', 'dave@example.com', '2024-02-05T10:00:00+00:00', \
'ENG-9 clean work', 'repo-a', 1, 3, 1, 0, 1)",
[format!("d{i}")],
)
.expect("seed clean commit");
}
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
assert_eq!(data.weekly_activity.len(), 1);
let wa = &data.weekly_activity[0];
assert!(
(wa.quality_score - 1.0).abs() < 1e-9,
"{}",
wa.quality_score
);
assert_eq!(wa.quality_tshirt, "5");
assert_eq!(wa.revert_count, 0);
assert_eq!(wa.bugfix_count, 0);
assert_eq!(wa.ticketed_count, 3);
assert_eq!(wa.abandoned_pr_count, 0);
assert_eq!(wa.commit_count_net, wa.commit_count);
}
#[test]
fn aggregator_counts_abandoned_prs() {
let db = Database::open_in_memory().expect("open db");
let conn = db.connection();
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, repository, \
files_changed, insertions, deletions, is_merge, ticketed) \
VALUES ('e1', 'eve', 'eve@example.com', '2024-01-15T10:00:00+00:00', \
'ENG-1 work', 'repo-a', 1, 5, 1, 0, 1)",
[],
)
.expect("seed commit");
conn.execute(
"INSERT INTO pull_requests (pr_number, title, author, state, created_at, merged_at) \
VALUES (10, 'wip', 'eve', 'closed', '2024-01-15T09:00:00+00:00', NULL)",
[],
)
.expect("seed abandoned pr");
conn.execute(
"INSERT INTO pull_requests (pr_number, title, author, state, created_at, merged_at) \
VALUES (11, 'done', 'eve', 'merged', '2024-01-15T08:00:00+00:00', \
'2024-01-15T12:00:00+00:00')",
[],
)
.expect("seed merged pr");
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
assert_eq!(data.weekly_activity.len(), 1);
assert_eq!(
data.weekly_activity[0].abandoned_pr_count, 1,
"exactly one closed-unmerged PR attributed to eve"
);
}
#[test]
fn weekly_csv_carries_the_ai_detection_method_split() {
let db = Database::open_in_memory().expect("open db");
let conn = db.connection();
for (sha, method) in [("s1", "trailer"), ("s2", "trailer"), ("s3", "message")] {
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, \
repository, files_changed, insertions, deletions, is_merge, ticketed, \
is_ai_assisted, ai_detection_method) \
VALUES (?1, 'Gina', 'gina@example.com', '2024-01-15T10:00:00+00:00', \
'ENG-9 work', 'repo-a', 1, 5, 1, 0, 1, 1, ?2)",
rusqlite::params![sha, method],
)
.expect("insert commit");
}
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
let dir = tmp_dir("weekly-ai-method-csv");
let path = csv_fmt::write_weekly_csv(&data, &dir).expect("write weekly");
let text = std::fs::read_to_string(&path).expect("read");
let header: Vec<&str> = text
.lines()
.next()
.expect("header line")
.split(',')
.collect();
let idx = |col: &str| {
header
.iter()
.position(|h| *h == col)
.unwrap_or_else(|| panic!("header missing {col}: {header:?}"))
};
let first_new = idx("ai_trailer_count");
assert_eq!(idx("ai_message_count"), first_new + 1);
assert_eq!(idx("ai_email_count"), first_new + 2);
assert!(
first_new > idx("commit_count_net"),
"the new columns must be appended, not inserted: {header:?}"
);
let row: Vec<&str> = text.lines().nth(1).expect("data row").split(',').collect();
assert_eq!(row[idx("ai_assisted_count")], "3");
assert_eq!(row[first_new], "2", "two trailer-detected commits");
assert_eq!(row[first_new + 1], "1");
assert_eq!(row[first_new + 2], "0");
std::fs::remove_dir_all(&dir).ok();
}
#[test]
fn weekly_csv_includes_quality_columns() {
let db = seed_quality_db();
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
let dir = tmp_dir("weekly-quality-csv");
let path = csv_fmt::write_weekly_csv(&data, &dir).expect("write weekly");
let text = std::fs::read_to_string(&path).expect("read");
let header = text.lines().next().expect("header line");
for col in [
"revert_count",
"bugfix_count",
"ticketed_count",
"quality_score",
"quality_tshirt",
"abandoned_pr_count",
"commit_count_net",
] {
assert!(header.contains(col), "header missing {col}: {header}");
}
assert!(
text.contains(",4,"),
"expected quality_tshirt 4 in row: {text}"
);
std::fs::remove_dir_all(&dir).ok();
}
#[test]
fn json_report_exposes_weekly_quality_fields() {
let db = seed_quality_db();
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
let dir = tmp_dir("json-quality");
let path = json_fmt::write_json(&data, &dir).expect("write json");
let parsed: serde_json::Value =
serde_json::from_str(&std::fs::read_to_string(&path).expect("read")).expect("valid json");
let wa = &parsed["weekly_activity"][0];
assert_eq!(wa["revert_count"], 1);
assert_eq!(wa["bugfix_count"], 1);
assert_eq!(wa["ticketed_count"], 2);
assert_eq!(wa["quality_tshirt"], "4");
assert!(wa["quality_score"].as_f64().expect("score is f64") > 0.68);
assert_eq!(wa["abandoned_pr_count"], 0);
assert_eq!(wa["commit_count"], 4);
assert_eq!(wa["commit_count_net"], 3);
std::fs::remove_dir_all(&dir).ok();
}
fn seed_db_with_authors() -> Database {
let db = Database::open_in_memory().expect("open db");
let conn = db.connection();
conn.execute(
"INSERT INTO authors (id, canonical_name, canonical_email, aliases) \
VALUES (1, 'Alice Smith', 'alice@example.com', '[]')",
[],
)
.expect("insert author alice");
conn.execute(
"INSERT INTO authors (id, canonical_name, canonical_email, aliases) \
VALUES (2, 'Bob Jones', 'bob@example.com', '[]')",
[],
)
.expect("insert author bob");
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, repository, \
files_changed, insertions, deletions, is_merge, author_id) \
VALUES ('aaa111', 'Alice Smith', 'alice@example.com', '2024-01-15T10:00:00+00:00', \
'feat: add login', 'repo-a', 3, 50, 5, 0, 1)",
[],
)
.expect("insert alice commit 1");
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, repository, \
files_changed, insertions, deletions, is_merge, author_id) \
VALUES ('aaa222', 'Alice Smith', 'alice@example.com', '2024-01-16T10:00:00+00:00', \
'feat: add logout', 'repo-a', 2, 20, 3, 0, 1)",
[],
)
.expect("insert alice commit 2");
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, repository, \
files_changed, insertions, deletions, is_merge, author_id) \
VALUES ('bbb111', 'Bob Jones', 'bob@example.com', '2024-01-22T11:00:00+00:00', \
'fix: edge case', 'repo-a', 1, 10, 2, 0, 2)",
[],
)
.expect("insert bob commit");
db
}
#[test]
fn aggregator_author_filter_returns_single_author() {
let db = seed_db_with_authors();
let cfg = baseline_config();
let data =
Aggregator::build_filtered(&db, &cfg, Some("alice@example.com")).expect("build filtered");
assert_eq!(
data.total_commits, 2,
"should contain only alice's 2 commits, got {}",
data.total_commits
);
assert_eq!(
data.total_authors, 1,
"should contain only 1 author (alice), got {}",
data.total_authors
);
assert_eq!(
data.authors[0].email, "alice@example.com",
"the sole author should be alice"
);
}
#[test]
fn aggregator_author_filter_case_insensitive() {
let db = seed_db_with_authors();
let cfg = baseline_config();
let data = Aggregator::build_filtered(&db, &cfg, Some("ALICE@EXAMPLE.COM"))
.expect("case-insensitive filter");
assert_eq!(data.total_commits, 2);
assert_eq!(data.total_authors, 1);
}
#[test]
fn aggregator_author_filter_unknown_email_errors() {
let db = seed_db_with_authors();
let cfg = baseline_config();
let result = Aggregator::build_filtered(&db, &cfg, Some("nobody@example.com"));
assert!(
result.is_err(),
"expected an error for unknown email, got Ok"
);
let msg = result.unwrap_err().to_string();
assert!(
msg.contains("nobody@example.com"),
"error should contain the supplied email; got: {msg}"
);
assert!(
msg.contains("tga aliases list"),
"error should suggest `tga aliases list`; got: {msg}"
);
}
#[test]
fn persist_weekly_quality_upserts_rows_and_is_idempotent() {
let db = seed_quality_db();
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate (first — writes quality rows)");
let wa = &data.weekly_activity[0];
let (score, tshirt, rc, bc, tc, cc): (f64, i64, i64, i64, i64, i64) = db
.connection()
.query_row(
"SELECT quality_score, quality_tshirt, revert_count, bugfix_count, \
ticketed_count, commit_count FROM fact_weekly_quality \
WHERE author_email = 'carol@example.com'",
[],
|r| {
Ok((
r.get(0)?,
r.get(1)?,
r.get(2)?,
r.get(3)?,
r.get(4)?,
r.get(5)?,
))
},
)
.expect("read persisted quality row");
assert!(
(score - wa.quality_score).abs() < 1e-9,
"persisted quality_score {score} must match aggregator score {}",
wa.quality_score
);
assert_eq!(tshirt, wa.quality_tshirt.parse::<i64>().unwrap_or(0));
assert_eq!(rc, wa.revert_count as i64);
assert_eq!(bc, wa.bugfix_count as i64);
assert_eq!(tc, wa.ticketed_count as i64);
assert_eq!(cc, wa.commit_count as i64);
Aggregator::persist_weekly_quality(&db, &data).expect("second persist");
let count: i64 = db
.connection()
.query_row(
"SELECT COUNT(*) FROM fact_weekly_quality WHERE author_email = 'carol@example.com'",
[],
|r| r.get(0),
)
.expect("count");
assert_eq!(count, 1, "UPSERT must not duplicate the grain row");
}
#[test]
fn avg_complexity_is_mean_of_non_null_values() {
let db = Database::open_in_memory().expect("open db");
let conn = db.connection();
conn.execute(
"INSERT INTO classifications (id, category, subcategory, confidence, method, complexity) \
VALUES (10, 'feature', NULL, 0.9, 'llm', 3)",
[],
)
.expect("insert classification with complexity");
conn.execute(
"INSERT INTO classifications (id, category, subcategory, confidence, method, complexity) \
VALUES (11, 'feature', NULL, 0.9, 'llm', 5)",
[],
)
.expect("insert classification complexity=5");
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, \
repository, files_changed, insertions, deletions, is_merge, classification_id) \
VALUES ('e1', 'Eve', 'eve@complexity.example', '2024-01-15T10:00:00+00:00', \
'feat: x', 'repo-c', 1, 5, 1, 0, 10)",
[],
)
.expect("commit with complexity=3");
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, \
repository, files_changed, insertions, deletions, is_merge, classification_id) \
VALUES ('e2', 'Eve', 'eve@complexity.example', '2024-01-16T10:00:00+00:00', \
'feat: y', 'repo-c', 1, 3, 1, 0, 11)",
[],
)
.expect("commit with complexity=5");
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, \
repository, files_changed, insertions, deletions, is_merge) \
VALUES ('e3', 'Eve', 'eve@complexity.example', '2024-01-17T10:00:00+00:00', \
'chore: no complexity', 'repo-c', 1, 1, 1, 0)",
[],
)
.expect("commit without complexity");
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
let wa = data
.weekly_activity
.iter()
.find(|r| r.author == "Eve" || r.author.contains("eve@complexity"))
.expect("Eve's weekly row must exist");
assert_eq!(
wa.avg_complexity,
Some(4.0),
"avg_complexity must be 4.0 (mean of 3 and 5)"
);
}
#[test]
fn avg_complexity_is_none_when_all_null() {
let db = seed_db();
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
for wa in &data.weekly_activity {
assert_eq!(
wa.avg_complexity, None,
"avg_complexity must be None when no classifications carry a complexity score"
);
}
}
#[test]
fn weekly_csv_includes_avg_complexity_column() {
let db = Database::open_in_memory().expect("open db");
let conn = db.connection();
conn.execute(
"INSERT INTO classifications (id, category, subcategory, confidence, method, complexity) \
VALUES (20, 'feature', NULL, 0.9, 'llm', 4)",
[],
)
.expect("insert classification");
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, \
repository, files_changed, insertions, deletions, is_merge, classification_id) \
VALUES ('cx1', 'Frank', 'frank@example.com', '2024-01-15T10:00:00+00:00', \
'feat: z', 'repo-a', 1, 5, 1, 0, 20)",
[],
)
.expect("insert commit");
let cfg = baseline_config();
let data = Aggregator::build(&db, &cfg).expect("aggregate");
let dir = tmp_dir("csv-complexity");
let path = csv_fmt::write_weekly_csv(&data, &dir).expect("write weekly csv");
let text = std::fs::read_to_string(&path).expect("read csv");
let header = text.lines().next().expect("header");
assert!(
header.contains("avg_complexity"),
"weekly CSV header must contain avg_complexity; got: {header}"
);
assert!(
text.contains("4.0000"),
"CSV should contain avg_complexity = 4.0000 for Frank's row; got:\n{text}"
);
std::fs::remove_dir_all(&dir).ok();
}
#[test]
fn aggregator_author_filter_none_returns_all() {
let db = seed_db_with_authors();
let cfg = baseline_config();
let unfiltered = Aggregator::build(&db, &cfg).expect("unfiltered");
let filtered_none = Aggregator::build_filtered(&db, &cfg, None).expect("filtered none");
assert_eq!(unfiltered.total_commits, filtered_none.total_commits);
assert_eq!(unfiltered.total_authors, filtered_none.total_authors);
}
#[test]
fn pipeline_author_filter_single_author() {
let db = seed_db_with_authors();
let dir = tmp_dir("pipeline-author");
let cfg = Config {
repositories: vec![RepositoryConfig {
path: PathBuf::from("/tmp/repo-a"),
name: Some("repo-a".into()),
..Default::default()
}],
output: Some(OutputConfig {
directory: Some(dir.clone()),
..Default::default()
}),
..Default::default()
};
let pipeline = ReportPipeline::new(cfg);
let stats = pipeline
.run_with_filter(&db, Some("alice@example.com"))
.expect("run with filter");
assert_eq!(stats.total_commits, 2, "filtered report: 2 alice commits");
assert_eq!(stats.total_authors, 1, "filtered report: 1 author");
assert_eq!(stats.files_written.len(), 14);
std::fs::remove_dir_all(&dir).ok();
}
#[test]
fn agentic_pct_keeps_unknown_in_the_denominator() {
let db = Database::open_in_memory().expect("open db");
let conn = db.connection();
let rows = [
("a1", "feat: agentic work", "full_agentic"),
("a2", "feat: ide completion", "ide_assisted"),
("a3", "Merge branch 'topic' into main", "unknown"),
("a4", "feat: hand written", "none"),
];
for (sha, msg, mode) in rows {
conn.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, \
repository, files_changed, insertions, deletions, is_merge, ticketed, \
agentic_mode) \
VALUES (?1, 'Dana', 'dana@example.com', '2024-01-15T10:00:00+00:00', ?2, \
'repo-a', 1, 5, 1, 0, 0, ?3)",
rusqlite::params![sha, msg, mode],
)
.expect("insert agentic commit");
}
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
assert_eq!(data.weekly_activity.len(), 1);
let wa = &data.weekly_activity[0];
assert_eq!(wa.commit_count_net, 4, "no reverts, so net is all four");
assert_eq!(wa.agentic_count, 1, "only full_agentic counts as agentic");
assert_eq!(wa.ide_assisted_count, 1, "unknown must not inflate this");
let (net, agentic, ide, pct): (i64, i64, i64, f64) = db
.connection()
.query_row(
"SELECT net_commits, agentic_count, ide_assisted_count, agentic_pct \
FROM fact_weekly_engineer WHERE author_email = 'dana@example.com'",
[],
|r| Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?)),
)
.expect("read persisted engineer row");
assert_eq!(net, 4);
assert_eq!(agentic, 1);
assert_eq!(ide, 1);
assert!(
(pct - 25.0).abs() < 1e-9,
"agentic_pct must be 1/4; 50.0 would bucket unknown as agentic and \
33.3 would drop it from the denominator. got {pct}"
);
}
struct CommitShape {
sha: &'static str,
author_name: &'static str,
author_email: &'static str,
committer_email: &'static str,
message: &'static str,
is_merge: bool,
expect: AgenticMode,
}
fn real_commit_shapes() -> Vec<CommitShape> {
vec\n",
is_merge: false,
expect: AgenticMode::FullAgentic,
},
CommitShape {
sha: "c3",
author_name: "Dana",
author_email: "dana@example.com",
committer_email: "dana@example.com",
message: "chore: tidy the import block\n\n\
Co-authored-by: Copilot <copilot@users.noreply.github.com>\n",
is_merge: false,
expect: AgenticMode::IdeAssisted,
},
CommitShape {
sha: "c4",
author_name: "Dana",
author_email: "dana@example.com",
committer_email: "noreply@github.com",
message: "feat(audit): checkpoint the sweep per repository (#5494)\n\n\
Co-authored-by: Dana Dev <dana@example.com>\n",
is_merge: false,
expect: AgenticMode::Unknown,
},
CommitShape {
sha: "c5",
author_name: "Dana",
author_email: "dana@example.com",
committer_email: "dana@example.com",
message: "Merge branch 'main' into feature/harvest-refs\n",
is_merge: true,
expect: AgenticMode::Unknown,
},
CommitShape {
sha: "c6",
author_name: "Dana",
author_email: "dana@example.com",
committer_email: "dana@example.com",
message: "docs: correct the fact-table column description\n",
is_merge: false,
expect: AgenticMode::None,
},
CommitShape {
sha: "c7",
author_name: "Dana",
author_email: "dana@example.com",
committer_email: "dana@example.com",
message: "Revert \"feat(tga): harvest branch and PR refs\"\n\n\
This reverts commit c1.\n",
is_merge: false,
expect: AgenticMode::None,
},
CommitShape {
sha: "c8",
author_name: "devin-ai-integration[bot]",
author_email: "devin-ai-integration[bot]@users.noreply.github.com",
committer_email: "devin-ai-integration[bot]@users.noreply.github.com",
message: "chore(deps): bump the lockfile\n",
is_merge: false,
expect: AgenticMode::FullAgentic,
},
]
}
#[test]
fn persist_weekly_engineer_upserts_rows() {
let db = Database::open_in_memory().expect("open db");
let shapes = real_commit_shapes();
for (i, c) in shapes.iter().enumerate() {
let detection = detect(&CommitSignals {
message: c.message,
author_email: c.author_email,
committer_email: c.committer_email,
});
assert_eq!(
detection.mode, c.expect,
"detection drifted for {}: {:?}",
c.sha, c.message
);
db.connection()
.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, \
repository, files_changed, insertions, deletions, is_merge, ticketed, \
agentic_mode) \
VALUES (?1, ?2, ?3, ?4, ?5, 'repo-a', 2, 20, 4, ?6, 0, ?7)",
rusqlite::params![
c.sha,
c.author_name,
c.author_email,
format!("2024-01-1{}T09:00:00+00:00", 5 + i % 5),
c.message,
c.is_merge as i64,
detection.mode.as_str(),
],
)
.expect("insert fixture commit");
}
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
let dana = data
.weekly_activity
.iter()
.find(|w| w.author.contains("Dana"))
.expect("Dana has a weekly bucket");
assert_eq!(dana.commit_count_net, 6, "7 commits less 1 revert");
assert_eq!(dana.agentic_count, 2, "c1 and c2 only");
assert_eq!(dana.ide_assisted_count, 1, "c3 only");
let row = |email: &str| -> (i64, i64, i64, f64) {
db.connection()
.query_row(
"SELECT net_commits, agentic_count, ide_assisted_count, agentic_pct \
FROM fact_weekly_engineer WHERE author_email = ?1",
[email],
|r| Ok((r.get(0)?, r.get(1)?, r.get(2)?, r.get(3)?)),
)
.expect("read persisted engineer row")
};
let (net, agentic, ide, pct) = row("dana@example.com");
assert_eq!((net, agentic, ide), (6, 2, 1));
assert!(
(pct - 200.0 / 6.0).abs() < 1e-9,
"agentic_pct must be 2/6; 42.9 would forget the revert, 28.6 would keep \
it in the denominator, 50.0 would count ide_assisted. got {pct}"
);
let (bot_net, bot_agentic, bot_ide, bot_pct) = row(shapes[7].author_email);
assert_eq!(
(bot_net, bot_agentic, bot_ide),
(1, 1, 0),
"the bot commit is its own grain key, not folded into Dana's"
);
assert!((bot_pct - 100.0).abs() < 1e-9, "got {bot_pct}");
let written = Aggregator::persist_weekly_engineer(&db, &data).expect("second persist");
assert_eq!(written, 2, "one row per author-week");
let total: i64 = db
.connection()
.query_row("SELECT COUNT(*) FROM fact_weekly_engineer", [], |r| {
r.get(0)
})
.expect("count rows");
assert_eq!(total, 2, "re-persisting must not duplicate the grain key");
assert_eq!(row("dana@example.com"), (6, 2, 1, pct));
}
fn seed_deployment_full(
db: &Database,
deploy_id: &str,
triggered_at: &str,
environment: &str,
status: &str,
git_sha: Option<&str>,
) {
db.connection()
.execute(
"INSERT INTO fact_deployments \
(deploy_id, repo, environment, triggered_at, status, git_sha) \
VALUES (?1, 'repo-a', ?2, ?3, ?4, ?5)",
rusqlite::params![deploy_id, environment, triggered_at, status, git_sha],
)
.expect("insert fact_deployments row");
}
fn seed_deployment(db: &Database, deploy_id: &str, triggered_at: &str) {
seed_deployment_full(db, deploy_id, triggered_at, "production", "success", None);
}
#[test]
fn dora_reads_fact_deployments_when_populated() {
let db = seed_db();
seed_deployment_full(
&db,
"deploy-linked",
"2024-01-15T12:00:00+00:00",
"production",
"success",
Some("aaa111"),
);
for i in 0..69 {
seed_deployment(&db, &format!("deploy-{i}"), "2024-01-18T00:00:00+00:00");
}
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
let dora = data.dora.as_ref().expect("dora present");
assert!(
dora.deployment_frequency > 0.0,
"70 production deploys in-period must not read as zero"
);
assert_eq!(dora.deployment_frequency_source, "fact_deployments");
assert_eq!(dora.lead_time_source, "measured");
assert_eq!(dora.lead_time_hours, Some(2.0));
assert_ne!(
dora.performance_level, "low",
"70 deploys / 2 weeks with a measured 2h lead must not classify the \
way the zero-deploy proxy would"
);
}
#[test]
fn dora_falls_back_to_pr_proxy_when_fact_deployments_empty() {
let db = seed_db();
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
let dora = data.dora.as_ref().expect("dora present");
assert_eq!(dora.deployment_frequency, 0.0);
assert_eq!(dora.deployment_frequency_source, "pr_merge_proxy");
assert_eq!(dora.lead_time_hours, None);
assert_eq!(dora.lead_time_source, "unmeasurable");
}
#[test]
fn dora_lead_time_is_unmeasurable_when_merged_prs_are_all_outlier_filtered() {
let db = seed_db();
db.connection()
.execute(
"INSERT INTO pull_requests \
(pr_number, title, author, state, created_at, merged_at) \
VALUES (1, 'pr', 'alice', 'merged', '2024-01-01T00:00:00+00:00', \
'2024-02-10T00:00:00+00:00')",
[],
)
.expect("insert outlier-filtered merged pr");
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
let dora = data.dora.as_ref().expect("dora present");
assert_eq!(dora.lead_time_hours, None);
assert_eq!(dora.lead_time_source, "unmeasurable");
}
#[test]
fn dora_ignores_fact_deployments_rows_outside_the_period() {
let db = seed_db();
for i in 0..5 {
seed_deployment(&db, &format!("in-period-{i}"), "2024-01-18T00:00:00+00:00");
}
for i in 0..3 {
seed_deployment(
&db,
&format!("out-of-period-{i}"),
"2023-06-01T00:00:00+00:00",
);
}
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
let dora = data.dora.as_ref().expect("dora present");
assert!(
(dora.deployment_frequency - 2.5).abs() < 1e-9,
"expected 5 in-period deploys / 2 weeks = 2.5, got {}",
dora.deployment_frequency
);
assert_eq!(dora.deployment_frequency_source, "fact_deployments");
}
#[test]
fn dora_falls_back_to_pr_proxy_when_all_fact_deployments_rows_are_outside_period() {
let db = seed_db();
for i in 0..4 {
seed_deployment(&db, &format!("out-{i}"), "2023-06-01T00:00:00+00:00");
}
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
let dora = data.dora.as_ref().expect("dora present");
assert_eq!(dora.deployment_frequency, 0.0);
assert_eq!(dora.deployment_frequency_source, "pr_merge_proxy");
}
#[test]
fn dora_excludes_non_production_or_failed_deployment_rows() {
let db = seed_db();
seed_deployment_full(
&db,
"staging-1",
"2024-01-18T00:00:00+00:00",
"staging",
"success",
None,
);
seed_deployment_full(
&db,
"failed-1",
"2024-01-18T01:00:00+00:00",
"production",
"failure",
None,
);
seed_deployment(&db, "prod-success-1", "2024-01-18T02:00:00+00:00");
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
let dora = data.dora.as_ref().expect("dora present");
assert_eq!(dora.deployment_frequency_source, "fact_deployments");
assert!(
(dora.deployment_frequency - 0.5).abs() < 1e-9,
"only the production/success row should count: 1 deploy / 2 weeks = 0.5, got {}",
dora.deployment_frequency
);
}
#[test]
fn dora_lead_time_falls_back_to_proxy_when_deploy_git_sha_is_unmatched() {
let db = seed_db();
seed_deployment_full(
&db,
"deploy-unmatched",
"2024-01-18T00:00:00+00:00",
"production",
"success",
Some("deadbeef"),
);
db.connection()
.execute(
"INSERT INTO pull_requests \
(pr_number, title, author, state, created_at, merged_at) \
VALUES (1, 'pr', 'alice', 'merged', '2024-01-15T08:00:00+00:00', \
'2024-01-15T10:00:00+00:00')",
[],
)
.expect("insert merged pr");
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
let dora = data.dora.as_ref().expect("dora present");
assert_eq!(dora.deployment_frequency_source, "fact_deployments");
assert_eq!(dora.lead_time_source, "proxy");
assert_eq!(dora.lead_time_hours, Some(2.0));
}
#[test]
fn dora_lead_time_is_unmeasurable_when_no_deploy_link_and_no_prs() {
let db = seed_db();
seed_deployment_full(
&db,
"deploy-unmatched",
"2024-01-18T00:00:00+00:00",
"production",
"success",
Some("deadbeef"),
);
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
let dora = data.dora.as_ref().expect("dora present");
assert_eq!(dora.deployment_frequency_source, "fact_deployments");
assert!(dora.deployment_frequency > 0.0);
assert_eq!(dora.lead_time_hours, None);
assert_eq!(dora.lead_time_source, "unmeasurable");
}
#[test]
fn dora_marks_proxy_source_distinctly_when_fact_deployments_query_fails() {
let db = seed_db();
db.connection()
.execute("DROP TABLE fact_deployments", [])
.expect("drop fact_deployments to simulate a missing/corrupt table");
let data = Aggregator::build(&db, &baseline_config()).expect("aggregate");
let dora = data.dora.as_ref().expect("dora present");
assert_eq!(dora.deployment_frequency, 0.0);
assert_eq!(
dora.deployment_frequency_source,
"pr_merge_proxy_query_failed"
);
}