use crate::git::GitContext;
use crate::language::Language;
use crate::report::{FunctionRiskReport, MetricsReport};
use crate::risk::RiskBand;
use anyhow::{Context, Result};
use rayon::prelude::*;
use serde::{Deserialize, Serialize};
use std::collections::HashMap;
use std::path::{Path, PathBuf};
#[cfg(test)]
use crate::report::RiskReport;
#[derive(Debug, Clone, PartialEq)]
pub enum TouchMode {
File,
PerFunction,
Hybrid { threshold: usize },
}
pub const SNAPSHOT_SCHEMA_VERSION: u32 = 2;
const SNAPSHOT_SCHEMA_MIN_VERSION: u32 = 1;
const INDEX_SCHEMA_VERSION: u32 = 1;
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
#[serde(rename_all = "snake_case")]
pub struct CommitInfo {
pub sha: String,
pub parents: Vec<String>,
pub timestamp: i64,
#[serde(skip_serializing_if = "Option::is_none")]
pub branch: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub message: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub author: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub is_fix_commit: Option<bool>,
#[serde(skip_serializing_if = "Option::is_none")]
pub is_revert_commit: Option<bool>,
#[serde(skip_serializing_if = "Vec::is_empty", default)]
pub ticket_ids: Vec<String>,
}
impl From<GitContext> for CommitInfo {
fn from(ctx: GitContext) -> Self {
CommitInfo {
sha: ctx.head_sha,
parents: ctx.parent_shas,
timestamp: ctx.timestamp,
branch: ctx.branch,
message: ctx.message,
author: ctx.author,
is_fix_commit: ctx.is_fix_commit,
is_revert_commit: ctx.is_revert_commit,
ticket_ids: ctx.ticket_ids,
}
}
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
#[serde(rename_all = "snake_case")]
pub struct AnalysisInfo {
pub scope: String,
#[serde(rename = "tool_version")]
pub tool_version: String,
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
#[serde(rename_all = "snake_case")]
pub struct ChurnMetrics {
pub lines_added: usize,
pub lines_deleted: usize,
pub net_change: i64,
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
#[serde(rename_all = "snake_case")]
pub struct PercentileFlags {
pub is_top_10_pct: bool,
pub is_top_5_pct: bool,
pub is_top_1_pct: bool,
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
#[serde(rename_all = "snake_case")]
pub struct CallGraphMetrics {
pub fan_in: usize,
pub fan_out: usize,
pub pagerank: f64,
pub betweenness: f64,
pub scc_id: usize,
pub scc_size: usize,
#[serde(default)]
pub is_entrypoint: bool,
#[serde(skip_serializing_if = "Option::is_none")]
pub dependency_depth: Option<usize>,
#[serde(skip_serializing_if = "Option::is_none")]
pub neighbor_churn: Option<usize>,
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
#[serde(rename_all = "snake_case")]
pub struct FunctionSnapshot {
pub function_id: String,
pub file: String,
pub line: u32,
pub language: Language,
pub metrics: MetricsReport,
pub lrs: f64,
pub band: RiskBand,
#[serde(skip_serializing_if = "Option::is_none")]
pub suppression_reason: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub churn: Option<ChurnMetrics>,
#[serde(skip_serializing_if = "Option::is_none")]
pub touch_count_30d: Option<usize>,
#[serde(skip_serializing_if = "Option::is_none")]
pub days_since_last_change: Option<u32>,
#[serde(skip_serializing_if = "Option::is_none")]
pub callgraph: Option<CallGraphMetrics>,
#[serde(skip_serializing_if = "Option::is_none")]
pub activity_risk: Option<f64>,
#[serde(skip_serializing_if = "Option::is_none")]
pub risk_factors: Option<crate::scoring::RiskFactors>,
#[serde(skip_serializing_if = "Option::is_none")]
pub percentile: Option<PercentileFlags>,
#[serde(skip_serializing_if = "Option::is_none")]
pub driver: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub driver_detail: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub quadrant: Option<String>,
#[serde(skip_serializing_if = "Vec::is_empty", default)]
pub patterns: Vec<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub pattern_details: Option<Vec<crate::patterns::PatternDetail>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub subsystem: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub authors_90d: Option<u32>,
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
#[serde(rename_all = "snake_case")]
pub struct BandStats {
pub count: usize,
pub sum_risk: f64,
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
#[serde(rename_all = "snake_case")]
pub struct CallGraphStats {
pub total_edges: usize,
pub avg_fan_in: f64,
pub scc_count: usize,
pub largest_scc_size: usize,
#[serde(default)]
pub betweenness_approximate: bool,
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
#[serde(rename_all = "snake_case")]
pub struct SnapshotSummary {
pub total_functions: usize,
pub total_activity_risk: f64,
pub top_1_pct_share: f64,
pub top_5_pct_share: f64,
pub top_10_pct_share: f64,
pub by_band: std::collections::BTreeMap<String, BandStats>,
#[serde(skip_serializing_if = "Option::is_none")]
pub call_graph: Option<CallGraphStats>,
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
#[serde(rename_all = "snake_case")]
pub struct Snapshot {
#[serde(rename = "schema_version")]
pub schema_version: u32,
pub commit: CommitInfo,
pub analysis: AnalysisInfo,
pub functions: Vec<FunctionSnapshot>,
#[serde(skip_serializing_if = "Option::is_none")]
pub summary: Option<SnapshotSummary>,
#[serde(skip_serializing_if = "Option::is_none")]
pub aggregates: Option<crate::aggregates::SnapshotAggregates>,
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
#[serde(rename_all = "snake_case")]
pub struct IndexEntry {
pub sha: String,
pub parents: Vec<String>,
pub timestamp: i64,
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
#[serde(rename_all = "snake_case")]
pub struct Index {
#[serde(rename = "schema_version")]
pub schema_version: u32,
#[serde(skip_serializing_if = "Option::is_none")]
pub compaction_level: Option<u32>,
pub commits: Vec<IndexEntry>,
}
const SUBSYSTEM_MANIFESTS: &[&str] = &[
"package.json",
"Cargo.toml",
"pyproject.toml",
"go.mod",
"setup.py",
"setup.cfg",
"Gemfile",
"build.gradle",
"pom.xml",
"mix.exs",
];
const SKIP_DIRS: &[&str] = &[
"node_modules",
".git",
"__pycache__",
".venv",
"venv",
"dist",
"build",
"out",
"target",
];
fn build_subsystem_cache(repo_root: &Path) -> HashMap<PathBuf, String> {
let mut cache: HashMap<PathBuf, String> = HashMap::new();
collect_manifest_dirs(repo_root, repo_root, &mut cache);
cache
}
fn collect_manifest_dirs(dir: &Path, repo_root: &Path, cache: &mut HashMap<PathBuf, String>) {
for manifest in SUBSYSTEM_MANIFESTS {
if dir.join(manifest).exists() {
let rel = dir
.strip_prefix(repo_root)
.unwrap_or(dir)
.to_string_lossy()
.replace('\\', "/");
cache.entry(dir.to_path_buf()).or_insert(rel);
break;
}
}
let entries = match std::fs::read_dir(dir) {
Ok(e) => e,
Err(_) => return,
};
for entry in entries.flatten() {
let ft = match entry.file_type() {
Ok(ft) => ft,
Err(_) => continue,
};
if !ft.is_dir() {
continue;
}
let name = entry.file_name();
let name_str = name.to_string_lossy();
if name_str.starts_with('.') || SKIP_DIRS.contains(&name_str.as_ref()) {
continue;
}
collect_manifest_dirs(&entry.path(), repo_root, cache);
}
}
fn subsystem_for_file(
abs_file: &Path,
repo_root: &Path,
cache: &HashMap<PathBuf, String>,
) -> Option<String> {
let mut best: Option<&str> = None;
let mut best_depth = 0usize;
let mut dir = abs_file.parent()?;
loop {
if let Some(label) = cache.get(dir) {
let depth = dir.components().count();
if depth >= best_depth {
best = Some(label.as_str());
best_depth = depth;
}
}
match dir.parent() {
Some(p) if p != dir && dir.starts_with(repo_root) => dir = p,
_ => break,
}
}
best.map(|s| s.to_string())
}
impl Snapshot {
pub fn new(git_context: GitContext, reports: Vec<FunctionRiskReport>) -> Self {
let mut functions: Vec<FunctionSnapshot> = reports
.into_iter()
.map(|report| {
let normalized_file = report.file.replace('\\', "/");
let function_symbol = if report.function.starts_with("<anonymous>") {
"<anonymous>"
} else {
&report.function
};
let function_id = format!("{}::{}", normalized_file, function_symbol);
FunctionSnapshot {
function_id,
file: normalized_file,
line: report.line,
language: report.language,
metrics: report.metrics,
lrs: report.lrs,
band: report.band,
suppression_reason: report.suppression_reason,
churn: None, touch_count_30d: None, days_since_last_change: None, callgraph: None, activity_risk: None,
risk_factors: None,
percentile: None,
driver: None,
driver_detail: None,
quadrant: None,
patterns: report.patterns,
pattern_details: None,
subsystem: None,
authors_90d: None,
}
})
.collect();
functions.sort_by(|a, b| a.function_id.cmp(&b.function_id));
Snapshot {
schema_version: SNAPSHOT_SCHEMA_VERSION,
commit: CommitInfo::from(git_context),
analysis: AnalysisInfo {
scope: "full".to_string(),
tool_version: env!("CARGO_PKG_VERSION").to_string(),
},
functions,
summary: None,
aggregates: None, }
}
pub fn populate_churn(
&mut self,
file_churns: &std::collections::HashMap<String, crate::git::FileChurn>,
) {
for function in &mut self.functions {
let file_path = &function.file;
if let Some(file_churn) = file_churns.get(file_path) {
let net_change = file_churn.lines_added as i64 - file_churn.lines_deleted as i64;
function.churn = Some(ChurnMetrics {
lines_added: file_churn.lines_added,
lines_deleted: file_churn.lines_deleted,
net_change,
});
}
}
}
pub fn populate_authors_90d(&mut self, repo_root: &Path) {
use std::collections::{HashMap, HashSet};
use std::process::Command;
let mut file_indices: HashMap<String, Vec<usize>> = HashMap::new();
for (i, func) in self.functions.iter().enumerate() {
file_indices.entry(func.file.clone()).or_default().push(i);
}
for (abs_path, indices) in &file_indices {
let rel = if let Ok(r) = std::path::Path::new(abs_path).strip_prefix(repo_root) {
r.to_string_lossy().to_string()
} else {
abs_path.clone()
};
let out = Command::new("git")
.args(["log", "--after=90.days.ago", "--format=%ae", "--", &rel])
.current_dir(repo_root)
.output();
let count = match out {
Ok(o) if o.status.success() => {
let text = String::from_utf8_lossy(&o.stdout);
text.lines()
.filter(|l| !l.is_empty())
.collect::<HashSet<_>>()
.len() as u32
}
_ => continue, };
for &i in indices {
self.functions[i].authors_90d = Some(count);
}
}
}
pub fn populate_subsystems(&mut self, repo_root: &Path) {
let cache = build_subsystem_cache(repo_root);
for function in &mut self.functions {
let path = Path::new(&function.file);
let abs = if path.is_absolute() {
path.to_path_buf()
} else {
repo_root.join(path)
};
function.subsystem = subsystem_for_file(&abs, repo_root, &cache);
}
}
fn populate_per_function_touch_metrics(
&mut self,
repo_root: &std::path::Path,
progress_fn: Option<&dyn Fn(usize, usize)>,
) -> anyhow::Result<()> {
let all: Vec<usize> = (0..self.functions.len()).collect();
self.populate_per_function_touch_for_indices(repo_root, &all, progress_fn)
}
fn populate_per_function_touch_for_indices(
&mut self,
repo_root: &std::path::Path,
indices: &[usize],
progress_fn: Option<&dyn Fn(usize, usize)>,
) -> anyhow::Result<()> {
let sha = self.commit.sha.clone();
let timestamp = self.commit.timestamp;
let mut cache = crate::touch_cache::read_touch_cache(repo_root).unwrap_or_default();
let total = indices.len();
const COLD_CACHE_WARN_THRESHOLD: usize = 50;
let cache_warm = cache.keys().any(|k| k.starts_with(&sha));
if !cache_warm && total >= COLD_CACHE_WARN_THRESHOLD {
let threads = rayon::current_num_threads().max(1);
let est_secs = (total * 9).div_ceil(1000 * threads);
eprintln!("touch cache: cold start for {total} functions (~{est_secs}s; fast on subsequent runs)");
}
const CHUNK_SIZE: usize = 512;
let mut completed = 0usize;
for chunk in indices.chunks(CHUNK_SIZE) {
let mut chunk_misses: Vec<(usize, String, String, u32, u32)> = Vec::new();
for &i in chunk {
let function = &self.functions[i];
let rel =
if let Ok(r) = std::path::Path::new(&function.file).strip_prefix(repo_root) {
r.to_string_lossy().replace('\\', "/")
} else {
function.file.replace('\\', "/")
};
let start_line = function.line;
let end_line =
(start_line + function.metrics.loc.saturating_sub(1)).max(start_line);
let key = crate::touch_cache::cache_key(&sha, &rel, start_line, end_line);
if let Some(&(count, days)) = cache.get(&key) {
self.functions[i].touch_count_30d = Some(count);
self.functions[i].days_since_last_change = days;
completed += 1;
} else {
chunk_misses.push((i, key, rel, start_line, end_line));
}
}
if let Some(f) = progress_fn {
f(completed, total);
}
if chunk_misses.is_empty() {
continue;
}
let results: Vec<(usize, String, (usize, Option<u32>))> = chunk_misses
.par_iter()
.map(|(idx, key, rel, start_line, end_line)| {
let value = match crate::git::function_touch_metrics_at(
repo_root,
rel,
*start_line,
*end_line,
timestamp,
) {
Ok((count, days)) => (count, days),
Err(_) => (0usize, None),
};
(*idx, key.clone(), value)
})
.collect();
for (idx, key, (count, days)) in results {
self.functions[idx].touch_count_30d = Some(count);
self.functions[idx].days_since_last_change = days;
cache.insert(key, (count, days));
completed += 1;
if let Some(f) = progress_fn {
f(completed, total);
}
}
}
let known_shas: Vec<String> = {
let mut commits = Index::load_or_new(&index_path(repo_root))
.map(|idx| idx.commits)
.unwrap_or_default();
commits.sort_by_key(|c| std::cmp::Reverse(c.timestamp));
let mut shas = vec![sha.clone()];
shas.extend(commits.into_iter().map(|e| e.sha).filter(|s| s != &sha));
shas
};
crate::touch_cache::evict_old_entries(&mut cache, &known_shas);
if let Err(e) = crate::touch_cache::write_touch_cache(repo_root, &cache) {
eprintln!("warning: failed to write touch cache: {e}");
}
Ok(())
}
fn populate_file_level_touch_metrics(
&mut self,
repo_root: &std::path::Path,
) -> anyhow::Result<()> {
use std::collections::HashMap;
let mut unique_files: HashMap<String, Vec<usize>> = HashMap::new();
for (idx, function) in self.functions.iter().enumerate() {
unique_files
.entry(function.file.clone())
.or_default()
.push(idx);
}
let abs_to_rel: HashMap<String, String> = unique_files
.keys()
.map(|abs| {
let rel = if let Ok(r) = std::path::Path::new(abs).strip_prefix(repo_root) {
r.to_string_lossy().to_string()
} else {
abs.clone()
};
(abs.clone(), rel)
})
.collect();
let batched = crate::git::batch_touch_metrics_at(repo_root, self.commit.timestamp)
.unwrap_or_else(|_| crate::git::BatchedTouchMetrics {
touch_count_30d: HashMap::new(),
days_since_last_change: HashMap::new(),
});
let stale_files: std::collections::HashSet<&str> = abs_to_rel
.values()
.map(|s| s.as_str())
.filter(|rel| !batched.days_since_last_change.contains_key(*rel))
.collect();
let stale_days =
crate::git::batch_last_touch_for_files(repo_root, &stale_files, self.commit.timestamp);
for (abs_path, function_indices) in &unique_files {
let rel = abs_to_rel
.get(abs_path)
.map(|s| s.as_str())
.unwrap_or(abs_path);
let touch_count = batched.touch_count_30d.get(rel).copied().or(Some(0));
let days_since = batched
.days_since_last_change
.get(rel)
.copied()
.or_else(|| stale_days.get(rel).copied());
for &idx in function_indices {
self.functions[idx].touch_count_30d = touch_count;
self.functions[idx].days_since_last_change = days_since;
}
}
Ok(())
}
pub fn populate_touch_metrics(
&mut self,
repo_root: &std::path::Path,
mode: TouchMode,
progress_fn: Option<&dyn Fn(usize, usize)>,
) -> anyhow::Result<()> {
match mode {
TouchMode::File => self.populate_file_level_touch_metrics(repo_root),
TouchMode::PerFunction => {
self.populate_per_function_touch_metrics(repo_root, progress_fn)
}
TouchMode::Hybrid { threshold } => {
self.populate_hybrid_touch_metrics(repo_root, threshold, progress_fn)
}
}
}
fn populate_hybrid_touch_metrics(
&mut self,
repo_root: &std::path::Path,
threshold: usize,
progress_fn: Option<&dyn Fn(usize, usize)>,
) -> anyhow::Result<()> {
self.populate_file_level_touch_metrics(repo_root)?;
let hot_indices: Vec<usize> = self
.functions
.iter()
.enumerate()
.filter_map(|(i, f)| {
if f.touch_count_30d.unwrap_or(0) >= threshold {
Some(i)
} else {
None
}
})
.collect();
if hot_indices.is_empty() {
return Ok(());
}
let total = self.functions.len();
eprintln!(
"hybrid touch: {}/{} functions qualify (≥{} touches/30d), refining with git log -L",
hot_indices.len(),
total,
threshold
);
self.populate_per_function_touch_for_indices(repo_root, &hot_indices, progress_fn)
}
pub fn adjust_recency_for_branch(
&mut self,
repo_root: &std::path::Path,
merge_base_sha: &str,
merge_base_ts: i64,
) {
let merge_base_age_days = ((self.commit.timestamp - merge_base_ts).max(0) / 86400) as u32;
let mut files_needing_lookup: std::collections::HashSet<String> =
std::collections::HashSet::new();
for func in &self.functions {
if func
.days_since_last_change
.is_some_and(|d| d < merge_base_age_days)
{
files_needing_lookup.insert(func.file.clone());
}
}
let mut pre_branch: std::collections::HashMap<String, Option<u32>> =
std::collections::HashMap::new();
for abs_file in &files_needing_lookup {
let rel = std::path::Path::new(abs_file)
.strip_prefix(repo_root)
.map(|p| p.to_string_lossy().replace('\\', "/"))
.unwrap_or_else(|_| abs_file.replace('\\', "/"));
let days = crate::git::days_since_last_change_at_sha(
repo_root,
&rel,
merge_base_sha,
self.commit.timestamp,
);
pre_branch.insert(abs_file.clone(), days);
}
for func in &mut self.functions {
if let Some(Some(pre_days)) = pre_branch.get(&func.file) {
func.days_since_last_change = Some(*pre_days);
}
}
}
pub fn populate_callgraph(
&mut self,
call_graph: &crate::callgraph::CallGraph,
exact_threshold: usize,
approx_k: usize,
) -> bool {
use std::collections::HashMap;
let n = call_graph.node_count();
let approximate = n > exact_threshold;
let pagerank_scores = call_graph.pagerank(0.85, 30, 1e-6);
let betweenness_scores = if approximate {
call_graph.betweenness_centrality_approx(approx_k)
} else {
call_graph.betweenness_centrality()
};
let scc_info = call_graph.find_strongly_connected_components();
let dependency_depths = call_graph.compute_dependency_depth();
let fan_in_map = call_graph.build_fan_in_map();
let mut churn_map: HashMap<String, usize> = HashMap::new();
for function in &self.functions {
if let Some(ref churn) = function.churn {
let total_churn = churn.lines_added + churn.lines_deleted;
churn_map.insert(function.function_id.clone(), total_churn);
}
}
for function in &mut self.functions {
let function_id = &function.function_id;
if call_graph.contains(function_id) {
let (scc_id, scc_size) = scc_info.get(function_id).copied().unwrap_or((0, 1));
let dependency_depth = dependency_depths.get(function_id).copied().flatten();
let neighbor_churn = if let Some(callees) = call_graph.callees_of(function_id) {
let total: usize = callees
.filter_map(|callee_id| churn_map.get(callee_id))
.sum();
if total > 0 {
Some(total)
} else {
None
}
} else {
None
};
function.callgraph = Some(CallGraphMetrics {
fan_in: fan_in_map.get(function_id).copied().unwrap_or(0),
fan_out: call_graph.fan_out(function_id),
pagerank: pagerank_scores.get(function_id).copied().unwrap_or(0.0),
betweenness: betweenness_scores.get(function_id).copied().unwrap_or(0.0),
scc_id,
scc_size,
is_entrypoint: call_graph.is_entry_point(function_id),
dependency_depth,
neighbor_churn,
});
}
}
approximate
}
pub fn compute_activity_risk(&mut self, weights: Option<&crate::scoring::ScoringWeights>) {
let default_weights = crate::scoring::ScoringWeights::default();
let weights = weights.unwrap_or(&default_weights);
for function in &mut self.functions {
let churn = function
.churn
.as_ref()
.map(|c| (c.lines_added, c.lines_deleted));
let (fan_in, scc_size, dependency_depth, neighbor_churn) =
if let Some(ref cg) = function.callgraph {
(
Some(cg.fan_in),
Some(cg.scc_size),
cg.dependency_depth,
cg.neighbor_churn,
)
} else {
(None, None, None, None)
};
let (activity_risk, risk_factors) = crate::scoring::compute_activity_risk(
&crate::scoring::ActivityRiskInput {
lrs: function.lrs,
churn,
touch_count_30d: function.touch_count_30d,
days_since_last_change: function.days_since_last_change,
fan_in,
scc_size,
dependency_depth,
neighbor_churn,
},
weights,
);
if activity_risk > function.lrs || risk_factors.churn > 0.0 {
function.activity_risk = Some(activity_risk);
function.risk_factors = Some(risk_factors);
}
}
}
pub fn populate_patterns(&mut self, thresholds: &crate::patterns::Thresholds) {
for function in &mut self.functions {
let t1 = crate::patterns::Tier1Input {
cc: function.metrics.cc as usize,
nd: function.metrics.nd as usize,
fo: function.metrics.fo as usize,
ns: function.metrics.ns as usize,
loc: function.metrics.loc as usize,
};
let (fan_in, scc_size, neighbor_churn, is_entrypoint) =
if let Some(ref cg) = function.callgraph {
(
Some(cg.fan_in),
Some(cg.scc_size),
cg.neighbor_churn,
cg.is_entrypoint,
)
} else {
(None, None, None, false)
};
let t2 = crate::patterns::Tier2Input {
fan_in,
scc_size,
churn_lines: None,
days_since_last_change: function.days_since_last_change,
neighbor_churn,
is_entrypoint,
};
function.patterns = crate::patterns::classify(&t1, &t2, thresholds);
}
}
pub fn populate_pattern_details(&mut self, thresholds: &crate::patterns::Thresholds) {
for function in &mut self.functions {
let t1 = crate::patterns::Tier1Input {
cc: function.metrics.cc as usize,
nd: function.metrics.nd as usize,
fo: function.metrics.fo as usize,
ns: function.metrics.ns as usize,
loc: function.metrics.loc as usize,
};
let (fan_in, scc_size, neighbor_churn, is_entrypoint) =
if let Some(ref cg) = function.callgraph {
(
Some(cg.fan_in),
Some(cg.scc_size),
cg.neighbor_churn,
cg.is_entrypoint,
)
} else {
(None, None, None, false)
};
let t2 = crate::patterns::Tier2Input {
fan_in,
scc_size,
churn_lines: None,
days_since_last_change: function.days_since_last_change,
neighbor_churn,
is_entrypoint,
};
function.pattern_details =
Some(crate::patterns::classify_detailed(&t1, &t2, thresholds));
}
}
pub fn compute_percentiles(&mut self) {
let n = self.functions.len();
if n == 0 {
return;
}
let mut scores: Vec<f64> = self
.functions
.iter()
.map(|f| f.activity_risk.unwrap_or(f.lrs))
.collect();
scores.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
let threshold_10 = scores[n.saturating_sub(1) * 90 / 100];
let threshold_5 = scores[n.saturating_sub(1) * 95 / 100];
let threshold_1 = scores[n.saturating_sub(1) * 99 / 100];
for function in &mut self.functions {
let score = function.activity_risk.unwrap_or(function.lrs);
function.percentile = Some(PercentileFlags {
is_top_10_pct: score >= threshold_10,
is_top_5_pct: score >= threshold_5,
is_top_1_pct: score >= threshold_1,
});
}
}
pub fn populate_driver_labels(&mut self, percentile: u8) {
let thresholds = compute_dimension_thresholds(&self.functions, percentile);
let mut sorted_cc: Vec<usize> = self
.functions
.iter()
.map(|f| f.metrics.cc as usize)
.collect();
let mut sorted_nd: Vec<usize> = self
.functions
.iter()
.map(|f| f.metrics.nd as usize)
.collect();
let mut sorted_fo: Vec<usize> = self
.functions
.iter()
.map(|f| f.callgraph.as_ref().map(|cg| cg.fan_out).unwrap_or(0))
.collect();
let mut sorted_fi: Vec<usize> = self
.functions
.iter()
.map(|f| f.callgraph.as_ref().map(|cg| cg.fan_in).unwrap_or(0))
.collect();
let mut sorted_touch: Vec<usize> = self
.functions
.iter()
.map(|f| f.touch_count_30d.unwrap_or(0))
.collect();
sorted_cc.sort_unstable();
sorted_nd.sort_unstable();
sorted_fo.sort_unstable();
sorted_fi.sort_unstable();
sorted_touch.sort_unstable();
for function in &mut self.functions {
let label = driving_dimension_label(function, &thresholds).to_string();
function.driver_detail = if label == "composite" {
compute_near_miss_detail(
function,
&sorted_cc,
&sorted_nd,
&sorted_fo,
&sorted_fi,
&sorted_touch,
)
} else {
None
};
function.driver = Some(label);
}
}
pub fn compute_quadrants(&mut self, driver_threshold_percentile: u8, ranker_applied: bool) {
if self.functions.is_empty() {
return;
}
let thresholds = compute_dimension_thresholds(&self.functions, driver_threshold_percentile);
let touch_p50 = thresholds.touch_med;
for function in &mut self.functions {
let touch_above_p50 = function
.touch_count_30d
.map(|t| t > touch_p50)
.unwrap_or(false);
let recently_changed = function
.days_since_last_change
.map(|d| d <= 30)
.unwrap_or(false);
let high_ranker_score =
ranker_applied && function.activity_risk.map(|r| r >= 0.7).unwrap_or(false);
let is_active = touch_above_p50 || recently_changed || high_ranker_score;
let is_high_risk = matches!(function.band, RiskBand::Critical | RiskBand::High);
function.quadrant = Some(
match (is_high_risk, is_active) {
(true, true) => "fire",
(true, false) => "debt",
(false, true) => "watch",
(false, false) => "ok",
}
.to_string(),
);
}
}
pub fn compute_summary(&mut self, betweenness_approximate: bool) {
let n = self.functions.len();
if n == 0 {
self.summary = Some(SnapshotSummary {
total_functions: 0,
total_activity_risk: 0.0,
top_1_pct_share: 0.0,
top_5_pct_share: 0.0,
top_10_pct_share: 0.0,
by_band: std::collections::BTreeMap::new(),
call_graph: None,
});
return;
}
let mut scored: Vec<f64> = self
.functions
.iter()
.map(|f| f.activity_risk.unwrap_or(f.lrs))
.collect();
scored.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
let total_risk: f64 = scored.iter().sum();
let (top_1_pct_share, top_5_pct_share, top_10_pct_share) =
compute_top_k_shares(&scored, total_risk);
self.summary = Some(SnapshotSummary {
total_functions: n,
total_activity_risk: total_risk,
top_1_pct_share,
top_5_pct_share,
top_10_pct_share,
by_band: compute_band_distribution(&self.functions),
call_graph: compute_call_graph_stats(&self.functions, n, betweenness_approximate),
});
}
pub fn to_jsonl(&self) -> Result<String> {
let commit_json =
serde_json::to_value(&self.commit).context("failed to serialize commit")?;
let mut lines = Vec::with_capacity(self.functions.len());
for func in &self.functions {
let mut obj = serde_json::to_value(func).context("failed to serialize function")?;
obj.as_object_mut()
.context("serialized function is not a JSON object")?
.insert("commit".to_string(), commit_json.clone());
lines.push(serde_json::to_string(&obj).context("failed to serialize JSONL line")?);
}
Ok(lines.join("\n"))
}
pub fn to_json(&self) -> Result<String> {
serde_json::to_string_pretty(self).context("failed to serialize snapshot to JSON")
}
pub fn write_json_to<W: std::io::Write>(&self, writer: &mut W) -> Result<()> {
serde_json::to_writer_pretty(&mut *writer, self)
.context("failed to write snapshot JSON")?;
writeln!(writer).context("failed to write trailing newline")
}
pub fn write_jsonl_to<W: std::io::Write>(&self, writer: &mut W) -> Result<()> {
let commit_json =
serde_json::to_value(&self.commit).context("failed to serialize commit")?;
for func in &self.functions {
let mut obj = serde_json::to_value(func).context("failed to serialize function")?;
obj.as_object_mut()
.context("serialized function is not a JSON object")?
.insert("commit".to_string(), commit_json.clone());
serde_json::to_writer(writer as &mut dyn std::io::Write, &obj)
.context("failed to write JSONL line")?;
writeln!(writer).context("failed to write JSONL newline")?;
}
Ok(())
}
pub fn from_json(json: &str) -> Result<Self> {
let snapshot: Snapshot =
serde_json::from_str(json).context("failed to deserialize snapshot from JSON")?;
if snapshot.schema_version < SNAPSHOT_SCHEMA_MIN_VERSION
|| snapshot.schema_version > SNAPSHOT_SCHEMA_VERSION
{
anyhow::bail!(
"unsupported schema version: got {}, supported range {}-{}",
snapshot.schema_version,
SNAPSHOT_SCHEMA_MIN_VERSION,
SNAPSHOT_SCHEMA_VERSION
);
}
Ok(snapshot)
}
/// Get the commit SHA for this snapshot
pub fn commit_sha(&self) -> &str {
&self.commit.sha
}
}
/// Returns (top_1_pct_share, top_5_pct_share, top_10_pct_share) from a
/// descending-sorted score slice and the pre-computed total.
fn compute_top_k_shares(scored: &[f64], total_risk: f64) -> (f64, f64, f64) {
let n = scored.len();
let safe_div = |a: f64, b: f64| if b > 0.0 { a / b } else { 0.0 };
let top1_sum: f64 = scored.iter().take((n / 100).max(1)).sum();
let top5_sum: f64 = scored.iter().take((n * 5 / 100).max(1)).sum();
let top10_sum: f64 = scored.iter().take((n / 10).max(1)).sum();
(
safe_div(top1_sum, total_risk),
safe_div(top5_sum, total_risk),
safe_div(top10_sum, total_risk),
)
}
/// Builds a band → BandStats map from the function list.
fn compute_band_distribution(
functions: &[FunctionSnapshot],
) -> std::collections::BTreeMap<String, BandStats> {
let mut by_band = std::collections::BTreeMap::new();
for func in functions {
let score = func.activity_risk.unwrap_or(func.lrs);
let entry = by_band
.entry(func.band.as_str().to_string())
.or_insert(BandStats {
count: 0,
sum_risk: 0.0,
});
entry.count += 1;
entry.sum_risk += score;
}
by_band
}
/// Computes call-graph-level summary statistics, or None if no call graph data.
fn compute_call_graph_stats(
functions: &[FunctionSnapshot],
n: usize,
betweenness_approximate: bool,
) -> Option<CallGraphStats> {
if !functions.iter().any(|f| f.callgraph.is_some()) {
return None;
}
let total_edges: usize = functions
.iter()
.filter_map(|f| f.callgraph.as_ref())
.map(|cg| cg.fan_out)
.sum();
let total_fan_in: usize = functions
.iter()
.filter_map(|f| f.callgraph.as_ref())
.map(|cg| cg.fan_in)
.sum();
let mut scc_sizes: std::collections::HashMap<usize, usize> = std::collections::HashMap::new();
for func in functions {
if let Some(ref cg) = func.callgraph {
if cg.scc_size > 1 {
scc_sizes.insert(cg.scc_id, cg.scc_size);
}
}
}
Some(CallGraphStats {
total_edges,
avg_fan_in: total_fan_in as f64 / n as f64,
scc_count: scc_sizes.len(),
largest_scc_size: scc_sizes.values().copied().max().unwrap_or(0),
betweenness_approximate,
})
}
/// Percentile-derived thresholds for driving dimension detection.
/// Computed once per snapshot from the distribution of all functions.
pub struct DimensionThresholds {
pub cc_high: usize, // Pth percentile of cc — "high_complexity" gate
pub cc_med: usize, // 50th percentile of cc — floor for "high_fanin_complex"
pub cc_low: usize, // (100-P)th percentile of cc — "low cc" in "high_churn_low_cc"
pub nd_high: usize, // Pth percentile of nd — "deep_nesting" gate
pub fan_out_high: usize, // Pth percentile of fan_out — "high_fanout_churning" gate
pub fan_in_high: usize, // Pth percentile of fan_in — "high_fanin_complex" gate
pub touch_high: usize, // Pth percentile of touch_count — "high churn" gate
pub touch_med: usize, // 50th percentile of touch_count — floor for "high_fanout_churning"
}
/// Compute percentile-derived thresholds from a slice of function snapshots.
pub fn compute_dimension_thresholds(
functions: &[FunctionSnapshot],
percentile: u8,
) -> DimensionThresholds {
let n = functions.len();
if n == 0 {
return DimensionThresholds {
cc_high: 0,
cc_med: 0,
cc_low: 0,
nd_high: 0,
fan_out_high: 0,
fan_in_high: 0,
touch_high: 0,
touch_med: 0,
};
}
let p = percentile as usize;
let anti_p = 100 - p;
let percentile_idx = |pct: usize| (pct * (n - 1)) / 100;
let mut cc_vals: Vec<usize> = functions.iter().map(|f| f.metrics.cc as usize).collect();
cc_vals.sort_unstable();
let cc_high = cc_vals[percentile_idx(p)];
let cc_med = cc_vals[percentile_idx(50)];
let cc_low = cc_vals[percentile_idx(anti_p)];
let mut nd_vals: Vec<usize> = functions.iter().map(|f| f.metrics.nd as usize).collect();
nd_vals.sort_unstable();
let nd_high = nd_vals[percentile_idx(p)];
let mut fo_vals: Vec<usize> = functions
.iter()
.map(|f| f.callgraph.as_ref().map(|cg| cg.fan_out).unwrap_or(0))
.collect();
fo_vals.sort_unstable();
let fan_out_high = fo_vals[percentile_idx(p)];
let mut fi_vals: Vec<usize> = functions
.iter()
.map(|f| f.callgraph.as_ref().map(|cg| cg.fan_in).unwrap_or(0))
.collect();
fi_vals.sort_unstable();
let fan_in_high = fi_vals[percentile_idx(p)];
let mut touch_vals: Vec<usize> = functions
.iter()
.map(|f| f.touch_count_30d.unwrap_or(0))
.collect();
touch_vals.sort_unstable();
let touch_high = touch_vals[percentile_idx(p)];
let touch_med = touch_vals[percentile_idx(50)];
DimensionThresholds {
cc_high,
cc_med,
cc_low,
nd_high,
fan_out_high,
fan_in_high,
touch_high,
touch_med,
}
}
/// Normalize a driver label string to a canonical `'static` str.
pub fn normalize_driver_label(label: &str) -> &'static str {
match label {
"cyclic_dep" => "cyclic_dep",
"high_complexity" => "high_complexity",
"high_churn_low_cc" => "high_churn_low_cc",
"high_fanout_churning" => "high_fanout_churning",
"deep_nesting" => "deep_nesting",
"high_fanin_complex" => "high_fanin_complex",
_ => "composite",
}
}
/// Map a (driver, quadrant) pair to a recommended action string.
///
/// `quadrant` is one of `"fire"`, `"debt"`, `"watch"`, `"ok"`, or `""` (unknown).
/// When quadrant context is available the action is more specific; the generic
/// driver-only text is used as a fallback.
pub fn driver_action_for_quadrant(driver: &str, quadrant: &str) -> &'static str {
match (driver, quadrant) {
("cyclic_dep", "fire") => "Break cycle now — circular dep is actively changing",
("cyclic_dep", _) => "Resolve dependency cycle",
("high_complexity", "fire") => "Extract sub-functions now — actively changing",
("high_complexity", "debt") => "Schedule CC reduction — stable, plan for next sprint",
("high_complexity", _) => "Reduce cyclomatic complexity",
("high_churn_low_cc", "fire") => "Add tests now — churning without a safety net",
("high_churn_low_cc", _) => "Add tests before next change",
("high_fanout_churning", "fire") => {
"Extract interface boundary — high coupling + active change"
}
("high_fanout_churning", _) => "Consider extracting an interface boundary",
("deep_nesting", "fire") => "Flatten nesting before next change",
("deep_nesting", "debt") => "Schedule flattening — deep nesting, currently quiet",
("deep_nesting", _) => "Flatten nesting depth",
("high_fanin_complex", "fire") => "Stabilize interface — many callers + active changes",
("high_fanin_complex", _) => "Stabilize interface — high fan-in makes changes risky",
(_, "fire") => "Actively risky — plan refactor this sprint",
_ => "Monitor: review complexity trends before next modification",
}
}
/// Map a driver label to its recommended action text (quadrant-agnostic fallback).
pub fn driver_action(label: &str) -> &'static str {
driver_action_for_quadrant(label, "")
}
/// Identify the primary driving dimension for a function's risk.
///
/// Returns a stable label: one of `"cyclic_dep"`, `"high_complexity"`,
/// `"high_churn_low_cc"`, `"high_fanout_churning"`, `"deep_nesting"`,
/// `"high_fanin_complex"`, or `"composite"`. Uses percentile-relative thresholds
/// derived from the snapshot's own distribution; `cyclic_dep` stays absolute.
pub fn driving_dimension_label(
func: &FunctionSnapshot,
thresholds: &DimensionThresholds,
) -> &'static str {
let in_cycle = func
.callgraph
.as_ref()
.map(|cg| cg.scc_size > 1)
.unwrap_or(false);
let fan_out = func.callgraph.as_ref().map(|cg| cg.fan_out).unwrap_or(0);
let fan_in = func.callgraph.as_ref().map(|cg| cg.fan_in).unwrap_or(0);
let touch_count = func.touch_count_30d.unwrap_or(0);
let cc = func.metrics.cc as usize;
let nd = func.metrics.nd as usize;
if in_cycle {
"cyclic_dep"
} else if cc > thresholds.cc_high {
"high_complexity"
} else if touch_count > thresholds.touch_high && cc < thresholds.cc_low {
"high_churn_low_cc"
} else if fan_out > thresholds.fan_out_high && touch_count > thresholds.touch_med {
"high_fanout_churning"
} else if nd > thresholds.nd_high {
"deep_nesting"
} else if fan_in > thresholds.fan_in_high && cc > thresholds.cc_med {
"high_fanin_complex"
} else {
"composite"
}
}
/// Compute near-miss detail string for composite-labeled functions.
///
/// Returns a string like "cc (P72), nd (P68)" listing the top dimensions that
/// are above the 40th percentile (above median) but below the firing threshold.
/// Returns None when no dimension is notable.
fn compute_near_miss_detail(
func: &FunctionSnapshot,
sorted_cc: &[usize],
sorted_nd: &[usize],
sorted_fo: &[usize],
sorted_fi: &[usize],
sorted_touch: &[usize],
) -> Option<String> {
let pct_rank = |v: usize, sorted: &[usize]| -> u8 {
if sorted.is_empty() {
return 0;
}
((sorted.partition_point(|&x| x < v) * 100) / sorted.len()) as u8
};
let mut near: Vec<(&str, u8)> = vec![
("cc", pct_rank(func.metrics.cc as usize, sorted_cc)),
("nd", pct_rank(func.metrics.nd as usize, sorted_nd)),
(
"fan_out",
pct_rank(
func.callgraph.as_ref().map(|cg| cg.fan_out).unwrap_or(0),
sorted_fo,
),
),
(
"fan_in",
pct_rank(
func.callgraph.as_ref().map(|cg| cg.fan_in).unwrap_or(0),
sorted_fi,
),
),
(
"touch",
pct_rank(func.touch_count_30d.unwrap_or(0), sorted_touch),
),
]
.into_iter()
.filter(|(_, rank)| *rank >= 40)
.collect();
near.sort_by_key(|a| std::cmp::Reverse(a.1));
near.truncate(3);
if near.is_empty() {
return None;
}
Some(
near.iter()
.map(|(name, rank)| format!("{} (P{})", name, rank))
.collect::<Vec<_>>()
.join(", "),
)
}
/// churn → touch_metrics → callgraph → activity_risk + percentiles + summary.
pub struct SnapshotEnricher {
snapshot: Snapshot,
betweenness_approximate: bool,
}
impl SnapshotEnricher {
/// Create a new enricher wrapping the given snapshot.
pub fn new(snapshot: Snapshot) -> Self {
SnapshotEnricher {
snapshot,
betweenness_approximate: false,
}
}
/// Detect and populate the `subsystem` field for every function.
///
/// Walks `repo_root` once to find manifest files (package.json, Cargo.toml,
/// pyproject.toml, go.mod, etc.) and tags each function with its nearest
/// manifest-containing ancestor directory relative to `repo_root`.
/// No-op (returns self unchanged) if `repo_root` does not exist.
pub fn with_subsystems(mut self, repo_root: &Path) -> Self {
if repo_root.exists() {
self.snapshot.populate_subsystems(repo_root);
}
self
}
/// Populate churn metrics from a file churn map.
pub fn with_churn(
mut self,
file_churns: &std::collections::HashMap<String, crate::git::FileChurn>,
) -> Self {
self.snapshot.populate_churn(file_churns);
self
}
/// Populate touch count and recency metrics from git.
/// On error, emits a warning to stderr and continues.
pub fn with_touch_metrics(
mut self,
repo_root: &Path,
mode: TouchMode,
progress_fn: Option<Box<dyn Fn(usize, usize)>>,
) -> Self {
if let Err(e) =
self.snapshot
.populate_touch_metrics(repo_root, mode, progress_fn.as_deref())
{
eprintln!("Warning: failed to populate touch metrics: {}", e);
}
self
}
/// Replace branch-inflated recency with pre-branch last-change dates.
/// No-op when merge_base is None (on main, or no divergence).
pub fn with_branch_recency_adjustment(
mut self,
repo_root: &Path,
merge_base: Option<&(String, i64)>,
) -> Self {
if let Some((sha, ts)) = merge_base {
self.snapshot.adjust_recency_for_branch(repo_root, sha, *ts);
}
self
}
/// Populate call graph metrics (PageRank, fan-in, betweenness, SCC, etc).
///
/// Betweenness is computed exactly when `call_graph.nodes.len() <= exact_threshold`,
/// and via k-source approximation otherwise.
pub fn with_callgraph(
mut self,
call_graph: &crate::callgraph::CallGraph,
exact_threshold: usize,
approx_k: usize,
) -> Self {
self.betweenness_approximate =
self.snapshot
.populate_callgraph(call_graph, exact_threshold, approx_k);
self
}
/// Compute activity risk, percentile flags, driver labels, and summary statistics.
///
/// Must be called after with_churn, with_touch_metrics, and with_callgraph.
pub fn enrich(
mut self,
weights: Option<&crate::scoring::ScoringWeights>,
driver_threshold_percentile: u8,
) -> Self {
self.snapshot.compute_activity_risk(weights);
self.snapshot.compute_percentiles();
self.snapshot
.populate_driver_labels(driver_threshold_percentile);
self.snapshot
.compute_quadrants(driver_threshold_percentile, false);
self.snapshot.compute_summary(self.betweenness_approximate);
self
}
/// Consume the enricher and return the fully enriched snapshot.
pub fn build(self) -> Snapshot {
self.snapshot
}
}
impl Index {
/// Create a new empty index (default compaction level 0 - full snapshots)
pub fn new() -> Self {
Index {
schema_version: INDEX_SCHEMA_VERSION,
compaction_level: Some(0), // Default: full snapshots
commits: Vec::new(),
}
}
/// Get compaction level (defaults to 0 if not set for backward compatibility)
pub fn compaction_level(&self) -> u32 {
self.compaction_level.unwrap_or(0)
}
/// Set compaction level
pub fn set_compaction_level(&mut self, level: u32) {
self.compaction_level = Some(level);
}
/// Load index from JSON file, or create new if file doesn't exist
pub fn load_or_new(path: &Path) -> Result<Self> {
if path.exists() {
let json = std::fs::read_to_string(path)
.with_context(|| format!("failed to read index file: {}", path.display()))?;
Self::from_json(&json)
} else {
Ok(Self::new())
}
}
/// Deserialize index from JSON string
pub fn from_json(json: &str) -> Result<Self> {
let index: Index =
serde_json::from_str(json).context("failed to deserialize index from JSON")?;
// Validate schema version
if index.schema_version != INDEX_SCHEMA_VERSION {
anyhow::bail!(
"index schema version mismatch: expected {}, got {}",
INDEX_SCHEMA_VERSION,
index.schema_version
);
}
Ok(index)
}
/// Serialize index to JSON string (deterministic ordering)
pub fn to_json(&self) -> Result<String> {
serde_json::to_string_pretty(self).context("failed to serialize index to JSON")
}
/// Add or update a commit entry in the index
///
/// If the commit already exists, this is idempotent (no-op).
/// Entries are kept sorted deterministically by timestamp, then SHA.
pub fn add_commit(&mut self, entry: IndexEntry) {
// Check if entry already exists
if !self.commits.iter().any(|c| c.sha == entry.sha) {
self.commits.push(entry);
// Sort deterministically: timestamp ascending, then SHA ASCII ascending
self.commits.sort_by(|a, b| {
a.timestamp
.cmp(&b.timestamp)
.then_with(|| a.sha.cmp(&b.sha))
});
}
}
/// Remove a commit entry by SHA
pub fn remove_commit(&mut self, sha: &str) {
self.commits.retain(|c| c.sha != sha);
}
/// Check if index contains a commit
pub fn contains(&self, sha: &str) -> bool {
self.commits.iter().any(|c| c.sha == sha)
}
}
impl Default for Index {
fn default() -> Self {
Self::new()
}
}
/// Delta encoding schema version
const DELTA_SNAPSHOT_SCHEMA_VERSION: u32 = 1;
/// Delta between two snapshots — stores only what changed.
///
/// Stored as `<sha>.delta.json.zst` in the snapshots directory.
/// Reconstruct the full snapshot by loading the base and calling `apply_delta`.
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
#[serde(rename_all = "snake_case")]
pub struct DeltaSnapshot {
pub schema_version: u32,
/// Schema version of the full snapshot this delta reconstructs.
pub snapshot_schema_version: u32,
pub commit: CommitInfo,
pub analysis: AnalysisInfo,
/// SHA of the snapshot used as the delta base.
pub base_sha: String,
pub added: Vec<FunctionSnapshot>,
pub modified: Vec<FunctionSnapshot>,
pub removed: Vec<String>,
/// Preserved from the original snapshot so HTML history charts remain intact
/// after reconstruction (callers filter on `summary.is_some()`).
#[serde(skip_serializing_if = "Option::is_none")]
pub summary: Option<SnapshotSummary>,
}
/// Compute a delta from `base` to `current`.
///
/// The returned `DeltaSnapshot` encodes only the functions that were added,
/// modified, or removed relative to `base`.
pub fn compute_delta(base: &Snapshot, current: &Snapshot) -> DeltaSnapshot {
use std::collections::HashMap;
let base_map: HashMap<&str, &FunctionSnapshot> = base
.functions
.iter()
.map(|f| (f.function_id.as_str(), f))
.collect();
let current_map: HashMap<&str, ()> = current
.functions
.iter()
.map(|f| (f.function_id.as_str(), ()))
.collect();
let added: Vec<FunctionSnapshot> = current
.functions
.iter()
.filter(|f| !base_map.contains_key(f.function_id.as_str()))
.cloned()
.collect();
let modified: Vec<FunctionSnapshot> = current
.functions
.iter()
.filter(|f| {
base_map
.get(f.function_id.as_str())
.map(|bf| *bf != *f)
.unwrap_or(false)
})
.cloned()
.collect();
let removed: Vec<String> = base
.functions
.iter()
.filter(|f| !current_map.contains_key(f.function_id.as_str()))
.map(|f| f.function_id.clone())
.collect();
DeltaSnapshot {
schema_version: DELTA_SNAPSHOT_SCHEMA_VERSION,
snapshot_schema_version: current.schema_version,
commit: current.commit.clone(),
analysis: current.analysis.clone(),
base_sha: base.commit.sha.clone(),
added,
modified,
removed,
summary: current.summary.clone(),
}
}
/// Reconstruct a full `Snapshot` by applying a delta on top of `base`.
pub fn apply_delta(base: Snapshot, delta: DeltaSnapshot) -> Snapshot {
use std::collections::HashMap;
let mut functions: HashMap<String, FunctionSnapshot> = base
.functions
.into_iter()
.map(|f| (f.function_id.clone(), f))
.collect();
for id in &delta.removed {
functions.remove(id);
}
for func in delta.modified {
functions.insert(func.function_id.clone(), func);
}
for func in delta.added {
functions.insert(func.function_id.clone(), func);
}
let mut result: Vec<FunctionSnapshot> = functions.into_values().collect();
result.sort_by(|a, b| a.function_id.cmp(&b.function_id));
Snapshot {
schema_version: delta.snapshot_schema_version,
commit: delta.commit,
analysis: delta.analysis,
functions: result,
summary: delta.summary,
aggregates: None,
}
}
/// Persist a delta snapshot to `<sha>.delta.json.zst`.
pub fn persist_delta(repo_root: &Path, delta: &DeltaSnapshot) -> Result<()> {
let path = delta_snapshot_path(repo_root, &delta.commit.sha);
let json = serde_json::to_string_pretty(delta).context("failed to serialize delta snapshot")?;
let compressed =
zstd::encode_all(json.as_bytes(), 3).context("failed to compress delta snapshot")?;
atomic_write_bytes(&path, &compressed)
.with_context(|| format!("failed to persist delta snapshot: {}", path.display()))
}
/// Get the path to the `.hotspots` directory in the repository root
pub fn hotspots_dir(repo_root: &Path) -> PathBuf {
repo_root.join(".hotspots")
}
/// Get the path to the snapshots directory
pub fn snapshots_dir(repo_root: &Path) -> PathBuf {
hotspots_dir(repo_root).join("snapshots")
}
/// Get the path to the index file
pub fn index_path(repo_root: &Path) -> PathBuf {
hotspots_dir(repo_root).join("index.json")
}
/// Get the path to a snapshot file for a given commit SHA
pub fn snapshot_path(repo_root: &Path, commit_sha: &str) -> PathBuf {
snapshots_dir(repo_root).join(format!("{}.json.zst", commit_sha))
}
/// Get the path to a delta snapshot file for a given commit SHA
pub fn delta_snapshot_path(repo_root: &Path, commit_sha: &str) -> PathBuf {
snapshots_dir(repo_root).join(format!("{}.delta.json.zst", commit_sha))
}
/// Return the path of the snapshot file that actually exists on disk,
/// trying `.json.zst` (new) before `.json` (legacy). Returns `None` if
/// neither exists.
pub fn snapshot_path_existing(repo_root: &Path, commit_sha: &str) -> Option<PathBuf> {
let zst = snapshot_path(repo_root, commit_sha);
if zst.exists() {
return Some(zst);
}
let json = snapshots_dir(repo_root).join(format!("{}.json", commit_sha));
if json.exists() {
return Some(json);
}
None
}
/// Load a snapshot for the given commit SHA from disk.
///
/// Handles full snapshots (`.json.zst`, `.json`), delta snapshots
/// (`.delta.json.zst`), and transparent reconstruction of delta chains.
/// Returns `None` if no snapshot or delta file exists for the SHA.
pub fn load_snapshot(repo_root: &Path, commit_sha: &str) -> Result<Option<Snapshot>> {
// Full snapshot takes priority.
if let Some(path) = snapshot_path_existing(repo_root, commit_sha) {
return Ok(Some(read_snapshot_file(&path)?));
}
// Fall back to delta reconstruction.
let dpath = delta_snapshot_path(repo_root, commit_sha);
if dpath.exists() {
let compressed = std::fs::read(&dpath)
.with_context(|| format!("failed to read delta: {}", dpath.display()))?;
let bytes = zstd::decode_all(compressed.as_slice())
.with_context(|| format!("failed to decompress delta: {}", dpath.display()))?;
let json = String::from_utf8(bytes).context("delta snapshot contains invalid UTF-8")?;
let delta: DeltaSnapshot = serde_json::from_str(&json)
.with_context(|| format!("failed to parse delta: {}", dpath.display()))?;
let base = load_snapshot(repo_root, &delta.base_sha)?.ok_or_else(|| {
anyhow::anyhow!(
"base snapshot {} not found for delta {}",
delta.base_sha,
commit_sha
)
})?;
return Ok(Some(apply_delta(base, delta)));
}
Ok(None)
}
/// Read and parse a snapshot from an arbitrary path, auto-detecting compression.
fn read_snapshot_file(path: &Path) -> Result<Snapshot> {
let is_compressed = path
.file_name()
.and_then(|n| n.to_str())
.map(|n| n.ends_with(".json.zst"))
.unwrap_or(false);
let json: String = if is_compressed {
let compressed = std::fs::read(path)
.with_context(|| format!("failed to read snapshot: {}", path.display()))?;
let bytes = zstd::decode_all(compressed.as_slice())
.with_context(|| format!("failed to decompress snapshot: {}", path.display()))?;
String::from_utf8(bytes).context("snapshot contains invalid UTF-8")?
} else {
std::fs::read_to_string(path)
.with_context(|| format!("failed to read snapshot: {}", path.display()))?
};
Snapshot::from_json(&json)
.with_context(|| format!("failed to parse snapshot: {}", path.display()))
}
/// Write data to file atomically using temp file + rename
pub fn atomic_write(path: &Path, contents: &str) -> Result<()> {
use std::fs;
use std::io::Write;
// Ensure parent directory exists
if let Some(parent) = path.parent() {
fs::create_dir_all(parent)
.with_context(|| format!("failed to create directory: {}", parent.display()))?;
}
// Create temp file in same directory
let temp_path = path.with_extension("tmp");
// Write to temp file
let mut file = fs::File::create(&temp_path)
.with_context(|| format!("failed to create temp file: {}", temp_path.display()))?;
file.write_all(contents.as_bytes())
.with_context(|| format!("failed to write to temp file: {}", temp_path.display()))?;
file.sync_all()
.with_context(|| format!("failed to sync temp file: {}", temp_path.display()))?;
drop(file);
// Atomic rename
fs::rename(&temp_path, path)
.with_context(|| format!("failed to rename temp file to: {}", path.display()))?;
Ok(())
}
/// Write binary data to file atomically using temp file + rename
pub fn atomic_write_bytes(path: &Path, contents: &[u8]) -> Result<()> {
use std::fs;
use std::io::Write;
if let Some(parent) = path.parent() {
fs::create_dir_all(parent)
.with_context(|| format!("failed to create directory: {}", parent.display()))?;
}
let temp_path = path.with_extension("tmp");
let mut file = fs::File::create(&temp_path)
.with_context(|| format!("failed to create temp file: {}", temp_path.display()))?;
file.write_all(contents)
.with_context(|| format!("failed to write to temp file: {}", temp_path.display()))?;
file.sync_all()
.with_context(|| format!("failed to sync temp file: {}", temp_path.display()))?;
drop(file);
fs::rename(&temp_path, path)
.with_context(|| format!("failed to rename temp file to: {}", path.display()))?;
Ok(())
}
/// Persist a snapshot to disk
///
/// # Atomic Writes
///
/// Uses temp file + rename pattern for atomic writes.
/// Persists a snapshot to disk.
///
/// When `force` is false, never overwrites an existing snapshot (fails if one already exists
/// and differs). When `force` is true, overwrites any existing snapshot.
///
/// # Errors
///
/// Returns error if:
/// - `force` is false and snapshot file already exists with different content
/// - Schema version mismatch (if reading existing file)
/// - I/O errors during write
pub fn persist_snapshot(repo_root: &Path, snapshot: &Snapshot, force: bool) -> Result<()> {
let snapshot_path = snapshot_path(repo_root, snapshot.commit_sha());
// Normalize through a parse-reserialize cycle to produce a canonical form.
// This handles float serialization quirks where serde_json may parse a float
// string to a slightly different f64 than what was computed (e.g. a 1-ULP
// difference due to the float parser's rounding). Both the on-disk snapshot
// (already round-tripped once) and the freshly-computed snapshot are brought
// to the same canonical representation before comparing.
let canonical_json = Snapshot::from_json(&snapshot.to_json()?)
.context("failed to normalize snapshot for canonical form")?
.to_json()?;
if !force {
if let Some(existing) = load_snapshot(repo_root, snapshot.commit_sha())? {
// Compare canonical forms (both normalized through one parse-reserialize cycle)
if existing.to_json()? == canonical_json {
return Ok(());
}
anyhow::bail!(
"snapshot already exists and differs: {} (snapshots are immutable; use --force to overwrite)",
snapshot_path.display()
);
}
}
// Compress and write atomically (zstd level 3 — fast with good ratio)
let compressed =
zstd::encode_all(canonical_json.as_bytes(), 3).context("failed to compress snapshot")?;
atomic_write_bytes(&snapshot_path, &compressed)
.with_context(|| format!("failed to persist snapshot: {}", snapshot_path.display()))?;
Ok(())
}
/// Append snapshot entry to index
///
/// Loads existing index, adds entry, and persists atomically.
pub fn append_to_index(repo_root: &Path, snapshot: &Snapshot) -> Result<()> {
let index_path = index_path(repo_root);
// Load existing index or create new
let mut index = Index::load_or_new(&index_path)?;
// Add commit entry
index.add_commit(IndexEntry {
sha: snapshot.commit.sha.clone(),
parents: snapshot.commit.parents.clone(),
timestamp: snapshot.commit.timestamp,
});
// Serialize and write atomically
let json = index.to_json()?;
atomic_write(&index_path, &json)
.with_context(|| format!("failed to update index: {}", index_path.display()))?;
Ok(())
}
/// Rebuild index from snapshots directory
///
/// Scans `.hotspots/snapshots/` and rebuilds `index.json` with deterministic ordering.
/// Useful for recovery if index is corrupted or missing.
///
/// # Ordering
///
/// Index entries are sorted by timestamp (ascending), then SHA (ASCII ascending),
/// ensuring byte-for-byte deterministic output.
pub fn rebuild_index(repo_root: &Path) -> Result<Index> {
let snapshots_dir = snapshots_dir(repo_root);
if !snapshots_dir.exists() {
return Ok(Index::new());
}
let mut index = Index::new();
// Read all snapshot files
let entries = std::fs::read_dir(&snapshots_dir).with_context(|| {
format!(
"failed to read snapshots directory: {}",
snapshots_dir.display()
)
})?;
for entry_result in entries {
let entry = entry_result?;
let path = entry.path();
// Only process snapshot files (.json.zst or legacy .json)
let file_name = path.file_name().and_then(|n| n.to_str()).unwrap_or("");
if !file_name.ends_with(".json.zst") && !file_name.ends_with(".json") {
continue;
}
// Read and parse snapshot (auto-detects compression)
let snapshot = match read_snapshot_file(&path) {
Ok(s) => s,
Err(e) => {
// Log error but continue (some snapshots may be corrupted)
eprintln!(
"Warning: failed to parse snapshot {}: {}",
path.display(),
e
);
continue;
}
};
// Add to index
index.add_commit(IndexEntry {
sha: snapshot.commit.sha,
parents: snapshot.commit.parents,
timestamp: snapshot.commit.timestamp,
});
}
Ok(index)
}
#[cfg(test)]
mod tests {
use super::*;
use crate::report::MetricsReport;
fn create_test_snapshot() -> Snapshot {
let git_context = GitContext {
head_sha: "abc123".to_string(),
parent_shas: vec!["def456".to_string()],
timestamp: 1705600000,
branch: Some("main".to_string()),
is_detached: false,
message: Some("test commit".to_string()),
author: Some("Test Author".to_string()),
is_fix_commit: Some(false),
is_revert_commit: Some(false),
ticket_ids: vec![],
};
let report = FunctionRiskReport {
file: "src/foo.ts".to_string(),
function: "handler".to_string(),
line: 42,
language: Language::TypeScript,
metrics: MetricsReport {
cc: 5,
nd: 2,
fo: 3,
ns: 1,
loc: 10,
},
risk: RiskReport {
r_cc: 2.0,
r_nd: 1.0,
r_fo: 1.0,
r_ns: 1.0,
},
lrs: 4.8,
band: RiskBand::Moderate,
suppression_reason: None,
patterns: vec![],
pattern_details: None,
callees: vec![],
};
Snapshot::new(git_context, vec![report])
}
#[test]
fn test_snapshot_serialization() {
let snapshot = create_test_snapshot();
// Serialize
let json = snapshot.to_json().expect("should serialize");
assert!(json.contains("\"schema_version\": 2"));
assert!(json.contains("\"sha\": \"abc123\""));
assert!(json.contains("\"function_id\""));
let deserialized = Snapshot::from_json(&json).expect("should deserialize");
assert_eq!(deserialized.commit.sha, snapshot.commit.sha);
assert_eq!(deserialized.functions.len(), snapshot.functions.len());
}
#[test]
fn test_function_id_format() {
let snapshot = create_test_snapshot();
assert_eq!(snapshot.functions[0].function_id, "src/foo.ts::handler");
}
#[test]
fn test_snapshot_enricher_with_churn() {
use crate::git::FileChurn;
let snapshot = create_test_snapshot();
let mut churn_map = std::collections::HashMap::new();
churn_map.insert(
"src/foo.ts".to_string(),
FileChurn {
file: "src/foo.ts".to_string(),
lines_added: 10,
lines_deleted: 5,
},
);
let snapshot = SnapshotEnricher::new(snapshot)
.with_churn(&churn_map)
.build();
let churn = snapshot.functions[0]
.churn
.as_ref()
.expect("churn should be set");
assert_eq!(churn.lines_added, 10);
assert_eq!(churn.lines_deleted, 5);
assert_eq!(churn.net_change, 5);
}
#[test]
fn test_snapshot_enricher_enrich_computes_summary() {
let snapshot = create_test_snapshot();
let snapshot = SnapshotEnricher::new(snapshot).enrich(None, 75).build();
let summary = snapshot.summary.as_ref().expect("summary should be set");
assert_eq!(summary.total_functions, 1);
}
#[test]
fn test_snapshot_enricher_enrich_computes_percentiles() {
let snapshot = create_test_snapshot();
let snapshot = SnapshotEnricher::new(snapshot).enrich(None, 75).build();
assert!(snapshot.functions[0].percentile.is_some());
}
#[test]
fn test_snapshot_enricher_build_passthrough() {
let snapshot = create_test_snapshot();
let built = SnapshotEnricher::new(snapshot.clone()).build();
assert_eq!(
built.functions[0].function_id,
snapshot.functions[0].function_id
);
assert_eq!(built.commit.sha, snapshot.commit.sha);
}
#[test]
fn test_index_ordering() {
let mut index = Index::new();
index.add_commit(IndexEntry {
sha: "zzz".to_string(),
parents: vec![],
timestamp: 2000,
});
index.add_commit(IndexEntry {
sha: "aaa".to_string(),
parents: vec![],
timestamp: 1000,
});
index.add_commit(IndexEntry {
sha: "mmm".to_string(),
parents: vec![],
timestamp: 2000,
});
assert_eq!(index.commits[0].sha, "aaa");
assert_eq!(index.commits[1].sha, "mmm");
assert_eq!(index.commits[2].sha, "zzz");
}
fn test_cache_key() -> String {
crate::touch_cache::cache_key("abc123", "src/foo.ts", 42, 51)
}
#[test]
fn test_populate_touch_metrics_applies_cache_hits() {
let snapshot = create_test_snapshot();
let mut cache = crate::touch_cache::TouchCache::new();
cache.insert(test_cache_key(), (7, Some(3)));
let dir = tempfile::tempdir().unwrap();
crate::touch_cache::write_touch_cache(dir.path(), &cache).unwrap();
let mut snapshot = snapshot;
snapshot
.populate_touch_metrics(dir.path(), crate::snapshot::TouchMode::PerFunction, None)
.unwrap();
assert_eq!(snapshot.functions[0].touch_count_30d, Some(7));
assert_eq!(snapshot.functions[0].days_since_last_change, Some(3));
}
#[test]
fn test_populate_touch_metrics_progress_fires_on_all_cache_hits() {
use std::sync::{Arc, Mutex};
let snapshot = create_test_snapshot();
let mut cache = crate::touch_cache::TouchCache::new();
cache.insert(test_cache_key(), (3, Some(1)));
let dir = tempfile::tempdir().unwrap();
crate::touch_cache::write_touch_cache(dir.path(), &cache).unwrap();
let calls: Arc<Mutex<Vec<(usize, usize)>>> = Arc::new(Mutex::new(Vec::new()));
let calls_ref = calls.clone();
let mut snapshot = snapshot;
snapshot
.populate_touch_metrics(
dir.path(),
crate::snapshot::TouchMode::PerFunction,
Some(&|i, n| {
calls_ref.lock().unwrap().push((i, n));
}),
)
.unwrap();
let calls = calls.lock().unwrap();
assert!(!calls.is_empty(), "progress should have been called");
assert_eq!(*calls.last().unwrap(), (1, 1));
}
#[test]
fn test_populate_touch_metrics_progress_fires_per_miss() {
use std::sync::{Arc, Mutex};
let snapshot = create_test_snapshot();
let dir = tempfile::tempdir().unwrap();
let calls: Arc<Mutex<Vec<(usize, usize)>>> = Arc::new(Mutex::new(Vec::new()));
let calls_ref = calls.clone();
let mut snapshot = snapshot;
let _ = snapshot.populate_touch_metrics(
dir.path(),
crate::snapshot::TouchMode::PerFunction,
Some(&|i, n| {
calls_ref.lock().unwrap().push((i, n));
}),
);
let calls = calls.lock().unwrap();
assert!(
calls.len() >= 2,
"expected at least 2 progress calls, got {}",
calls.len()
);
assert_eq!(calls[0], (0, 1));
assert_eq!(*calls.last().unwrap(), (1, 1));
}
}