Skip to main content

git_slop/
analyze.rs

1use std::collections::{BTreeMap, HashMap};
2use std::path::{Path, PathBuf};
3use std::sync::LazyLock;
4
5use anyhow::{Context, Result, bail};
6use chrono::{SecondsFormat, Utc};
7use regex::Regex;
8use serde_json::{Value, json};
9use sha2::{Digest, Sha256};
10use tiktoken_rs::{
11    CoreBPE, cl100k_base, o200k_base, o200k_harmony, p50k_base, p50k_edit, r50k_base,
12};
13use unicode_normalization::UnicodeNormalization;
14
15use crate::config;
16use crate::git;
17use crate::health;
18use crate::history;
19use crate::inventory;
20use crate::model::{Analysis, FileAnalysis, FindResult};
21use crate::overlays;
22use crate::report;
23use crate::scoring;
24
25static CAMEL_CASE_RE: LazyLock<Regex> =
26    LazyLock::new(|| Regex::new(r"([a-z0-9])([A-Z])").expect("valid camel-case regex"));
27static NUMBER_RE: LazyLock<Regex> =
28    LazyLock::new(|| Regex::new(r"\b\d+(?:\.\d+)?\b").expect("valid number regex"));
29static WORD_RE: LazyLock<Regex> =
30    LazyLock::new(|| Regex::new(r"[a-z][a-z0-9_]{1,}").expect("valid word regex"));
31
32fn replace_quoted_strings(text: &str) -> String {
33    let mut result = String::with_capacity(text.len());
34    let mut chars = text.chars().peekable();
35    while let Some(character) = chars.next() {
36        if !matches!(character, '\'' | '"' | '`') {
37            result.push(character);
38            continue;
39        }
40        result.push_str(" str ");
41        let quote = character;
42        let mut escaped = false;
43        for next in chars.by_ref() {
44            if escaped {
45                escaped = false;
46                continue;
47            }
48            if next == '\\' {
49                escaped = true;
50            } else if next == quote {
51                break;
52            }
53        }
54    }
55    result
56}
57
58fn structural_tokens(path: &str, text: &str) -> Vec<String> {
59    let normalized: String = text.nfkc().collect();
60    let normalized = CAMEL_CASE_RE.replace_all(&normalized, "$1 $2");
61    let normalized = normalized.replace(['-', '/'], " ");
62    let normalized = replace_quoted_strings(&normalized);
63    let normalized = NUMBER_RE.replace_all(&normalized, " 0 ");
64    let lower = normalized.to_ascii_lowercase();
65    let mut tokens: Vec<String> = WORD_RE
66        .find_iter(&lower)
67        .map(|item| item.as_str().to_string())
68        .collect();
69    tokens.extend(
70        path.replace(['-', '_', '.'], "/")
71            .to_ascii_lowercase()
72            .split('/')
73            .filter(|item| !item.is_empty())
74            .map(ToOwned::to_owned),
75    );
76    tokens
77}
78
79fn content_fingerprint(text: &str) -> String {
80    hex::encode(Sha256::digest(text.as_bytes()))
81}
82
83fn top_terms(tokens: &[String]) -> Vec<String> {
84    let mut counts: BTreeMap<&str, usize> = BTreeMap::new();
85    for token in tokens {
86        *counts.entry(token).or_default() += 1;
87    }
88    let mut ranked: Vec<(&str, usize)> = counts.into_iter().collect();
89    ranked.sort_by(|left, right| right.1.cmp(&left.1).then_with(|| left.0.cmp(right.0)));
90    ranked
91        .into_iter()
92        .take(12)
93        .map(|(term, _)| term.to_string())
94        .collect()
95}
96
97fn configured_context_encoder(config: &Value) -> Result<CoreBPE> {
98    let tokenizer_name = match config.pointer("/tokenization/context_tokenizer_name") {
99        Some(Value::String(name)) if !name.trim().is_empty() => name.as_str(),
100        Some(Value::String(_)) => {
101            bail!("tokenization.context_tokenizer_name must not be empty")
102        }
103        Some(_) => bail!("tokenization.context_tokenizer_name must be a string"),
104        None => "cl100k_base",
105    };
106    let encoder = match tokenizer_name {
107        "cl100k_base" => cl100k_base(),
108        "o200k_base" => o200k_base(),
109        "o200k_harmony" => o200k_harmony(),
110        "p50k_base" => p50k_base(),
111        "p50k_edit" => p50k_edit(),
112        "r50k_base" => r50k_base(),
113        unsupported => {
114            bail!(
115                "unsupported tokenization.context_tokenizer_name {unsupported:?}; \
116                 supported encodings: cl100k_base, o200k_base, o200k_harmony, \
117                 p50k_base, p50k_edit, r50k_base"
118            )
119        }
120    };
121    encoder.with_context(|| format!("failed to initialize {tokenizer_name} tokenizer"))
122}
123
124fn action_queue(files: &[FileAnalysis]) -> Vec<Value> {
125    let mut files: Vec<&FileAnalysis> = files.iter().collect();
126    files.sort_by(|left, right| {
127        right
128            .slop_score
129            .total_cmp(&left.slop_score)
130            .then_with(|| right.tokens.cmp(&left.tokens))
131            .then_with(|| left.path.cmp(&right.path))
132    });
133    files
134        .into_iter()
135        .take(25)
136        .map(|file| {
137            let non_context_reasons = file.reason_codes.iter().any(|reason| {
138                !matches!(reason.as_str(), "critical_token_cost" | "high_token_cost")
139            });
140            json!({
141                "path": file.path,
142                "slop_score": file.slop_score,
143                "slop_band": file.slop_band,
144                "context_band": file.context_band,
145                "tokens": file.tokens,
146                "age_days": file.age_days,
147                "revisions_window": file.revisions_window,
148                "churn_pressure": file.churn_pressure,
149                "reason_codes": file.reason_codes,
150                "is_pure_context_hotspot": !file.reason_codes.is_empty() && !non_context_reasons
151            })
152        })
153        .collect()
154}
155
156pub fn run_find() -> Result<FindResult> {
157    let repo_root = git::resolve_repo_root()?;
158    run_find_in(&repo_root)
159}
160
161pub fn run_find_in(repo_root: &Path) -> Result<FindResult> {
162    config::ensure_state_dirs(repo_root)?;
163    let loaded_config = config::load(repo_root)?;
164    let repo = git::repo_metadata(repo_root)?;
165    let tracked_paths = git::list_tracked_files(repo_root)?;
166    let (inventory_files, skipped) = inventory::build(repo_root, &tracked_paths, &loaded_config)?;
167    let encoder = configured_context_encoder(&loaded_config)?;
168    let mut token_counts = BTreeMap::new();
169    let mut line_counts = BTreeMap::new();
170    let mut token_data = HashMap::new();
171    for file in &inventory_files {
172        let count = encoder.encode_ordinary(&file.text).len();
173        let structural = structural_tokens(&file.path, &file.text);
174        let fingerprint = content_fingerprint(&file.text);
175        token_counts.insert(file.path.clone(), count);
176        line_counts.insert(file.path.clone(), file.lines);
177        token_data.insert(
178            file.path.clone(),
179            (
180                structural.len(),
181                top_terms(&structural),
182                structural,
183                fingerprint,
184            ),
185        );
186    }
187    let analyzed_paths: Vec<String> = inventory_files
188        .iter()
189        .map(|file| file.path.clone())
190        .collect();
191    let now = Utc::now();
192    let (history_by_path, commits, _repo_baselines) = history::analyze_history(
193        repo_root,
194        &analyzed_paths,
195        &token_counts,
196        &line_counts,
197        &loaded_config,
198        now,
199    )?;
200    let mut files = Vec::with_capacity(inventory_files.len());
201    for file in inventory_files {
202        let tokens = token_counts.get(&file.path).copied().unwrap_or_default();
203        let (structural_token_count, top_structural_terms, structural_tokens, content_fingerprint) =
204            token_data
205                .remove(&file.path)
206                .unwrap_or_else(|| (0, Vec::new(), Vec::new(), String::new()));
207        let history = history_by_path.get(&file.path).cloned().unwrap_or_default();
208        files.push(FileAnalysis {
209            path: file.path,
210            bytes: file.bytes,
211            lines: file.lines,
212            blank_lines: file.blank_lines,
213            code_lines: file.code_lines,
214            comment_lines: file.comment_lines,
215            language: file.language,
216            profile: file.profile,
217            classification: file.classification,
218            tokens,
219            context_band: scoring::context_band_for_tokens(tokens, &loaded_config),
220            context_pressure: scoring::context_pressure_for_tokens(tokens, &loaded_config),
221            content_fingerprint,
222            structural_tokens,
223            structural_token_count,
224            top_structural_terms,
225            age_days: history.age_days,
226            revisions_window: history.revisions_window,
227            recency_weighted_commits: history.recency_weighted_commits,
228            added_window: history.added_window,
229            deleted_window: history.deleted_window,
230            churn_lines_window: history.line_churn_window,
231            line_churn_window: history.line_churn_window,
232            token_churn_window: history.token_churn_window,
233            relative_churn_window: history.relative_churn_window,
234            late_churn_spike: history.late_churn_spike,
235            author_count_window: history.author_count_window,
236            author_entropy: history.author_entropy,
237            top_author_share: history.top_author_share,
238            days_since_non_bot_edit: history.days_since_non_bot_edit,
239            recent_maintainer_diversity: history.recent_maintainer_diversity,
240            age_pressure: 0.0,
241            revision_norm: 0.0,
242            relative_churn_norm: 0.0,
243            churn_pressure: 0.0,
244            slop_score: 0.0,
245            slop_band: "low".to_string(),
246            reason_codes: Vec::new(),
247            costs: json!({}),
248            overlays: json!({}),
249        });
250    }
251    scoring::apply_scoring(&mut files, &loaded_config);
252    let organization = overlays::analyze(&mut files, &commits, &loaded_config)?;
253    let folders = scoring::build_folder_analyses(&files, &loaded_config);
254    let queue = action_queue(&files);
255    let generated_at = now.to_rfc3339_opts(SecondsFormat::Secs, true);
256    let analyzed_revision_at = repo.head_commit_timestamp.clone();
257    let analysis = Analysis {
258        repo_root: PathBuf::from(repo_root),
259        repo,
260        config: loaded_config,
261        generated_at,
262        analyzed_revision_at,
263        skipped,
264        tracked_file_count: tracked_paths.len(),
265        files,
266        folders,
267        commits,
268        organization,
269        action_queue: queue,
270        report: Value::Null,
271    };
272    let rollup = health::build_health_rollup(&analysis);
273    let result = report::write_report_bundle(&analysis, &rollup)?;
274    if result.report.get("schema_version").and_then(Value::as_u64) != Some(4) {
275        bail!("internal error: report writer did not produce schema 4");
276    }
277    Ok(result)
278}
279
280#[cfg(test)]
281mod tests {
282    use serde_json::json;
283    use tiktoken_rs::{cl100k_base, r50k_base};
284
285    use super::{
286        action_queue, configured_context_encoder, replace_quoted_strings, structural_tokens,
287    };
288    use crate::model::FileAnalysis;
289    use crate::scoring;
290
291    fn file(path: &str, relative_churn: f64) -> FileAnalysis {
292        FileAnalysis {
293            path: path.to_string(),
294            bytes: 400,
295            lines: 100,
296            blank_lines: 0,
297            code_lines: 100,
298            comment_lines: 0,
299            language: "Rust".to_string(),
300            profile: "agent_context".to_string(),
301            classification: "source".to_string(),
302            tokens: 100,
303            context_band: "compact".to_string(),
304            context_pressure: 0.0,
305            content_fingerprint: String::new(),
306            structural_tokens: Vec::new(),
307            structural_token_count: 0,
308            top_structural_terms: Vec::new(),
309            age_days: 0,
310            revisions_window: 1,
311            recency_weighted_commits: 0.0,
312            added_window: 0,
313            deleted_window: 0,
314            churn_lines_window: 0,
315            line_churn_window: 0,
316            token_churn_window: 0,
317            relative_churn_window: relative_churn,
318            late_churn_spike: 0.0,
319            author_count_window: 0,
320            author_entropy: 0.0,
321            top_author_share: 0.0,
322            days_since_non_bot_edit: None,
323            recent_maintainer_diversity: 0,
324            age_pressure: 0.0,
325            revision_norm: 0.0,
326            relative_churn_norm: 0.0,
327            churn_pressure: 0.0,
328            slop_score: 0.0,
329            slop_band: String::new(),
330            reason_codes: Vec::new(),
331            costs: json!({}),
332            overlays: json!({}),
333        }
334    }
335
336    #[test]
337    fn structural_normalization_is_deterministic() {
338        let tokens = structural_tokens(
339            "src/my_file.rs",
340            "let camelCase = \"secret 123\"; // hello-world",
341        );
342        assert!(tokens.contains(&"camel".to_string()));
343        assert!(tokens.contains(&"case".to_string()));
344        assert!(tokens.contains(&"str".to_string()));
345        assert!(tokens.contains(&"my".to_string()));
346        assert_eq!(
347            replace_quoted_strings("'one' \"two\" `three`"),
348            " str   str   str "
349        );
350    }
351
352    #[test]
353    fn configured_tokenizer_is_used_exactly_and_unknown_names_fail_closed() {
354        let text = "お誕生日おめでとう";
355        let configured = configured_context_encoder(&json!({"tokenization": {
356            "context_tokenizer_name": "r50k_base"
357        }}))
358        .unwrap();
359        assert_eq!(
360            configured.encode_ordinary(text).len(),
361            r50k_base().unwrap().encode_ordinary(text).len()
362        );
363        assert_ne!(
364            configured.encode_ordinary(text).len(),
365            cl100k_base().unwrap().encode_ordinary(text).len()
366        );
367
368        let result = configured_context_encoder(&json!({"tokenization": {
369            "context_tokenizer_name": "not-a-real-encoding"
370        }}));
371        let error = match result {
372            Ok(_) => panic!("unsupported tokenizer must fail closed"),
373            Err(error) => error,
374        };
375        assert!(
376            error
377                .to_string()
378                .contains("unsupported tokenization.context_tokenizer_name")
379        );
380    }
381
382    #[test]
383    fn action_queue_prioritizes_line_relative_churn_signal() {
384        let mut files = vec![file("src/quiet.rs", 0.1), file("src/volatile.rs", 2.0)];
385        scoring::apply_scoring(&mut files, &json!({}));
386        let queue = action_queue(&files);
387
388        assert_eq!(queue[0]["path"], "src/volatile.rs");
389        assert_eq!(queue[0]["reason_codes"][1], "high_relative_churn");
390        assert_eq!(queue[0]["is_pure_context_hotspot"], false);
391    }
392}