Skip to main content

git_slop/
analyze.rs

1use std::collections::{BTreeMap, HashMap};
2use std::fs;
3use std::path::{Component, Path, PathBuf};
4use std::sync::LazyLock;
5use std::time::Instant;
6
7use anyhow::{Context, Result, bail};
8use chrono::{SecondsFormat, Utc};
9use regex::Regex;
10use serde::{Deserialize, Serialize};
11use serde_json::{Value, json};
12use sha2::{Digest, Sha256};
13use tiktoken_rs::{
14    CoreBPE, cl100k_base, o200k_base, o200k_harmony, p50k_base, p50k_edit, r50k_base,
15};
16use unicode_normalization::UnicodeNormalization;
17use unicode_segmentation::UnicodeSegmentation;
18
19use crate::config;
20use crate::estimate;
21use crate::git;
22use crate::health;
23use crate::history;
24use crate::inventory;
25use crate::model::{Analysis, FileAnalysis, FindResult, ScopeIdentity};
26use crate::overlays;
27use crate::report;
28use crate::scoring;
29
30static CAMEL_CASE_RE: LazyLock<Regex> =
31    LazyLock::new(|| Regex::new(r"([a-z0-9])([A-Z])").expect("valid camel-case regex"));
32static ACRONYM_BOUNDARY_RE: LazyLock<Regex> =
33    LazyLock::new(|| Regex::new(r"([A-Z]+)([A-Z][a-z])").expect("valid acronym regex"));
34static NUMBER_RE: LazyLock<Regex> =
35    LazyLock::new(|| Regex::new(r"\b\d+(?:\.\d+)?\b").expect("valid number regex"));
36#[derive(Debug, Serialize, Deserialize)]
37struct CachedTokenData {
38    token_count: usize,
39    structural_tokens: Vec<String>,
40    content_fingerprint: String,
41}
42
43fn token_cache_key(text: &str, tokenizer: &str, large_file_bytes: usize, mode: &str) -> String {
44    let mut digest = Sha256::new();
45    digest.update(b"git-slop-token-cache-v3\0");
46    digest.update(tokenizer.as_bytes());
47    digest.update([0]);
48    digest.update(large_file_bytes.to_le_bytes());
49    digest.update(mode.as_bytes());
50    digest.update([0]);
51    digest.update(text.as_bytes());
52    hex::encode(digest.finalize())
53}
54
55fn load_cached_tokens(path: &Path) -> Option<CachedTokenData> {
56    serde_json::from_slice(&fs::read(path).ok()?).ok()
57}
58
59fn write_cached_tokens(path: &Path, cached: &CachedTokenData) -> Result<()> {
60    if let Some(parent) = path.parent() {
61        fs::create_dir_all(parent)?;
62    }
63    let temporary = path.with_extension(format!("json.tmp-{}", std::process::id()));
64    fs::write(&temporary, serde_json::to_vec(cached)?)?;
65    fs::rename(&temporary, path)?;
66    Ok(())
67}
68
69fn enforce_cache_limits(root: &Path, max_entries: usize, max_bytes: u64) -> Result<(usize, u64)> {
70    let mut entries = fs::read_dir(root)
71        .into_iter()
72        .flatten()
73        .filter_map(Result::ok)
74        .filter_map(|entry| {
75            let metadata = entry.metadata().ok()?;
76            metadata
77                .is_file()
78                .then_some((entry.path(), metadata.modified().ok(), metadata.len()))
79        })
80        .collect::<Vec<_>>();
81    entries.sort_by_key(|(path, modified, _)| (*modified, path.clone()));
82    let mut total_bytes = entries.iter().map(|entry| entry.2).sum::<u64>();
83    let mut total_entries = entries.len();
84    for (path, _, bytes) in entries {
85        if total_entries <= max_entries && total_bytes <= max_bytes {
86            break;
87        }
88        if fs::remove_file(&path).is_ok() {
89            total_entries = total_entries.saturating_sub(1);
90            total_bytes = total_bytes.saturating_sub(bytes);
91        }
92    }
93    Ok((total_entries, total_bytes))
94}
95
96fn replace_quoted_strings(text: &str) -> String {
97    let mut result = String::with_capacity(text.len());
98    let mut chars = text.chars().peekable();
99    let mut previous = None;
100    while let Some(character) = chars.next() {
101        if !matches!(character, '\'' | '"' | '`') {
102            result.push(character);
103            previous = Some(character);
104            continue;
105        }
106        if character == '\''
107            && previous.is_some_and(char::is_alphanumeric)
108            && chars.peek().is_some_and(|next| next.is_alphanumeric())
109        {
110            result.push(character);
111            previous = Some(character);
112            continue;
113        }
114        result.push_str(" str ");
115        let quote = character;
116        let mut escaped = false;
117        for next in chars.by_ref() {
118            if escaped {
119                escaped = false;
120                continue;
121            }
122            if next == '\\' {
123                escaped = true;
124            } else if next == quote {
125                break;
126            }
127        }
128        previous = Some(' ');
129    }
130    result
131}
132
133fn structural_mode(path: &str) -> &'static str {
134    if matches!(
135        Path::new(path).extension().and_then(|value| value.to_str()),
136        Some("md" | "mdx" | "txt")
137    ) {
138        "prose"
139    } else {
140        "code"
141    }
142}
143
144fn structural_content_tokens(mode: &str, text: &str) -> Vec<String> {
145    let normalized: String = text.nfkc().collect();
146    let normalized = normalized.replace(['\u{2018}', '\u{2019}'], "'");
147    let normalized = ACRONYM_BOUNDARY_RE.replace_all(&normalized, "$1 $2");
148    let normalized = CAMEL_CASE_RE.replace_all(&normalized, "$1 $2");
149    let normalized = normalized.replace(['-', '/'], " ");
150    let normalized = if mode == "prose" {
151        normalized
152    } else {
153        replace_quoted_strings(&normalized)
154    };
155    let normalized = NUMBER_RE.replace_all(&normalized, " 0 ");
156    let lower = normalized.to_lowercase();
157    lower
158        .unicode_words()
159        .flat_map(|word| word.split('_'))
160        .map(|word| {
161            word.split_once('\'')
162                .filter(|(prefix, suffix)| prefix.chars().count() == 1 && !suffix.is_empty())
163                .map_or(word, |(_, suffix)| suffix)
164        })
165        .filter(|item| item.chars().count() > 1)
166        .map(ToOwned::to_owned)
167        .collect()
168}
169
170fn structural_path_tokens(path: &str) -> Vec<String> {
171    path.replace(['-', '_', '.'], "/")
172        .to_ascii_lowercase()
173        .split('/')
174        .filter(|item| !item.is_empty())
175        .map(ToOwned::to_owned)
176        .collect()
177}
178
179#[cfg(test)]
180fn structural_tokens(path: &str, text: &str) -> Vec<String> {
181    let mut tokens = structural_content_tokens(structural_mode(path), text);
182    tokens.extend(structural_path_tokens(path));
183    tokens
184}
185
186fn content_fingerprint(text: &str) -> String {
187    hex::encode(Sha256::digest(text.as_bytes()))
188}
189
190fn top_terms(tokens: &[String], limit: usize) -> Vec<String> {
191    let mut counts: BTreeMap<&str, usize> = BTreeMap::new();
192    for token in tokens {
193        *counts.entry(token).or_default() += 1;
194    }
195    let mut ranked: Vec<(&str, usize)> = counts.into_iter().collect();
196    ranked.sort_by(|left, right| right.1.cmp(&left.1).then_with(|| left.0.cmp(right.0)));
197    ranked
198        .into_iter()
199        .take(limit)
200        .map(|(term, _)| term.to_string())
201        .collect()
202}
203
204fn has_inline_tests(language: &str, text: &str) -> bool {
205    match language {
206        "Rust" => text.contains("#[cfg(test)]") || text.contains("#[test]"),
207        "Go" => text.contains("func Test") || text.contains("func Benchmark"),
208        "Python" => text.contains("def test_") || text.contains("class Test"),
209        "JavaScript" | "JSX" | "TypeScript" | "TSX" => {
210            text.contains("describe(") || text.contains("test(") || text.contains("it(")
211        }
212        "Swift" => text.contains("XCTestCase") || text.contains("@Test"),
213        _ => false,
214    }
215}
216
217fn configured_context_encoder(config: &Value) -> Result<CoreBPE> {
218    let tokenizer_name = match config.pointer("/tokenization/context_tokenizer_name") {
219        Some(Value::String(name)) if !name.trim().is_empty() => name.as_str(),
220        Some(Value::String(_)) => {
221            bail!("tokenization.context_tokenizer_name must not be empty")
222        }
223        Some(_) => bail!("tokenization.context_tokenizer_name must be a string"),
224        None => "cl100k_base",
225    };
226    let encoder = match tokenizer_name {
227        "cl100k_base" => cl100k_base(),
228        "o200k_base" => o200k_base(),
229        "o200k_harmony" => o200k_harmony(),
230        "p50k_base" => p50k_base(),
231        "p50k_edit" => p50k_edit(),
232        "r50k_base" => r50k_base(),
233        unsupported => {
234            bail!(
235                "unsupported tokenization.context_tokenizer_name {unsupported:?}; \
236                 supported encodings: cl100k_base, o200k_base, o200k_harmony, \
237                 p50k_base, p50k_edit, r50k_base"
238            )
239        }
240    };
241    encoder.with_context(|| format!("failed to initialize {tokenizer_name} tokenizer"))
242}
243
244fn action_queue(files: &[FileAnalysis]) -> Vec<Value> {
245    let mut files: Vec<&FileAnalysis> = files.iter().collect();
246    files.sort_by(|left, right| {
247        right
248            .slop_score
249            .total_cmp(&left.slop_score)
250            .then_with(|| right.tokens.cmp(&left.tokens))
251            .then_with(|| left.path.cmp(&right.path))
252    });
253    files
254        .into_iter()
255        .map(|file| {
256            let non_context_reasons = file.reason_codes.iter().any(|reason| {
257                !matches!(reason.as_str(), "critical_token_cost" | "high_token_cost")
258            });
259            json!({
260                "path": file.path,
261                "slop_score": file.slop_score,
262                "slop_band": file.slop_band,
263                "context_band": file.context_band,
264                "tokens": file.tokens,
265                "age_days": file.age_days,
266                "revisions_window": file.revisions_window,
267                "churn_pressure": file.churn_pressure,
268                "reason_codes": file.reason_codes,
269                "is_pure_context_hotspot": !file.reason_codes.is_empty() && !non_context_reasons
270            })
271        })
272        .collect()
273}
274
275pub fn run_find() -> Result<FindResult> {
276    let repo_root = git::resolve_repo_root()?;
277    run_find_in(&repo_root)
278}
279
280pub fn run_find_in(repo_root: &Path) -> Result<FindResult> {
281    run_find_scoped(repo_root, false, None, false)
282}
283
284pub fn run_find_in_with_options(repo_root: &Path, allow_shallow: bool) -> Result<FindResult> {
285    run_find_scoped(repo_root, allow_shallow, None, false)
286}
287
288#[derive(Debug, Clone, Default)]
289pub struct FindOptions {
290    pub allow_shallow: bool,
291    pub scope: Option<String>,
292    pub progress: bool,
293    pub allow_empty_scope: bool,
294}
295
296fn normalize_scope(value: Option<&str>) -> Result<Option<String>> {
297    let Some(raw) = value.map(str::trim) else {
298        return Ok(None);
299    };
300    if raw.is_empty() || raw == "." {
301        return Ok(None);
302    }
303    let path = Path::new(raw);
304    if path.is_absolute() {
305        bail!("--scope must be repo-relative, received {raw:?}");
306    }
307    let mut parts = Vec::new();
308    for component in path.components() {
309        match component {
310            Component::Normal(part) => parts.push(
311                part.to_str()
312                    .ok_or_else(|| anyhow::anyhow!("--scope must be valid UTF-8"))?,
313            ),
314            Component::CurDir => {}
315            Component::ParentDir | Component::RootDir | Component::Prefix(_) => {
316                bail!("--scope must not escape the repository, received {raw:?}");
317            }
318        }
319    }
320    let normalized = parts.join("/");
321    Ok((!normalized.is_empty()).then_some(normalized))
322}
323
324fn selected_path_digest(paths: &[String]) -> String {
325    let mut digest = Sha256::new();
326    for path in paths {
327        digest.update(path.as_bytes());
328        digest.update([0]);
329    }
330    hex::encode(digest.finalize())
331}
332
333fn selected_content_digest(repo_root: &Path, paths: &[String]) -> Result<String> {
334    let mut digest = Sha256::new();
335    for path in paths {
336        digest.update(path.as_bytes());
337        digest.update([0]);
338        let absolute = repo_root.join(path);
339        let metadata = fs::symlink_metadata(&absolute)
340            .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?;
341        let bytes = if metadata.file_type().is_symlink() {
342            fs::read_link(&absolute)
343                .with_context(|| format!("selected tracked link changed or disappeared: {path}"))?
344                .to_string_lossy()
345                .into_owned()
346                .into_bytes()
347        } else if metadata.is_dir() {
348            b"<gitlink>".to_vec()
349        } else {
350            fs::read(&absolute)
351                .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?
352        };
353        digest.update(bytes);
354        digest.update([0]);
355    }
356    Ok(hex::encode(digest.finalize()))
357}
358
359pub fn run_find_scoped(
360    repo_root: &Path,
361    allow_shallow: bool,
362    scope: Option<&str>,
363    progress: bool,
364) -> Result<FindResult> {
365    run_find_with_options(
366        repo_root,
367        &FindOptions {
368            allow_shallow,
369            scope: scope.map(ToOwned::to_owned),
370            progress,
371            allow_empty_scope: false,
372        },
373    )
374}
375
376pub fn run_find_with_options(repo_root: &Path, options: &FindOptions) -> Result<FindResult> {
377    let allow_shallow = options.allow_shallow;
378    let scope = options.scope.as_deref();
379    let progress = options.progress;
380    let allow_empty_scope = options.allow_empty_scope;
381    let started = Instant::now();
382    let phase = |name: &str| {
383        if progress {
384            eprintln!("git-slop: {name} ({:.1}s)", started.elapsed().as_secs_f64());
385        }
386    };
387    phase("preflight");
388    config::ensure_runtime_gitignore(repo_root)?;
389    let _scan_lock = config::acquire_scan_lock(repo_root)?;
390    let loaded_config = config::load(repo_root)?;
391    let mut repo = git::repo_metadata(repo_root)?;
392    if repo.is_shallow && !allow_shallow {
393        bail!(
394            "repository history is shallow; rerun with git slop find --allow-shallow to acknowledge incomplete history"
395        );
396    }
397    let all_tracked_paths = git::list_tracked_files(repo_root)?;
398    let scope = normalize_scope(scope)?;
399    if let Some(scope) = scope.as_deref() {
400        if !repo_root.join(scope).exists() {
401            bail!("--scope does not exist in the repository: {scope}");
402        }
403    }
404    let tracked_paths = all_tracked_paths
405        .iter()
406        .filter(|path| {
407            scope
408                .as_deref()
409                .is_none_or(|scope| *path == scope || path.starts_with(&format!("{scope}/")))
410        })
411        .cloned()
412        .collect::<Vec<_>>();
413    if tracked_paths.is_empty() && !allow_empty_scope {
414        bail!(
415            "{} selected no tracked paths; pass --allow-empty-scope only when an empty report is intentional",
416            scope.as_deref().map_or_else(
417                || "repository".to_string(),
418                |scope| format!("--scope {scope:?}")
419            )
420        );
421    }
422    let scope_identity = ScopeIdentity {
423        mode: if scope.is_some() {
424            "scoped"
425        } else {
426            "repository"
427        }
428        .to_string(),
429        path: scope.clone(),
430        selected_path_count: tracked_paths.len(),
431        selected_path_digest: selected_path_digest(&tracked_paths),
432    };
433    let starting_content_digest = selected_content_digest(repo_root, &tracked_paths)?;
434    let estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
435    if estimate.estimated_peak_memory_bytes > estimate.memory_budget_bytes {
436        bail!(
437            "analysis bounded before inventory: estimated {} MiB exceeds resources.memory_budget_mb={}; narrow --scope or raise the explicit budget",
438            estimate.estimated_peak_memory_bytes.div_ceil(1024 * 1024),
439            estimate.memory_budget_bytes / 1024 / 1024
440        );
441    }
442    let (inventory_files, skipped) = inventory::build(repo_root, &tracked_paths, &loaded_config)?;
443    phase("inventory");
444    let encoder = configured_context_encoder(&loaded_config)?;
445    let mut token_counts = BTreeMap::new();
446    let mut line_counts = BTreeMap::new();
447    let mut token_data = HashMap::new();
448    let tokenizer = config::pointer_str(&loaded_config, "/tokenization/context_tokenizer_name")
449        .unwrap_or("cl100k_base")
450        .to_string();
451    let large_file_bytes =
452        config::pointer_u64(&loaded_config, "/resources/large_file_bytes", 2_097_152) as usize;
453    let cache_root = config::cache_dir(repo_root).join("token-v3");
454    let mut cache_cleanup_warnings = Vec::new();
455    for version in ["token-v1", "token-v2"] {
456        let legacy_cache = config::cache_dir(repo_root).join(version);
457        if legacy_cache.exists() {
458            if let Err(error) = fs::remove_dir_all(&legacy_cache) {
459                cache_cleanup_warnings.push(format!(
460                    "failed to remove legacy cache {}: {error}",
461                    legacy_cache.display()
462                ));
463            }
464        }
465    }
466    if let Ok(entries) = fs::read_dir(&cache_root) {
467        for entry in entries.filter_map(Result::ok) {
468            if entry.file_name().to_string_lossy().contains(".tmp-") {
469                if let Err(error) = fs::remove_file(entry.path()) {
470                    cache_cleanup_warnings.push(format!(
471                        "failed to remove abandoned cache temporary {}: {error}",
472                        entry.path().display()
473                    ));
474                }
475            }
476        }
477    }
478    let mut cache_hits = 0usize;
479    let mut cache_misses = 0usize;
480    let mut structurally_skipped_large_files = 0usize;
481    for file in &inventory_files {
482        if file.bytes > large_file_bytes {
483            structurally_skipped_large_files += 1;
484        }
485        let mode = structural_mode(&file.path);
486        let cache_key = token_cache_key(&file.text, &tokenizer, large_file_bytes, mode);
487        let cache_path = cache_root.join(format!("{cache_key}.json"));
488        let cached = if let Some(cached) = load_cached_tokens(&cache_path) {
489            cache_hits += 1;
490            cached
491        } else {
492            cache_misses += 1;
493            let cached = CachedTokenData {
494                token_count: encoder.encode_ordinary(&file.text).len(),
495                structural_tokens: if file.bytes > large_file_bytes {
496                    Vec::new()
497                } else {
498                    structural_content_tokens(mode, &file.text)
499                },
500                content_fingerprint: content_fingerprint(&file.text),
501            };
502            write_cached_tokens(&cache_path, &cached)?;
503            cached
504        };
505        let count = cached.token_count;
506        let mut structural = cached.structural_tokens;
507        if file.bytes <= large_file_bytes {
508            structural.extend(structural_path_tokens(&file.path));
509        }
510        let top_term_limit =
511            config::pointer_u64(&loaded_config, "/semantic_drift/top_term_limit", 25) as usize;
512        let fingerprint = cached.content_fingerprint;
513        token_counts.insert(file.path.clone(), count);
514        line_counts.insert(file.path.clone(), file.lines);
515        token_data.insert(
516            file.path.clone(),
517            (
518                structural.len(),
519                top_terms(&structural, top_term_limit),
520                structural,
521                fingerprint,
522            ),
523        );
524    }
525    let (cache_entries, cache_bytes) = enforce_cache_limits(
526        &cache_root,
527        config::pointer_u64(&loaded_config, "/resources/cache_max_entries", 10_000) as usize,
528        config::pointer_u64(&loaded_config, "/resources/cache_max_bytes", 536_870_912),
529    )?;
530    phase("tokenization");
531    repo.analyzed_content_digest = Some(starting_content_digest.clone());
532    let analyzed_paths: Vec<String> = inventory_files
533        .iter()
534        .map(|file| file.path.clone())
535        .collect();
536    let now = Utc::now();
537    let (history_by_path, commits, history_diagnostics) = history::analyze_history(
538        repo_root,
539        &analyzed_paths,
540        &token_counts,
541        &line_counts,
542        &loaded_config,
543        now,
544    )?;
545    phase("history");
546    let mut files = Vec::with_capacity(inventory_files.len());
547    for file in inventory_files {
548        let tokens = token_counts.get(&file.path).copied().unwrap_or_default();
549        let inline_tests = has_inline_tests(&file.language, &file.text);
550        let (structural_token_count, top_structural_terms, structural_tokens, content_fingerprint) =
551            token_data
552                .remove(&file.path)
553                .unwrap_or_else(|| (0, Vec::new(), Vec::new(), String::new()));
554        let history = history_by_path.get(&file.path).cloned().unwrap_or_default();
555        files.push(FileAnalysis {
556            path: file.path,
557            bytes: file.bytes,
558            lines: file.lines,
559            blank_lines: file.blank_lines,
560            code_lines: file.code_lines,
561            comment_lines: file.comment_lines,
562            language: file.language,
563            profile: file.profile,
564            classification: file.classification,
565            has_inline_tests: inline_tests,
566            tokens,
567            context_band: scoring::context_band_for_tokens(tokens, &loaded_config),
568            context_pressure: scoring::context_pressure_for_tokens(tokens, &loaded_config),
569            content_fingerprint,
570            structural_tokens,
571            structural_token_count,
572            top_structural_terms,
573            age_days: history.age_days,
574            revisions_window: history.revisions_window,
575            recency_weighted_commits: history.recency_weighted_commits,
576            added_window: history.added_window,
577            deleted_window: history.deleted_window,
578            churn_lines_window: history.line_churn_window,
579            line_churn_window: history.line_churn_window,
580            token_churn_window: history.token_churn_window,
581            relative_churn_window: history.relative_churn_window,
582            late_churn_spike: history.late_churn_spike,
583            author_count_window: history.author_count_window,
584            author_entropy: history.author_entropy,
585            top_author_share: history.top_author_share,
586            days_since_non_bot_edit: history.days_since_non_bot_edit,
587            recent_maintainer_diversity: history.recent_maintainer_diversity,
588            age_pressure: 0.0,
589            revision_norm: 0.0,
590            relative_churn_norm: 0.0,
591            churn_pressure: 0.0,
592            slop_score: 0.0,
593            slop_band: "low".to_string(),
594            reason_codes: Vec::new(),
595            costs: json!({}),
596            overlays: json!({}),
597        });
598    }
599    scoring::apply_scoring(&mut files, &loaded_config);
600    let organization = overlays::analyze(&mut files, &commits, &loaded_config)?;
601    phase("relationships");
602    let folders = scoring::build_folder_analyses(&files, &loaded_config);
603    let queue = action_queue(&files);
604    let generated_at = now.to_rfc3339_opts(SecondsFormat::Secs, true);
605    let analyzed_revision_at = repo.head_commit_timestamp.clone();
606    let ending_worktree = git::worktree_state(repo_root)?;
607    if ending_worktree.digest != repo.worktree_state_digest {
608        bail!("repository changed during analysis; no mixed-snapshot report was published");
609    }
610    if selected_content_digest(repo_root, &tracked_paths)? != starting_content_digest {
611        bail!(
612            "selected file content changed during analysis; no mixed-snapshot report was published"
613        );
614    }
615    let analysis = Analysis {
616        repo_root: PathBuf::from(repo_root),
617        repo,
618        config: loaded_config,
619        generated_at,
620        analyzed_revision_at,
621        skipped,
622        tracked_file_count: all_tracked_paths.len(),
623        scope: scope_identity,
624        files,
625        folders,
626        organization,
627        action_queue: queue,
628        diagnostics: json!({
629            "analysis_elapsed_ms_before_report": started.elapsed().as_millis(),
630            "estimate": estimate,
631            "cache_hits": cache_hits,
632            "cache_misses": cache_misses,
633            "cache_entries": cache_entries,
634            "cache_bytes": cache_bytes,
635            "cache_cleanup_warnings": cache_cleanup_warnings,
636            "structurally_skipped_large_files": structurally_skipped_large_files,
637            "analysis_status": if structurally_skipped_large_files > 0 { "degraded_large_files" } else { "complete" },
638            "history": history_diagnostics,
639            "scope": scope
640        }),
641    };
642    let rollup = health::build_health_rollup(&analysis);
643    let result = report::write_report_bundle(&analysis, &rollup)?;
644    phase("report writing");
645    if result.report.get("schema_version").and_then(Value::as_u64) != Some(4) {
646        bail!("internal error: report writer did not produce schema 4");
647    }
648    Ok(result)
649}
650
651#[cfg(test)]
652mod tests {
653    use serde_json::json;
654    use tiktoken_rs::{cl100k_base, r50k_base};
655
656    use super::{
657        action_queue, configured_context_encoder, replace_quoted_strings, structural_tokens,
658    };
659    use crate::model::FileAnalysis;
660    use crate::scoring;
661
662    fn file(path: &str, relative_churn: f64) -> FileAnalysis {
663        FileAnalysis {
664            path: path.to_string(),
665            bytes: 400,
666            lines: 100,
667            blank_lines: 0,
668            code_lines: 100,
669            comment_lines: 0,
670            language: "Rust".to_string(),
671            profile: "agent_context".to_string(),
672            classification: "source".to_string(),
673            has_inline_tests: false,
674            tokens: 100,
675            context_band: "compact".to_string(),
676            context_pressure: 0.0,
677            content_fingerprint: String::new(),
678            structural_tokens: Vec::new(),
679            structural_token_count: 0,
680            top_structural_terms: Vec::new(),
681            age_days: 0,
682            revisions_window: 1,
683            recency_weighted_commits: 0.0,
684            added_window: 0,
685            deleted_window: 0,
686            churn_lines_window: 0,
687            line_churn_window: 0,
688            token_churn_window: 0,
689            relative_churn_window: relative_churn,
690            late_churn_spike: 0.0,
691            author_count_window: 0,
692            author_entropy: 0.0,
693            top_author_share: 0.0,
694            days_since_non_bot_edit: None,
695            recent_maintainer_diversity: 0,
696            age_pressure: 0.0,
697            revision_norm: 0.0,
698            relative_churn_norm: 0.0,
699            churn_pressure: 0.0,
700            slop_score: 0.0,
701            slop_band: String::new(),
702            reason_codes: Vec::new(),
703            costs: json!({}),
704            overlays: json!({}),
705        }
706    }
707
708    #[test]
709    fn structural_normalization_is_deterministic() {
710        let tokens = structural_tokens(
711            "src/my_file.rs",
712            "let camelCase = \"secret 123\"; // hello-world",
713        );
714        assert!(tokens.contains(&"camel".to_string()));
715        assert!(tokens.contains(&"case".to_string()));
716        assert!(tokens.contains(&"str".to_string()));
717        assert!(tokens.contains(&"my".to_string()));
718        assert_eq!(
719            replace_quoted_strings("'one' \"two\" `three`"),
720            " str   str   str "
721        );
722    }
723
724    #[test]
725    fn structural_normalization_preserves_unicode_and_apostrophe_words() {
726        let tokens = structural_tokens("docs/café.md", "L’équipe can’t rename HTTPServer_value");
727        assert!(tokens.contains(&"équipe".to_string()));
728        assert!(tokens.contains(&"can't".to_string()));
729        assert!(tokens.contains(&"http".to_string()));
730        assert!(tokens.contains(&"server".to_string()));
731        assert!(tokens.contains(&"value".to_string()));
732        assert!(tokens.iter().any(|token| token.contains("café")));
733    }
734
735    #[test]
736    fn configured_tokenizer_is_used_exactly_and_unknown_names_fail_closed() {
737        let text = "お誕生日おめでとう";
738        let configured = configured_context_encoder(&json!({"tokenization": {
739            "context_tokenizer_name": "r50k_base"
740        }}))
741        .unwrap();
742        assert_eq!(
743            configured.encode_ordinary(text).len(),
744            r50k_base().unwrap().encode_ordinary(text).len()
745        );
746        assert_ne!(
747            configured.encode_ordinary(text).len(),
748            cl100k_base().unwrap().encode_ordinary(text).len()
749        );
750
751        let result = configured_context_encoder(&json!({"tokenization": {
752            "context_tokenizer_name": "not-a-real-encoding"
753        }}));
754        let error = match result {
755            Ok(_) => panic!("unsupported tokenizer must fail closed"),
756            Err(error) => error,
757        };
758        assert!(
759            error
760                .to_string()
761                .contains("unsupported tokenization.context_tokenizer_name")
762        );
763    }
764
765    #[test]
766    fn action_queue_prioritizes_line_relative_churn_signal() {
767        let mut files = vec![file("src/quiet.rs", 0.1), file("src/volatile.rs", 2.0)];
768        scoring::apply_scoring(&mut files, &json!({}));
769        let queue = action_queue(&files);
770
771        assert_eq!(queue[0]["path"], "src/volatile.rs");
772        assert_eq!(queue[0]["reason_codes"][1], "high_relative_churn");
773        assert_eq!(queue[0]["is_pure_context_hotspot"], false);
774    }
775}