1use std::collections::{BTreeMap, HashMap};
2use std::fs;
3use std::path::{Component, Path, PathBuf};
4use std::sync::LazyLock;
5use std::time::Instant;
6
7use anyhow::{Context, Result, bail};
8use chrono::{SecondsFormat, Utc};
9use regex::Regex;
10use serde::{Deserialize, Serialize};
11use serde_json::{Value, json};
12use sha2::{Digest, Sha256};
13use tiktoken_rs::{
14 CoreBPE, cl100k_base, o200k_base, o200k_harmony, p50k_base, p50k_edit, r50k_base,
15};
16use unicode_normalization::UnicodeNormalization;
17use unicode_segmentation::UnicodeSegmentation;
18
19use crate::config;
20use crate::estimate;
21use crate::git;
22use crate::health;
23use crate::history;
24use crate::inventory;
25use crate::model::{Analysis, FileAnalysis, FindResult, ScopeIdentity};
26use crate::overlays;
27use crate::report;
28use crate::scoring;
29
30static CAMEL_CASE_RE: LazyLock<Regex> =
31 LazyLock::new(|| Regex::new(r"([a-z0-9])([A-Z])").expect("valid camel-case regex"));
32static ACRONYM_BOUNDARY_RE: LazyLock<Regex> =
33 LazyLock::new(|| Regex::new(r"([A-Z]+)([A-Z][a-z])").expect("valid acronym regex"));
34static NUMBER_RE: LazyLock<Regex> =
35 LazyLock::new(|| Regex::new(r"\b\d+(?:\.\d+)?\b").expect("valid number regex"));
36#[derive(Debug, Serialize, Deserialize)]
37struct CachedTokenData {
38 token_count: usize,
39 structural_tokens: Vec<String>,
40 content_fingerprint: String,
41}
42
43fn token_cache_key(text: &str, tokenizer: &str, large_file_bytes: usize, mode: &str) -> String {
44 let mut digest = Sha256::new();
45 digest.update(b"git-slop-token-cache-v3\0");
46 digest.update(tokenizer.as_bytes());
47 digest.update([0]);
48 digest.update(large_file_bytes.to_le_bytes());
49 digest.update(mode.as_bytes());
50 digest.update([0]);
51 digest.update(text.as_bytes());
52 hex::encode(digest.finalize())
53}
54
55fn load_cached_tokens(path: &Path) -> Option<CachedTokenData> {
56 serde_json::from_slice(&fs::read(path).ok()?).ok()
57}
58
59fn write_cached_tokens(path: &Path, cached: &CachedTokenData) -> Result<()> {
60 if let Some(parent) = path.parent() {
61 fs::create_dir_all(parent)?;
62 }
63 let temporary = path.with_extension(format!("json.tmp-{}", std::process::id()));
64 fs::write(&temporary, serde_json::to_vec(cached)?)?;
65 fs::rename(&temporary, path)?;
66 Ok(())
67}
68
69fn enforce_cache_limits(root: &Path, max_entries: usize, max_bytes: u64) -> Result<(usize, u64)> {
70 let mut entries = fs::read_dir(root)
71 .into_iter()
72 .flatten()
73 .filter_map(Result::ok)
74 .filter_map(|entry| {
75 let metadata = entry.metadata().ok()?;
76 metadata
77 .is_file()
78 .then_some((entry.path(), metadata.modified().ok(), metadata.len()))
79 })
80 .collect::<Vec<_>>();
81 entries.sort_by_key(|(path, modified, _)| (*modified, path.clone()));
82 let mut total_bytes = entries.iter().map(|entry| entry.2).sum::<u64>();
83 let mut total_entries = entries.len();
84 for (path, _, bytes) in entries {
85 if total_entries <= max_entries && total_bytes <= max_bytes {
86 break;
87 }
88 if fs::remove_file(&path).is_ok() {
89 total_entries = total_entries.saturating_sub(1);
90 total_bytes = total_bytes.saturating_sub(bytes);
91 }
92 }
93 Ok((total_entries, total_bytes))
94}
95
96fn replace_quoted_strings(text: &str) -> String {
97 let mut result = String::with_capacity(text.len());
98 let mut chars = text.chars().peekable();
99 let mut previous = None;
100 while let Some(character) = chars.next() {
101 if !matches!(character, '\'' | '"' | '`') {
102 result.push(character);
103 previous = Some(character);
104 continue;
105 }
106 if character == '\''
107 && previous.is_some_and(char::is_alphanumeric)
108 && chars.peek().is_some_and(|next| next.is_alphanumeric())
109 {
110 result.push(character);
111 previous = Some(character);
112 continue;
113 }
114 result.push_str(" str ");
115 let quote = character;
116 let mut escaped = false;
117 for next in chars.by_ref() {
118 if escaped {
119 escaped = false;
120 continue;
121 }
122 if next == '\\' {
123 escaped = true;
124 } else if next == quote {
125 break;
126 }
127 }
128 previous = Some(' ');
129 }
130 result
131}
132
133fn structural_mode(path: &str) -> &'static str {
134 if matches!(
135 Path::new(path).extension().and_then(|value| value.to_str()),
136 Some("md" | "mdx" | "txt")
137 ) {
138 "prose"
139 } else {
140 "code"
141 }
142}
143
144fn structural_content_tokens(mode: &str, text: &str) -> Vec<String> {
145 let normalized: String = text.nfkc().collect();
146 let normalized = normalized.replace(['\u{2018}', '\u{2019}'], "'");
147 let normalized = ACRONYM_BOUNDARY_RE.replace_all(&normalized, "$1 $2");
148 let normalized = CAMEL_CASE_RE.replace_all(&normalized, "$1 $2");
149 let normalized = normalized.replace(['-', '/'], " ");
150 let normalized = if mode == "prose" {
151 normalized
152 } else {
153 replace_quoted_strings(&normalized)
154 };
155 let normalized = NUMBER_RE.replace_all(&normalized, " 0 ");
156 let lower = normalized.to_lowercase();
157 lower
158 .unicode_words()
159 .flat_map(|word| word.split('_'))
160 .map(|word| {
161 word.split_once('\'')
162 .filter(|(prefix, suffix)| prefix.chars().count() == 1 && !suffix.is_empty())
163 .map_or(word, |(_, suffix)| suffix)
164 })
165 .filter(|item| item.chars().count() > 1)
166 .map(ToOwned::to_owned)
167 .collect()
168}
169
170fn structural_path_tokens(path: &str) -> Vec<String> {
171 path.replace(['-', '_', '.'], "/")
172 .to_ascii_lowercase()
173 .split('/')
174 .filter(|item| !item.is_empty())
175 .map(ToOwned::to_owned)
176 .collect()
177}
178
179#[cfg(test)]
180fn structural_tokens(path: &str, text: &str) -> Vec<String> {
181 let mut tokens = structural_content_tokens(structural_mode(path), text);
182 tokens.extend(structural_path_tokens(path));
183 tokens
184}
185
186fn content_fingerprint(text: &str) -> String {
187 hex::encode(Sha256::digest(text.as_bytes()))
188}
189
190fn top_terms(tokens: &[String], limit: usize) -> Vec<String> {
191 let mut counts: BTreeMap<&str, usize> = BTreeMap::new();
192 for token in tokens {
193 *counts.entry(token).or_default() += 1;
194 }
195 let mut ranked: Vec<(&str, usize)> = counts.into_iter().collect();
196 ranked.sort_by(|left, right| right.1.cmp(&left.1).then_with(|| left.0.cmp(right.0)));
197 ranked
198 .into_iter()
199 .take(limit)
200 .map(|(term, _)| term.to_string())
201 .collect()
202}
203
204fn has_inline_tests(language: &str, text: &str) -> bool {
205 match language {
206 "Rust" => text.contains("#[cfg(test)]") || text.contains("#[test]"),
207 "Go" => text.contains("func Test") || text.contains("func Benchmark"),
208 "Python" => text.contains("def test_") || text.contains("class Test"),
209 "JavaScript" | "JSX" | "TypeScript" | "TSX" => {
210 text.contains("describe(") || text.contains("test(") || text.contains("it(")
211 }
212 "Swift" => text.contains("XCTestCase") || text.contains("@Test"),
213 _ => false,
214 }
215}
216
217fn configured_context_encoder(config: &Value) -> Result<CoreBPE> {
218 let tokenizer_name = match config.pointer("/tokenization/context_tokenizer_name") {
219 Some(Value::String(name)) if !name.trim().is_empty() => name.as_str(),
220 Some(Value::String(_)) => {
221 bail!("tokenization.context_tokenizer_name must not be empty")
222 }
223 Some(_) => bail!("tokenization.context_tokenizer_name must be a string"),
224 None => "cl100k_base",
225 };
226 let encoder = match tokenizer_name {
227 "cl100k_base" => cl100k_base(),
228 "o200k_base" => o200k_base(),
229 "o200k_harmony" => o200k_harmony(),
230 "p50k_base" => p50k_base(),
231 "p50k_edit" => p50k_edit(),
232 "r50k_base" => r50k_base(),
233 unsupported => {
234 bail!(
235 "unsupported tokenization.context_tokenizer_name {unsupported:?}; \
236 supported encodings: cl100k_base, o200k_base, o200k_harmony, \
237 p50k_base, p50k_edit, r50k_base"
238 )
239 }
240 };
241 encoder.with_context(|| format!("failed to initialize {tokenizer_name} tokenizer"))
242}
243
244fn action_queue(files: &[FileAnalysis]) -> Vec<Value> {
245 let mut files: Vec<&FileAnalysis> = files.iter().collect();
246 files.sort_by(|left, right| {
247 right
248 .slop_score
249 .total_cmp(&left.slop_score)
250 .then_with(|| right.tokens.cmp(&left.tokens))
251 .then_with(|| left.path.cmp(&right.path))
252 });
253 files
254 .into_iter()
255 .map(|file| {
256 let non_context_reasons = file.reason_codes.iter().any(|reason| {
257 !matches!(reason.as_str(), "critical_token_cost" | "high_token_cost")
258 });
259 json!({
260 "path": file.path,
261 "slop_score": file.slop_score,
262 "slop_band": file.slop_band,
263 "context_band": file.context_band,
264 "tokens": file.tokens,
265 "age_days": file.age_days,
266 "revisions_window": file.revisions_window,
267 "churn_pressure": file.churn_pressure,
268 "reason_codes": file.reason_codes,
269 "is_pure_context_hotspot": !file.reason_codes.is_empty() && !non_context_reasons
270 })
271 })
272 .collect()
273}
274
275pub fn run_find() -> Result<FindResult> {
276 let repo_root = git::resolve_repo_root()?;
277 run_find_in(&repo_root)
278}
279
280pub fn run_find_in(repo_root: &Path) -> Result<FindResult> {
281 run_find_scoped(repo_root, false, None, false)
282}
283
284pub fn run_find_in_with_options(repo_root: &Path, allow_shallow: bool) -> Result<FindResult> {
285 run_find_scoped(repo_root, allow_shallow, None, false)
286}
287
288#[derive(Debug, Clone, Default)]
289pub struct FindOptions {
290 pub allow_shallow: bool,
291 pub scope: Option<String>,
292 pub progress: bool,
293 pub allow_empty_scope: bool,
294}
295
296fn normalize_scope(value: Option<&str>) -> Result<Option<String>> {
297 let Some(raw) = value.map(str::trim) else {
298 return Ok(None);
299 };
300 if raw.is_empty() || raw == "." {
301 return Ok(None);
302 }
303 let path = Path::new(raw);
304 if path.is_absolute() {
305 bail!("--scope must be repo-relative, received {raw:?}");
306 }
307 let mut parts = Vec::new();
308 for component in path.components() {
309 match component {
310 Component::Normal(part) => parts.push(
311 part.to_str()
312 .ok_or_else(|| anyhow::anyhow!("--scope must be valid UTF-8"))?,
313 ),
314 Component::CurDir => {}
315 Component::ParentDir | Component::RootDir | Component::Prefix(_) => {
316 bail!("--scope must not escape the repository, received {raw:?}");
317 }
318 }
319 }
320 let normalized = parts.join("/");
321 Ok((!normalized.is_empty()).then_some(normalized))
322}
323
324fn selected_path_digest(paths: &[String]) -> String {
325 let mut digest = Sha256::new();
326 for path in paths {
327 digest.update(path.as_bytes());
328 digest.update([0]);
329 }
330 hex::encode(digest.finalize())
331}
332
333fn selected_content_digest(repo_root: &Path, paths: &[String]) -> Result<String> {
334 let mut digest = Sha256::new();
335 for path in paths {
336 digest.update(path.as_bytes());
337 digest.update([0]);
338 let absolute = repo_root.join(path);
339 let metadata = fs::symlink_metadata(&absolute)
340 .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?;
341 let bytes = if metadata.file_type().is_symlink() {
342 fs::read_link(&absolute)
343 .with_context(|| format!("selected tracked link changed or disappeared: {path}"))?
344 .to_string_lossy()
345 .into_owned()
346 .into_bytes()
347 } else if metadata.is_dir() {
348 b"<gitlink>".to_vec()
349 } else {
350 fs::read(&absolute)
351 .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?
352 };
353 digest.update(bytes);
354 digest.update([0]);
355 }
356 Ok(hex::encode(digest.finalize()))
357}
358
359pub fn run_find_scoped(
360 repo_root: &Path,
361 allow_shallow: bool,
362 scope: Option<&str>,
363 progress: bool,
364) -> Result<FindResult> {
365 run_find_with_options(
366 repo_root,
367 &FindOptions {
368 allow_shallow,
369 scope: scope.map(ToOwned::to_owned),
370 progress,
371 allow_empty_scope: false,
372 },
373 )
374}
375
376pub fn run_find_with_options(repo_root: &Path, options: &FindOptions) -> Result<FindResult> {
377 let allow_shallow = options.allow_shallow;
378 let scope = options.scope.as_deref();
379 let progress = options.progress;
380 let allow_empty_scope = options.allow_empty_scope;
381 let started = Instant::now();
382 let phase = |name: &str| {
383 if progress {
384 eprintln!("git-slop: {name} ({:.1}s)", started.elapsed().as_secs_f64());
385 }
386 };
387 phase("preflight");
388 config::ensure_runtime_gitignore(repo_root)?;
389 let _scan_lock = config::acquire_scan_lock(repo_root)?;
390 let loaded_config = config::load(repo_root)?;
391 let mut repo = git::repo_metadata(repo_root)?;
392 if repo.is_shallow && !allow_shallow {
393 bail!(
394 "repository history is shallow; rerun with git slop find --allow-shallow to acknowledge incomplete history"
395 );
396 }
397 let all_tracked_paths = git::list_tracked_files(repo_root)?;
398 let scope = normalize_scope(scope)?;
399 if let Some(scope) = scope.as_deref() {
400 if !repo_root.join(scope).exists() {
401 bail!("--scope does not exist in the repository: {scope}");
402 }
403 }
404 let tracked_paths = all_tracked_paths
405 .iter()
406 .filter(|path| {
407 scope
408 .as_deref()
409 .is_none_or(|scope| *path == scope || path.starts_with(&format!("{scope}/")))
410 })
411 .cloned()
412 .collect::<Vec<_>>();
413 if tracked_paths.is_empty() && !allow_empty_scope {
414 bail!(
415 "{} selected no tracked paths; pass --allow-empty-scope only when an empty report is intentional",
416 scope.as_deref().map_or_else(
417 || "repository".to_string(),
418 |scope| format!("--scope {scope:?}")
419 )
420 );
421 }
422 let scope_identity = ScopeIdentity {
423 mode: if scope.is_some() {
424 "scoped"
425 } else {
426 "repository"
427 }
428 .to_string(),
429 path: scope.clone(),
430 selected_path_count: tracked_paths.len(),
431 selected_path_digest: selected_path_digest(&tracked_paths),
432 };
433 let starting_content_digest = selected_content_digest(repo_root, &tracked_paths)?;
434 let estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
435 if estimate.estimated_peak_memory_bytes > estimate.memory_budget_bytes {
436 bail!(
437 "analysis bounded before inventory: estimated {} MiB exceeds resources.memory_budget_mb={}; narrow --scope or raise the explicit budget",
438 estimate.estimated_peak_memory_bytes.div_ceil(1024 * 1024),
439 estimate.memory_budget_bytes / 1024 / 1024
440 );
441 }
442 let (inventory_files, skipped) = inventory::build(repo_root, &tracked_paths, &loaded_config)?;
443 phase("inventory");
444 let encoder = configured_context_encoder(&loaded_config)?;
445 let mut token_counts = BTreeMap::new();
446 let mut line_counts = BTreeMap::new();
447 let mut token_data = HashMap::new();
448 let tokenizer = config::pointer_str(&loaded_config, "/tokenization/context_tokenizer_name")
449 .unwrap_or("cl100k_base")
450 .to_string();
451 let large_file_bytes =
452 config::pointer_u64(&loaded_config, "/resources/large_file_bytes", 2_097_152) as usize;
453 let cache_root = config::cache_dir(repo_root).join("token-v3");
454 let mut cache_cleanup_warnings = Vec::new();
455 for version in ["token-v1", "token-v2"] {
456 let legacy_cache = config::cache_dir(repo_root).join(version);
457 if legacy_cache.exists() {
458 if let Err(error) = fs::remove_dir_all(&legacy_cache) {
459 cache_cleanup_warnings.push(format!(
460 "failed to remove legacy cache {}: {error}",
461 legacy_cache.display()
462 ));
463 }
464 }
465 }
466 if let Ok(entries) = fs::read_dir(&cache_root) {
467 for entry in entries.filter_map(Result::ok) {
468 if entry.file_name().to_string_lossy().contains(".tmp-") {
469 if let Err(error) = fs::remove_file(entry.path()) {
470 cache_cleanup_warnings.push(format!(
471 "failed to remove abandoned cache temporary {}: {error}",
472 entry.path().display()
473 ));
474 }
475 }
476 }
477 }
478 let mut cache_hits = 0usize;
479 let mut cache_misses = 0usize;
480 let mut structurally_skipped_large_files = 0usize;
481 for file in &inventory_files {
482 if file.bytes > large_file_bytes {
483 structurally_skipped_large_files += 1;
484 }
485 let mode = structural_mode(&file.path);
486 let cache_key = token_cache_key(&file.text, &tokenizer, large_file_bytes, mode);
487 let cache_path = cache_root.join(format!("{cache_key}.json"));
488 let cached = if let Some(cached) = load_cached_tokens(&cache_path) {
489 cache_hits += 1;
490 cached
491 } else {
492 cache_misses += 1;
493 let cached = CachedTokenData {
494 token_count: encoder.encode_ordinary(&file.text).len(),
495 structural_tokens: if file.bytes > large_file_bytes {
496 Vec::new()
497 } else {
498 structural_content_tokens(mode, &file.text)
499 },
500 content_fingerprint: content_fingerprint(&file.text),
501 };
502 write_cached_tokens(&cache_path, &cached)?;
503 cached
504 };
505 let count = cached.token_count;
506 let mut structural = cached.structural_tokens;
507 if file.bytes <= large_file_bytes {
508 structural.extend(structural_path_tokens(&file.path));
509 }
510 let top_term_limit =
511 config::pointer_u64(&loaded_config, "/semantic_drift/top_term_limit", 25) as usize;
512 let fingerprint = cached.content_fingerprint;
513 token_counts.insert(file.path.clone(), count);
514 line_counts.insert(file.path.clone(), file.lines);
515 token_data.insert(
516 file.path.clone(),
517 (
518 structural.len(),
519 top_terms(&structural, top_term_limit),
520 structural,
521 fingerprint,
522 ),
523 );
524 }
525 let (cache_entries, cache_bytes) = enforce_cache_limits(
526 &cache_root,
527 config::pointer_u64(&loaded_config, "/resources/cache_max_entries", 10_000) as usize,
528 config::pointer_u64(&loaded_config, "/resources/cache_max_bytes", 536_870_912),
529 )?;
530 phase("tokenization");
531 repo.analyzed_content_digest = Some(starting_content_digest.clone());
532 let analyzed_paths: Vec<String> = inventory_files
533 .iter()
534 .map(|file| file.path.clone())
535 .collect();
536 let now = Utc::now();
537 let (history_by_path, commits, history_diagnostics) = history::analyze_history(
538 repo_root,
539 &analyzed_paths,
540 &token_counts,
541 &line_counts,
542 &loaded_config,
543 now,
544 )?;
545 phase("history");
546 let mut files = Vec::with_capacity(inventory_files.len());
547 for file in inventory_files {
548 let tokens = token_counts.get(&file.path).copied().unwrap_or_default();
549 let inline_tests = has_inline_tests(&file.language, &file.text);
550 let (structural_token_count, top_structural_terms, structural_tokens, content_fingerprint) =
551 token_data
552 .remove(&file.path)
553 .unwrap_or_else(|| (0, Vec::new(), Vec::new(), String::new()));
554 let history = history_by_path.get(&file.path).cloned().unwrap_or_default();
555 files.push(FileAnalysis {
556 path: file.path,
557 bytes: file.bytes,
558 lines: file.lines,
559 blank_lines: file.blank_lines,
560 code_lines: file.code_lines,
561 comment_lines: file.comment_lines,
562 language: file.language,
563 profile: file.profile,
564 classification: file.classification,
565 has_inline_tests: inline_tests,
566 tokens,
567 context_band: scoring::context_band_for_tokens(tokens, &loaded_config),
568 context_pressure: scoring::context_pressure_for_tokens(tokens, &loaded_config),
569 content_fingerprint,
570 structural_tokens,
571 structural_token_count,
572 top_structural_terms,
573 age_days: history.age_days,
574 revisions_window: history.revisions_window,
575 recency_weighted_commits: history.recency_weighted_commits,
576 added_window: history.added_window,
577 deleted_window: history.deleted_window,
578 churn_lines_window: history.line_churn_window,
579 line_churn_window: history.line_churn_window,
580 token_churn_window: history.token_churn_window,
581 relative_churn_window: history.relative_churn_window,
582 late_churn_spike: history.late_churn_spike,
583 author_count_window: history.author_count_window,
584 author_entropy: history.author_entropy,
585 top_author_share: history.top_author_share,
586 days_since_non_bot_edit: history.days_since_non_bot_edit,
587 recent_maintainer_diversity: history.recent_maintainer_diversity,
588 age_pressure: 0.0,
589 revision_norm: 0.0,
590 relative_churn_norm: 0.0,
591 churn_pressure: 0.0,
592 slop_score: 0.0,
593 slop_band: "low".to_string(),
594 reason_codes: Vec::new(),
595 costs: json!({}),
596 overlays: json!({}),
597 });
598 }
599 scoring::apply_scoring(&mut files, &loaded_config);
600 let organization = overlays::analyze(&mut files, &commits, &loaded_config)?;
601 phase("relationships");
602 let folders = scoring::build_folder_analyses(&files, &loaded_config);
603 let queue = action_queue(&files);
604 let generated_at = now.to_rfc3339_opts(SecondsFormat::Secs, true);
605 let analyzed_revision_at = repo.head_commit_timestamp.clone();
606 let ending_worktree = git::worktree_state(repo_root)?;
607 if ending_worktree.digest != repo.worktree_state_digest {
608 bail!("repository changed during analysis; no mixed-snapshot report was published");
609 }
610 if selected_content_digest(repo_root, &tracked_paths)? != starting_content_digest {
611 bail!(
612 "selected file content changed during analysis; no mixed-snapshot report was published"
613 );
614 }
615 let analysis = Analysis {
616 repo_root: PathBuf::from(repo_root),
617 repo,
618 config: loaded_config,
619 generated_at,
620 analyzed_revision_at,
621 skipped,
622 tracked_file_count: all_tracked_paths.len(),
623 scope: scope_identity,
624 files,
625 folders,
626 organization,
627 action_queue: queue,
628 diagnostics: json!({
629 "analysis_elapsed_ms_before_report": started.elapsed().as_millis(),
630 "estimate": estimate,
631 "cache_hits": cache_hits,
632 "cache_misses": cache_misses,
633 "cache_entries": cache_entries,
634 "cache_bytes": cache_bytes,
635 "cache_cleanup_warnings": cache_cleanup_warnings,
636 "structurally_skipped_large_files": structurally_skipped_large_files,
637 "analysis_status": if structurally_skipped_large_files > 0 { "degraded_large_files" } else { "complete" },
638 "history": history_diagnostics,
639 "scope": scope
640 }),
641 };
642 let rollup = health::build_health_rollup(&analysis);
643 let result = report::write_report_bundle(&analysis, &rollup)?;
644 phase("report writing");
645 if result.report.get("schema_version").and_then(Value::as_u64) != Some(4) {
646 bail!("internal error: report writer did not produce schema 4");
647 }
648 Ok(result)
649}
650
651#[cfg(test)]
652mod tests {
653 use serde_json::json;
654 use tiktoken_rs::{cl100k_base, r50k_base};
655
656 use super::{
657 action_queue, configured_context_encoder, replace_quoted_strings, structural_tokens,
658 };
659 use crate::model::FileAnalysis;
660 use crate::scoring;
661
662 fn file(path: &str, relative_churn: f64) -> FileAnalysis {
663 FileAnalysis {
664 path: path.to_string(),
665 bytes: 400,
666 lines: 100,
667 blank_lines: 0,
668 code_lines: 100,
669 comment_lines: 0,
670 language: "Rust".to_string(),
671 profile: "agent_context".to_string(),
672 classification: "source".to_string(),
673 has_inline_tests: false,
674 tokens: 100,
675 context_band: "compact".to_string(),
676 context_pressure: 0.0,
677 content_fingerprint: String::new(),
678 structural_tokens: Vec::new(),
679 structural_token_count: 0,
680 top_structural_terms: Vec::new(),
681 age_days: 0,
682 revisions_window: 1,
683 recency_weighted_commits: 0.0,
684 added_window: 0,
685 deleted_window: 0,
686 churn_lines_window: 0,
687 line_churn_window: 0,
688 token_churn_window: 0,
689 relative_churn_window: relative_churn,
690 late_churn_spike: 0.0,
691 author_count_window: 0,
692 author_entropy: 0.0,
693 top_author_share: 0.0,
694 days_since_non_bot_edit: None,
695 recent_maintainer_diversity: 0,
696 age_pressure: 0.0,
697 revision_norm: 0.0,
698 relative_churn_norm: 0.0,
699 churn_pressure: 0.0,
700 slop_score: 0.0,
701 slop_band: String::new(),
702 reason_codes: Vec::new(),
703 costs: json!({}),
704 overlays: json!({}),
705 }
706 }
707
708 #[test]
709 fn structural_normalization_is_deterministic() {
710 let tokens = structural_tokens(
711 "src/my_file.rs",
712 "let camelCase = \"secret 123\"; // hello-world",
713 );
714 assert!(tokens.contains(&"camel".to_string()));
715 assert!(tokens.contains(&"case".to_string()));
716 assert!(tokens.contains(&"str".to_string()));
717 assert!(tokens.contains(&"my".to_string()));
718 assert_eq!(
719 replace_quoted_strings("'one' \"two\" `three`"),
720 " str str str "
721 );
722 }
723
724 #[test]
725 fn structural_normalization_preserves_unicode_and_apostrophe_words() {
726 let tokens = structural_tokens("docs/café.md", "L’équipe can’t rename HTTPServer_value");
727 assert!(tokens.contains(&"équipe".to_string()));
728 assert!(tokens.contains(&"can't".to_string()));
729 assert!(tokens.contains(&"http".to_string()));
730 assert!(tokens.contains(&"server".to_string()));
731 assert!(tokens.contains(&"value".to_string()));
732 assert!(tokens.iter().any(|token| token.contains("café")));
733 }
734
735 #[test]
736 fn configured_tokenizer_is_used_exactly_and_unknown_names_fail_closed() {
737 let text = "お誕生日おめでとう";
738 let configured = configured_context_encoder(&json!({"tokenization": {
739 "context_tokenizer_name": "r50k_base"
740 }}))
741 .unwrap();
742 assert_eq!(
743 configured.encode_ordinary(text).len(),
744 r50k_base().unwrap().encode_ordinary(text).len()
745 );
746 assert_ne!(
747 configured.encode_ordinary(text).len(),
748 cl100k_base().unwrap().encode_ordinary(text).len()
749 );
750
751 let result = configured_context_encoder(&json!({"tokenization": {
752 "context_tokenizer_name": "not-a-real-encoding"
753 }}));
754 let error = match result {
755 Ok(_) => panic!("unsupported tokenizer must fail closed"),
756 Err(error) => error,
757 };
758 assert!(
759 error
760 .to_string()
761 .contains("unsupported tokenization.context_tokenizer_name")
762 );
763 }
764
765 #[test]
766 fn action_queue_prioritizes_line_relative_churn_signal() {
767 let mut files = vec![file("src/quiet.rs", 0.1), file("src/volatile.rs", 2.0)];
768 scoring::apply_scoring(&mut files, &json!({}));
769 let queue = action_queue(&files);
770
771 assert_eq!(queue[0]["path"], "src/volatile.rs");
772 assert_eq!(queue[0]["reason_codes"][1], "high_relative_churn");
773 assert_eq!(queue[0]["is_pure_context_hotspot"], false);
774 }
775}