1use std::collections::{BTreeMap, HashMap};
2use std::path::{Path, PathBuf};
3use std::sync::LazyLock;
4
5use anyhow::{Context, Result, bail};
6use chrono::{SecondsFormat, Utc};
7use regex::Regex;
8use serde_json::{Value, json};
9use sha2::{Digest, Sha256};
10use tiktoken_rs::{
11 CoreBPE, cl100k_base, o200k_base, o200k_harmony, p50k_base, p50k_edit, r50k_base,
12};
13use unicode_normalization::UnicodeNormalization;
14
15use crate::config;
16use crate::git;
17use crate::health;
18use crate::history;
19use crate::inventory;
20use crate::model::{Analysis, FileAnalysis, FindResult};
21use crate::overlays;
22use crate::report;
23use crate::scoring;
24
25static CAMEL_CASE_RE: LazyLock<Regex> =
26 LazyLock::new(|| Regex::new(r"([a-z0-9])([A-Z])").expect("valid camel-case regex"));
27static NUMBER_RE: LazyLock<Regex> =
28 LazyLock::new(|| Regex::new(r"\b\d+(?:\.\d+)?\b").expect("valid number regex"));
29static WORD_RE: LazyLock<Regex> =
30 LazyLock::new(|| Regex::new(r"[a-z][a-z0-9_]{1,}").expect("valid word regex"));
31
32fn replace_quoted_strings(text: &str) -> String {
33 let mut result = String::with_capacity(text.len());
34 let mut chars = text.chars().peekable();
35 while let Some(character) = chars.next() {
36 if !matches!(character, '\'' | '"' | '`') {
37 result.push(character);
38 continue;
39 }
40 result.push_str(" str ");
41 let quote = character;
42 let mut escaped = false;
43 for next in chars.by_ref() {
44 if escaped {
45 escaped = false;
46 continue;
47 }
48 if next == '\\' {
49 escaped = true;
50 } else if next == quote {
51 break;
52 }
53 }
54 }
55 result
56}
57
58fn structural_tokens(path: &str, text: &str) -> Vec<String> {
59 let normalized: String = text.nfkc().collect();
60 let normalized = CAMEL_CASE_RE.replace_all(&normalized, "$1 $2");
61 let normalized = normalized.replace(['-', '/'], " ");
62 let normalized = replace_quoted_strings(&normalized);
63 let normalized = NUMBER_RE.replace_all(&normalized, " 0 ");
64 let lower = normalized.to_ascii_lowercase();
65 let mut tokens: Vec<String> = WORD_RE
66 .find_iter(&lower)
67 .map(|item| item.as_str().to_string())
68 .collect();
69 tokens.extend(
70 path.replace(['-', '_', '.'], "/")
71 .to_ascii_lowercase()
72 .split('/')
73 .filter(|item| !item.is_empty())
74 .map(ToOwned::to_owned),
75 );
76 tokens
77}
78
79fn content_fingerprint(text: &str) -> String {
80 hex::encode(Sha256::digest(text.as_bytes()))
81}
82
83fn top_terms(tokens: &[String]) -> Vec<String> {
84 let mut counts: BTreeMap<&str, usize> = BTreeMap::new();
85 for token in tokens {
86 *counts.entry(token).or_default() += 1;
87 }
88 let mut ranked: Vec<(&str, usize)> = counts.into_iter().collect();
89 ranked.sort_by(|left, right| right.1.cmp(&left.1).then_with(|| left.0.cmp(right.0)));
90 ranked
91 .into_iter()
92 .take(12)
93 .map(|(term, _)| term.to_string())
94 .collect()
95}
96
97fn configured_context_encoder(config: &Value) -> Result<CoreBPE> {
98 let tokenizer_name = match config.pointer("/tokenization/context_tokenizer_name") {
99 Some(Value::String(name)) if !name.trim().is_empty() => name.as_str(),
100 Some(Value::String(_)) => {
101 bail!("tokenization.context_tokenizer_name must not be empty")
102 }
103 Some(_) => bail!("tokenization.context_tokenizer_name must be a string"),
104 None => "cl100k_base",
105 };
106 let encoder = match tokenizer_name {
107 "cl100k_base" => cl100k_base(),
108 "o200k_base" => o200k_base(),
109 "o200k_harmony" => o200k_harmony(),
110 "p50k_base" => p50k_base(),
111 "p50k_edit" => p50k_edit(),
112 "r50k_base" => r50k_base(),
113 unsupported => {
114 bail!(
115 "unsupported tokenization.context_tokenizer_name {unsupported:?}; \
116 supported encodings: cl100k_base, o200k_base, o200k_harmony, \
117 p50k_base, p50k_edit, r50k_base"
118 )
119 }
120 };
121 encoder.with_context(|| format!("failed to initialize {tokenizer_name} tokenizer"))
122}
123
124fn action_queue(files: &[FileAnalysis]) -> Vec<Value> {
125 let mut files: Vec<&FileAnalysis> = files.iter().collect();
126 files.sort_by(|left, right| {
127 right
128 .slop_score
129 .total_cmp(&left.slop_score)
130 .then_with(|| right.tokens.cmp(&left.tokens))
131 .then_with(|| left.path.cmp(&right.path))
132 });
133 files
134 .into_iter()
135 .take(25)
136 .map(|file| {
137 let non_context_reasons = file.reason_codes.iter().any(|reason| {
138 !matches!(reason.as_str(), "critical_token_cost" | "high_token_cost")
139 });
140 json!({
141 "path": file.path,
142 "slop_score": file.slop_score,
143 "slop_band": file.slop_band,
144 "context_band": file.context_band,
145 "tokens": file.tokens,
146 "age_days": file.age_days,
147 "revisions_window": file.revisions_window,
148 "churn_pressure": file.churn_pressure,
149 "reason_codes": file.reason_codes,
150 "is_pure_context_hotspot": !file.reason_codes.is_empty() && !non_context_reasons
151 })
152 })
153 .collect()
154}
155
156pub fn run_find() -> Result<FindResult> {
157 let repo_root = git::resolve_repo_root()?;
158 run_find_in(&repo_root)
159}
160
161pub fn run_find_in(repo_root: &Path) -> Result<FindResult> {
162 config::ensure_state_dirs(repo_root)?;
163 let loaded_config = config::load(repo_root)?;
164 let repo = git::repo_metadata(repo_root)?;
165 let tracked_paths = git::list_tracked_files(repo_root)?;
166 let (inventory_files, skipped) = inventory::build(repo_root, &tracked_paths, &loaded_config)?;
167 let encoder = configured_context_encoder(&loaded_config)?;
168 let mut token_counts = BTreeMap::new();
169 let mut line_counts = BTreeMap::new();
170 let mut token_data = HashMap::new();
171 for file in &inventory_files {
172 let count = encoder.encode_ordinary(&file.text).len();
173 let structural = structural_tokens(&file.path, &file.text);
174 let fingerprint = content_fingerprint(&file.text);
175 token_counts.insert(file.path.clone(), count);
176 line_counts.insert(file.path.clone(), file.lines);
177 token_data.insert(
178 file.path.clone(),
179 (
180 structural.len(),
181 top_terms(&structural),
182 structural,
183 fingerprint,
184 ),
185 );
186 }
187 let analyzed_paths: Vec<String> = inventory_files
188 .iter()
189 .map(|file| file.path.clone())
190 .collect();
191 let now = Utc::now();
192 let (history_by_path, commits, _repo_baselines) = history::analyze_history(
193 repo_root,
194 &analyzed_paths,
195 &token_counts,
196 &line_counts,
197 &loaded_config,
198 now,
199 )?;
200 let mut files = Vec::with_capacity(inventory_files.len());
201 for file in inventory_files {
202 let tokens = token_counts.get(&file.path).copied().unwrap_or_default();
203 let (structural_token_count, top_structural_terms, structural_tokens, content_fingerprint) =
204 token_data
205 .remove(&file.path)
206 .unwrap_or_else(|| (0, Vec::new(), Vec::new(), String::new()));
207 let history = history_by_path.get(&file.path).cloned().unwrap_or_default();
208 files.push(FileAnalysis {
209 path: file.path,
210 bytes: file.bytes,
211 lines: file.lines,
212 blank_lines: file.blank_lines,
213 code_lines: file.code_lines,
214 comment_lines: file.comment_lines,
215 language: file.language,
216 profile: file.profile,
217 classification: file.classification,
218 tokens,
219 context_band: scoring::context_band_for_tokens(tokens, &loaded_config),
220 context_pressure: scoring::context_pressure_for_tokens(tokens, &loaded_config),
221 content_fingerprint,
222 structural_tokens,
223 structural_token_count,
224 top_structural_terms,
225 age_days: history.age_days,
226 revisions_window: history.revisions_window,
227 recency_weighted_commits: history.recency_weighted_commits,
228 added_window: history.added_window,
229 deleted_window: history.deleted_window,
230 churn_lines_window: history.line_churn_window,
231 line_churn_window: history.line_churn_window,
232 token_churn_window: history.token_churn_window,
233 relative_churn_window: history.relative_churn_window,
234 late_churn_spike: history.late_churn_spike,
235 author_count_window: history.author_count_window,
236 author_entropy: history.author_entropy,
237 top_author_share: history.top_author_share,
238 days_since_non_bot_edit: history.days_since_non_bot_edit,
239 recent_maintainer_diversity: history.recent_maintainer_diversity,
240 age_pressure: 0.0,
241 revision_norm: 0.0,
242 relative_churn_norm: 0.0,
243 churn_pressure: 0.0,
244 slop_score: 0.0,
245 slop_band: "low".to_string(),
246 reason_codes: Vec::new(),
247 costs: json!({}),
248 overlays: json!({}),
249 });
250 }
251 scoring::apply_scoring(&mut files, &loaded_config);
252 let organization = overlays::analyze(&mut files, &commits, &loaded_config)?;
253 let folders = scoring::build_folder_analyses(&files, &loaded_config);
254 let queue = action_queue(&files);
255 let generated_at = now.to_rfc3339_opts(SecondsFormat::Secs, true);
256 let analyzed_revision_at = repo.head_commit_timestamp.clone();
257 let analysis = Analysis {
258 repo_root: PathBuf::from(repo_root),
259 repo,
260 config: loaded_config,
261 generated_at,
262 analyzed_revision_at,
263 skipped,
264 tracked_file_count: tracked_paths.len(),
265 files,
266 folders,
267 commits,
268 organization,
269 action_queue: queue,
270 report: Value::Null,
271 };
272 let rollup = health::build_health_rollup(&analysis);
273 let result = report::write_report_bundle(&analysis, &rollup)?;
274 if result.report.get("schema_version").and_then(Value::as_u64) != Some(4) {
275 bail!("internal error: report writer did not produce schema 4");
276 }
277 Ok(result)
278}
279
280#[cfg(test)]
281mod tests {
282 use serde_json::json;
283 use tiktoken_rs::{cl100k_base, r50k_base};
284
285 use super::{
286 action_queue, configured_context_encoder, replace_quoted_strings, structural_tokens,
287 };
288 use crate::model::FileAnalysis;
289 use crate::scoring;
290
291 fn file(path: &str, relative_churn: f64) -> FileAnalysis {
292 FileAnalysis {
293 path: path.to_string(),
294 bytes: 400,
295 lines: 100,
296 blank_lines: 0,
297 code_lines: 100,
298 comment_lines: 0,
299 language: "Rust".to_string(),
300 profile: "agent_context".to_string(),
301 classification: "source".to_string(),
302 tokens: 100,
303 context_band: "compact".to_string(),
304 context_pressure: 0.0,
305 content_fingerprint: String::new(),
306 structural_tokens: Vec::new(),
307 structural_token_count: 0,
308 top_structural_terms: Vec::new(),
309 age_days: 0,
310 revisions_window: 1,
311 recency_weighted_commits: 0.0,
312 added_window: 0,
313 deleted_window: 0,
314 churn_lines_window: 0,
315 line_churn_window: 0,
316 token_churn_window: 0,
317 relative_churn_window: relative_churn,
318 late_churn_spike: 0.0,
319 author_count_window: 0,
320 author_entropy: 0.0,
321 top_author_share: 0.0,
322 days_since_non_bot_edit: None,
323 recent_maintainer_diversity: 0,
324 age_pressure: 0.0,
325 revision_norm: 0.0,
326 relative_churn_norm: 0.0,
327 churn_pressure: 0.0,
328 slop_score: 0.0,
329 slop_band: String::new(),
330 reason_codes: Vec::new(),
331 costs: json!({}),
332 overlays: json!({}),
333 }
334 }
335
336 #[test]
337 fn structural_normalization_is_deterministic() {
338 let tokens = structural_tokens(
339 "src/my_file.rs",
340 "let camelCase = \"secret 123\"; // hello-world",
341 );
342 assert!(tokens.contains(&"camel".to_string()));
343 assert!(tokens.contains(&"case".to_string()));
344 assert!(tokens.contains(&"str".to_string()));
345 assert!(tokens.contains(&"my".to_string()));
346 assert_eq!(
347 replace_quoted_strings("'one' \"two\" `three`"),
348 " str str str "
349 );
350 }
351
352 #[test]
353 fn configured_tokenizer_is_used_exactly_and_unknown_names_fail_closed() {
354 let text = "お誕生日おめでとう";
355 let configured = configured_context_encoder(&json!({"tokenization": {
356 "context_tokenizer_name": "r50k_base"
357 }}))
358 .unwrap();
359 assert_eq!(
360 configured.encode_ordinary(text).len(),
361 r50k_base().unwrap().encode_ordinary(text).len()
362 );
363 assert_ne!(
364 configured.encode_ordinary(text).len(),
365 cl100k_base().unwrap().encode_ordinary(text).len()
366 );
367
368 let result = configured_context_encoder(&json!({"tokenization": {
369 "context_tokenizer_name": "not-a-real-encoding"
370 }}));
371 let error = match result {
372 Ok(_) => panic!("unsupported tokenizer must fail closed"),
373 Err(error) => error,
374 };
375 assert!(
376 error
377 .to_string()
378 .contains("unsupported tokenization.context_tokenizer_name")
379 );
380 }
381
382 #[test]
383 fn action_queue_prioritizes_line_relative_churn_signal() {
384 let mut files = vec![file("src/quiet.rs", 0.1), file("src/volatile.rs", 2.0)];
385 scoring::apply_scoring(&mut files, &json!({}));
386 let queue = action_queue(&files);
387
388 assert_eq!(queue[0]["path"], "src/volatile.rs");
389 assert_eq!(queue[0]["reason_codes"][1], "high_relative_churn");
390 assert_eq!(queue[0]["is_pure_context_hotspot"], false);
391 }
392}