1use std::collections::{BTreeMap, HashMap};
2use std::fs;
3use std::path::{Component, Path, PathBuf};
4use std::sync::LazyLock;
5use std::time::Instant;
6
7use anyhow::{Context, Result, bail};
8use chrono::{SecondsFormat, Utc};
9use regex::Regex;
10use rusqlite::{Connection, OptionalExtension, params};
11use serde::{Deserialize, Serialize};
12use serde_json::{Value, json};
13use sha2::{Digest, Sha256};
14use tiktoken_rs::{
15 CoreBPE, cl100k_base, o200k_base, o200k_harmony, p50k_base, p50k_edit, r50k_base,
16};
17use unicode_normalization::UnicodeNormalization;
18use unicode_segmentation::UnicodeSegmentation;
19
20use crate::config;
21use crate::error::{ClassifiedError, ErrorKind};
22use crate::estimate;
23use crate::git;
24use crate::health;
25use crate::history;
26use crate::inventory;
27use crate::model::{Analysis, FileAnalysis, FindResult, ScopeIdentity};
28use crate::overlays;
29use crate::report;
30use crate::scoring;
31
32static CAMEL_CASE_RE: LazyLock<Regex> =
33 LazyLock::new(|| Regex::new(r"([a-z0-9])([A-Z])").expect("valid camel-case regex"));
34static ACRONYM_BOUNDARY_RE: LazyLock<Regex> =
35 LazyLock::new(|| Regex::new(r"([A-Z]+)([A-Z][a-z])").expect("valid acronym regex"));
36static NUMBER_RE: LazyLock<Regex> =
37 LazyLock::new(|| Regex::new(r"\b\d+(?:\.\d+)?\b").expect("valid number regex"));
38#[derive(Debug, Serialize, Deserialize)]
39struct CachedTokenData {
40 token_count: usize,
41 structural_tokens: Vec<String>,
42 content_fingerprint: String,
43}
44
45fn token_cache_key(text: &str, tokenizer: &str, large_file_bytes: usize, mode: &str) -> String {
46 let mut digest = Sha256::new();
47 digest.update(b"git-slop-token-cache-v3\0");
48 digest.update(tokenizer.as_bytes());
49 digest.update([0]);
50 digest.update(large_file_bytes.to_le_bytes());
51 digest.update(mode.as_bytes());
52 digest.update([0]);
53 digest.update(text.as_bytes());
54 hex::encode(digest.finalize())
55}
56
57struct TokenCache {
58 connection: Connection,
59}
60
61#[derive(Default)]
62struct CacheStats {
63 entries: usize,
64 bytes: u64,
65 failed_evictions: usize,
66}
67
68impl TokenCache {
69 fn open(path: &Path) -> Result<Self> {
70 if let Some(parent) = path.parent() {
71 fs::create_dir_all(parent)?;
72 }
73 let connection = Connection::open(path)
74 .with_context(|| format!("failed to open packed cache {}", path.display()))?;
75 connection.execute_batch(
76 "PRAGMA journal_mode=WAL;
77 PRAGMA synchronous=NORMAL;
78 CREATE TABLE IF NOT EXISTS token_cache (
79 cache_key TEXT PRIMARY KEY,
80 payload BLOB NOT NULL,
81 payload_bytes INTEGER NOT NULL,
82 accessed_at INTEGER NOT NULL
83 );
84 CREATE INDEX IF NOT EXISTS token_cache_accessed ON token_cache(accessed_at, cache_key);",
85 )?;
86 Ok(Self { connection })
87 }
88
89 fn get(&self, key: &str) -> Result<Option<CachedTokenData>> {
90 let payload: Option<Vec<u8>> = self
91 .connection
92 .query_row(
93 "SELECT payload FROM token_cache WHERE cache_key = ?1",
94 [key],
95 |row| row.get(0),
96 )
97 .optional()?;
98 let Some(payload) = payload else {
99 return Ok(None);
100 };
101 self.connection.execute(
102 "UPDATE token_cache SET accessed_at = unixepoch() WHERE cache_key = ?1",
103 [key],
104 )?;
105 Ok(serde_json::from_slice(&payload).ok())
106 }
107
108 fn put(&self, key: &str, value: &CachedTokenData) -> Result<()> {
109 let payload = serde_json::to_vec(value)?;
110 let payload_bytes = payload.len() as u64;
111 self.connection.execute(
112 "INSERT INTO token_cache(cache_key, payload, payload_bytes, accessed_at)
113 VALUES(?1, ?2, ?3, unixepoch())
114 ON CONFLICT(cache_key) DO UPDATE SET
115 payload = excluded.payload,
116 payload_bytes = excluded.payload_bytes,
117 accessed_at = excluded.accessed_at",
118 params![key, payload, payload_bytes],
119 )?;
120 Ok(())
121 }
122
123 fn enforce_limits(&self, max_entries: usize, max_bytes: u64) -> Result<CacheStats> {
124 let mut stats = self.stats()?;
125 while stats.entries > max_entries || stats.bytes > max_bytes {
126 let candidate: Option<(String, u64)> = self
127 .connection
128 .query_row(
129 "SELECT cache_key, payload_bytes FROM token_cache ORDER BY accessed_at, cache_key LIMIT 1",
130 [],
131 |row| Ok((row.get(0)?, row.get(1)?)),
132 )
133 .optional()?;
134 let Some((key, bytes)) = candidate else {
135 break;
136 };
137 match self
138 .connection
139 .execute("DELETE FROM token_cache WHERE cache_key = ?1", [&key])
140 {
141 Ok(1) => {
142 stats.entries = stats.entries.saturating_sub(1);
143 stats.bytes = stats.bytes.saturating_sub(bytes);
144 }
145 _ => {
146 stats.failed_evictions += 1;
147 break;
148 }
149 }
150 }
151 Ok(stats)
152 }
153
154 fn stats(&self) -> Result<CacheStats> {
155 let (entries, bytes): (u64, u64) = self.connection.query_row(
156 "SELECT COUNT(*), COALESCE(SUM(payload_bytes), 0) FROM token_cache",
157 [],
158 |row| Ok((row.get(0)?, row.get(1)?)),
159 )?;
160 Ok(CacheStats {
161 entries: entries as usize,
162 bytes,
163 failed_evictions: 0,
164 })
165 }
166}
167
168fn replace_quoted_strings(text: &str) -> String {
169 let mut result = String::with_capacity(text.len());
170 let mut chars = text.chars().peekable();
171 let mut previous = None;
172 while let Some(character) = chars.next() {
173 if !matches!(character, '\'' | '"' | '`') {
174 result.push(character);
175 previous = Some(character);
176 continue;
177 }
178 if character == '\''
179 && previous.is_some_and(char::is_alphanumeric)
180 && chars.peek().is_some_and(|next| next.is_alphanumeric())
181 {
182 result.push(character);
183 previous = Some(character);
184 continue;
185 }
186 result.push_str(" str ");
187 let quote = character;
188 let mut escaped = false;
189 for next in chars.by_ref() {
190 if escaped {
191 escaped = false;
192 continue;
193 }
194 if next == '\\' {
195 escaped = true;
196 } else if next == quote {
197 break;
198 }
199 }
200 previous = Some(' ');
201 }
202 result
203}
204
205fn structural_mode(path: &str) -> &'static str {
206 match Path::new(path).extension().and_then(|value| value.to_str()) {
207 Some("md" | "mdx") => "markdown",
208 Some("txt") => "prose",
209 Some("sql") => "sql",
210 Some("html" | "htm" | "xml" | "svg") => "markup",
211 _ => "code",
212 }
213}
214
215fn structural_categories(mode: &str, text: &str) -> Value {
216 match mode {
217 "markdown" => {
218 let mut fenced = false;
219 let mut prose_lines = 0usize;
220 let mut fenced_code_lines = 0usize;
221 for line in text.lines() {
222 if line.trim_start().starts_with("```") {
223 fenced = !fenced;
224 } else if fenced {
225 fenced_code_lines += 1;
226 } else {
227 prose_lines += 1;
228 }
229 }
230 json!({"mode": mode, "prose_lines": prose_lines, "fenced_code_lines": fenced_code_lines})
231 }
232 "sql" => {
233 json!({"mode": mode, "query_lines": text.lines().count(), "string_literals_normalized": true})
234 }
235 "markup" => {
236 json!({"mode": mode, "markup_lines": text.lines().count(), "tag_and_text_categories": true})
237 }
238 "prose" => json!({"mode": mode, "prose_lines": text.lines().count()}),
239 _ => {
240 json!({"mode": "code", "code_lines": text.lines().count(), "string_literals_normalized": true})
241 }
242 }
243}
244
245fn structural_content_tokens(mode: &str, text: &str) -> Vec<String> {
246 let normalized: String = text.nfkc().collect();
247 let normalized = normalized.replace(['\u{2018}', '\u{2019}'], "'");
248 let normalized = ACRONYM_BOUNDARY_RE.replace_all(&normalized, "$1 $2");
249 let normalized = CAMEL_CASE_RE.replace_all(&normalized, "$1 $2");
250 let normalized = normalized.replace(['-', '/'], " ");
251 let normalized = if matches!(mode, "prose" | "markdown") {
252 normalized
253 } else {
254 replace_quoted_strings(&normalized)
255 };
256 let normalized = NUMBER_RE.replace_all(&normalized, " 0 ");
257 let lower = normalized.to_lowercase();
258 lower
259 .unicode_words()
260 .flat_map(|word| word.split('_'))
261 .map(|word| {
262 word.split_once('\'')
263 .filter(|(prefix, suffix)| prefix.chars().count() == 1 && !suffix.is_empty())
264 .map_or(word, |(_, suffix)| suffix)
265 })
266 .filter(|item| item.chars().count() > 1)
267 .map(ToOwned::to_owned)
268 .collect()
269}
270
271fn structural_path_tokens(path: &str) -> Vec<String> {
272 path.replace(['-', '_', '.'], "/")
273 .to_ascii_lowercase()
274 .split('/')
275 .filter(|item| !item.is_empty())
276 .map(ToOwned::to_owned)
277 .collect()
278}
279
280#[cfg(test)]
281fn structural_tokens(path: &str, text: &str) -> Vec<String> {
282 let mut tokens = structural_content_tokens(structural_mode(path), text);
283 tokens.extend(structural_path_tokens(path));
284 tokens
285}
286
287fn content_fingerprint(text: &str) -> String {
288 hex::encode(Sha256::digest(text.as_bytes()))
289}
290
291fn top_terms(tokens: &[String], limit: usize) -> Vec<String> {
292 let mut counts: BTreeMap<&str, usize> = BTreeMap::new();
293 for token in tokens {
294 *counts.entry(token).or_default() += 1;
295 }
296 let mut ranked: Vec<(&str, usize)> = counts.into_iter().collect();
297 ranked.sort_by(|left, right| right.1.cmp(&left.1).then_with(|| left.0.cmp(right.0)));
298 ranked
299 .into_iter()
300 .take(limit)
301 .map(|(term, _)| term.to_string())
302 .collect()
303}
304
305fn has_inline_tests(language: &str, text: &str) -> bool {
306 match language {
307 "Rust" => text.contains("#[cfg(test)]") || text.contains("#[test]"),
308 "Go" => text.contains("func Test") || text.contains("func Benchmark"),
309 "Python" => text.contains("def test_") || text.contains("class Test"),
310 "JavaScript" | "JSX" | "TypeScript" | "TSX" => {
311 text.contains("describe(") || text.contains("test(") || text.contains("it(")
312 }
313 "Swift" => text.contains("XCTestCase") || text.contains("@Test"),
314 _ => false,
315 }
316}
317
318fn configured_context_encoder(config: &Value) -> Result<CoreBPE> {
319 let tokenizer_name = match config.pointer("/tokenization/context_tokenizer_name") {
320 Some(Value::String(name)) if !name.trim().is_empty() => name.as_str(),
321 Some(Value::String(_)) => {
322 bail!("tokenization.context_tokenizer_name must not be empty")
323 }
324 Some(_) => bail!("tokenization.context_tokenizer_name must be a string"),
325 None => "cl100k_base",
326 };
327 let encoder = match tokenizer_name {
328 "cl100k_base" => cl100k_base(),
329 "o200k_base" => o200k_base(),
330 "o200k_harmony" => o200k_harmony(),
331 "p50k_base" => p50k_base(),
332 "p50k_edit" => p50k_edit(),
333 "r50k_base" => r50k_base(),
334 unsupported => {
335 bail!(
336 "unsupported tokenization.context_tokenizer_name {unsupported:?}; \
337 supported encodings: cl100k_base, o200k_base, o200k_harmony, \
338 p50k_base, p50k_edit, r50k_base"
339 )
340 }
341 };
342 encoder.with_context(|| format!("failed to initialize {tokenizer_name} tokenizer"))
343}
344
345fn action_queue(files: &[FileAnalysis]) -> Vec<Value> {
346 let mut files: Vec<&FileAnalysis> = files.iter().collect();
347 files.sort_by(|left, right| {
348 right
349 .slop_score
350 .total_cmp(&left.slop_score)
351 .then_with(|| right.tokens.cmp(&left.tokens))
352 .then_with(|| left.path.cmp(&right.path))
353 });
354 files
355 .into_iter()
356 .filter(|file| {
357 !file.reason_codes.is_empty()
358 || matches!(file.context_band.as_str(), "warning" | "critical")
359 || matches!(file.slop_band.as_str(), "high" | "critical")
360 })
361 .map(|file| {
362 let non_context_reasons = file.reason_codes.iter().any(|reason| {
363 !matches!(reason.as_str(), "critical_token_cost" | "high_token_cost")
364 });
365 json!({
366 "path": file.path,
367 "slop_score": file.slop_score,
368 "slop_band": file.slop_band,
369 "context_band": file.context_band,
370 "tokens": file.tokens,
371 "age_days": file.age_days,
372 "revisions_window": file.revisions_window,
373 "churn_pressure": file.churn_pressure,
374 "reason_codes": file.reason_codes,
375 "is_pure_context_hotspot": !file.reason_codes.is_empty() && !non_context_reasons
376 })
377 })
378 .collect()
379}
380
381pub fn run_find() -> Result<FindResult> {
382 let repo_root = git::resolve_repo_root()?;
383 run_find_in(&repo_root)
384}
385
386pub fn run_find_in(repo_root: &Path) -> Result<FindResult> {
387 run_find_scoped(repo_root, false, None, false)
388}
389
390pub fn run_find_in_with_options(repo_root: &Path, allow_shallow: bool) -> Result<FindResult> {
391 run_find_scoped(repo_root, allow_shallow, None, false)
392}
393
394#[derive(Debug, Clone, Default)]
395pub struct FindOptions {
396 pub allow_shallow: bool,
397 pub scope: Option<String>,
398 pub progress: bool,
399 pub allow_empty_scope: bool,
400 pub state_dir: Option<PathBuf>,
401 pub output_dir: Option<PathBuf>,
402 pub no_cache: bool,
403 pub allow_degraded: bool,
404 pub report_profile: String,
405 pub compression: String,
406}
407
408fn normalize_scope(value: Option<&str>) -> Result<Option<String>> {
409 let Some(raw) = value.map(str::trim) else {
410 return Ok(None);
411 };
412 if raw.is_empty() || raw == "." {
413 return Ok(None);
414 }
415 let path = Path::new(raw);
416 if path.is_absolute() {
417 bail!("--scope must be repo-relative, received {raw:?}");
418 }
419 let mut parts = Vec::new();
420 for component in path.components() {
421 match component {
422 Component::Normal(part) => parts.push(
423 part.to_str()
424 .ok_or_else(|| anyhow::anyhow!("--scope must be valid UTF-8"))?,
425 ),
426 Component::CurDir => {}
427 Component::ParentDir | Component::RootDir | Component::Prefix(_) => {
428 bail!("--scope must not escape the repository, received {raw:?}");
429 }
430 }
431 }
432 let normalized = parts.join("/");
433 Ok((!normalized.is_empty()).then_some(normalized))
434}
435
436fn selected_path_digest(paths: &[String]) -> String {
437 let mut digest = Sha256::new();
438 for path in paths {
439 digest.update(path.as_bytes());
440 digest.update([0]);
441 }
442 hex::encode(digest.finalize())
443}
444
445fn measure_rss_checkpoint(
446 checkpoint: &'static str,
447 memory_budget_bytes: u128,
448 allow_degraded: bool,
449 peak_rss_bytes: &mut Option<u64>,
450 exceeded_checkpoints: &mut Vec<&'static str>,
451) -> Result<()> {
452 let Some(rss_bytes) = estimate::current_rss_bytes() else {
453 return Ok(());
454 };
455 *peak_rss_bytes = Some(peak_rss_bytes.unwrap_or_default().max(rss_bytes));
456 if u128::from(rss_bytes) <= memory_budget_bytes {
457 return Ok(());
458 }
459 exceeded_checkpoints.push(checkpoint);
460 if allow_degraded {
461 return Ok(());
462 }
463 Err(ClassifiedError::new(
464 ErrorKind::ResourceLimit,
465 "measured_memory_budget_exceeded",
466 format!(
467 "analysis stopped at {checkpoint}: measured RSS {} MiB exceeds resources.memory_budget_mb={}; narrow --scope, use --allow-degraded, or raise the explicit budget",
468 rss_bytes.div_ceil(1024 * 1024),
469 memory_budget_bytes / 1024 / 1024
470 ),
471 )
472 .at("/resources/memory_budget_mb")
473 .into())
474}
475
476fn selected_content_digest(repo_root: &Path, paths: &[String]) -> Result<String> {
477 let mut digest = Sha256::new();
478 for path in paths {
479 digest.update(path.as_bytes());
480 digest.update([0]);
481 let absolute = repo_root.join(path);
482 let metadata = fs::symlink_metadata(&absolute)
483 .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?;
484 let bytes = if metadata.file_type().is_symlink() {
485 fs::read_link(&absolute)
486 .with_context(|| format!("selected tracked link changed or disappeared: {path}"))?
487 .to_string_lossy()
488 .into_owned()
489 .into_bytes()
490 } else if metadata.is_dir() {
491 b"<gitlink>".to_vec()
492 } else {
493 fs::read(&absolute)
494 .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?
495 };
496 digest.update(bytes);
497 digest.update([0]);
498 }
499 Ok(hex::encode(digest.finalize()))
500}
501
502pub fn run_find_scoped(
503 repo_root: &Path,
504 allow_shallow: bool,
505 scope: Option<&str>,
506 progress: bool,
507) -> Result<FindResult> {
508 run_find_with_options(
509 repo_root,
510 &FindOptions {
511 allow_shallow,
512 scope: scope.map(ToOwned::to_owned),
513 progress,
514 allow_empty_scope: false,
515 ..FindOptions::default()
516 },
517 )
518}
519
520pub fn run_find_with_options(repo_root: &Path, options: &FindOptions) -> Result<FindResult> {
521 let allow_shallow = options.allow_shallow;
522 let scope = options.scope.as_deref();
523 let progress = options.progress;
524 let allow_empty_scope = options.allow_empty_scope;
525 let resolve_root = |value: Option<&Path>, fallback: PathBuf| {
526 value.map_or(fallback, |path| {
527 if path.is_absolute() {
528 path.to_path_buf()
529 } else {
530 repo_root.join(path)
531 }
532 })
533 };
534 let state_root = resolve_root(options.state_dir.as_deref(), config::slop_dir(repo_root));
535 let output_root = resolve_root(options.output_dir.as_deref(), config::slop_dir(repo_root));
536 let started = Instant::now();
537 let phase = |name: &str| {
538 if progress {
539 eprintln!("git-slop: {name} ({:.1}s)", started.elapsed().as_secs_f64());
540 }
541 };
542 phase("preflight");
543 let _scan_lock = config::acquire_scan_lock(repo_root)?;
544 let loaded_config = config::load(repo_root).map_err(|error| {
545 ClassifiedError::new(
546 ErrorKind::Contract,
547 "invalid_configuration",
548 format!("{error:#}"),
549 )
550 .at("/.slop/config.yaml")
551 })?;
552 let mut repo = git::repo_metadata(repo_root)?;
553 let runtime_exclusions = [
554 state_root.join("cache"),
555 output_root.join("latest"),
556 output_root.join("runs"),
557 ]
558 .into_iter()
559 .filter_map(|path| {
560 path.strip_prefix(repo_root)
561 .ok()
562 .map(|value| value.to_string_lossy().replace('\\', "/"))
563 })
564 .collect::<Vec<_>>();
565 let starting_worktree = git::worktree_state_excluding(repo_root, &runtime_exclusions)?;
566 repo.worktree_clean = starting_worktree.clean;
567 repo.staged_change_count = starting_worktree.staged_change_count;
568 repo.modified_tracked_file_count = starting_worktree.modified_tracked_file_count;
569 repo.untracked_file_count = starting_worktree.untracked_file_count;
570 repo.worktree_state_digest = starting_worktree.digest;
571 if repo.is_shallow && !allow_shallow {
572 bail!(
573 "repository history is shallow; rerun with git slop find --allow-shallow to acknowledge incomplete history"
574 );
575 }
576 let all_tracked_paths = git::list_tracked_files(repo_root)?;
577 let scope = normalize_scope(scope)?;
578 if let Some(scope) = scope.as_deref() {
579 if fs::symlink_metadata(repo_root.join(scope)).is_err() {
580 bail!("--scope does not exist in the repository: {scope}");
581 }
582 }
583 let mut tracked_paths = all_tracked_paths
584 .iter()
585 .filter(|path| {
586 scope
587 .as_deref()
588 .is_none_or(|scope| *path == scope || path.starts_with(&format!("{scope}/")))
589 })
590 .cloned()
591 .collect::<Vec<_>>();
592 if tracked_paths.is_empty() && !allow_empty_scope {
593 bail!(
594 "{} selected no tracked paths; pass --allow-empty-scope only when an empty report is intentional",
595 scope.as_deref().map_or_else(
596 || "repository".to_string(),
597 |scope| format!("--scope {scope:?}")
598 )
599 );
600 }
601 let original_selected_path_count = tracked_paths.len();
602 let initial_estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
603 if initial_estimate.estimated_peak_memory_bytes > initial_estimate.memory_budget_bytes
604 && options.allow_degraded
605 {
606 let mut low = 0usize;
607 let mut high = tracked_paths.len();
608 while low < high {
609 let middle = (low + high).div_ceil(2);
610 let candidate = estimate::build(repo_root, &tracked_paths[..middle], &loaded_config);
611 if candidate.estimated_peak_memory_bytes <= candidate.memory_budget_bytes {
612 low = middle;
613 } else {
614 high = middle.saturating_sub(1);
615 }
616 }
617 tracked_paths.truncate(low);
618 }
619 let scope_identity = ScopeIdentity {
620 mode: if scope.is_some() {
621 "scoped"
622 } else {
623 "repository"
624 }
625 .to_string(),
626 path: scope.clone(),
627 selected_path_count: tracked_paths.len(),
628 selected_path_digest: selected_path_digest(&tracked_paths),
629 };
630 let starting_content_digest = selected_content_digest(repo_root, &tracked_paths)?;
631 let estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
632 if estimate.estimated_peak_memory_bytes > estimate.memory_budget_bytes {
633 return Err(ClassifiedError::new(
634 ErrorKind::ResourceLimit,
635 "estimated_memory_budget_exceeded",
636 format!(
637 "analysis bounded before inventory: estimated {} MiB exceeds resources.memory_budget_mb={}; narrow --scope, use --allow-degraded, or raise the explicit budget",
638 estimate.estimated_peak_memory_bytes.div_ceil(1024 * 1024),
639 estimate.memory_budget_bytes / 1024 / 1024
640 ),
641 )
642 .at("/resources/memory_budget_mb")
643 .into());
644 }
645 let mut measured_peak_rss_bytes = None;
646 let mut memory_budget_exceeded_checkpoints = Vec::new();
647 measure_rss_checkpoint(
648 "pre_inventory",
649 estimate.memory_budget_bytes,
650 options.allow_degraded,
651 &mut measured_peak_rss_bytes,
652 &mut memory_budget_exceeded_checkpoints,
653 )?;
654 let (inventory_files, skipped) = inventory::build(repo_root, &tracked_paths, &loaded_config)?;
655 phase("inventory");
656 measure_rss_checkpoint(
657 "post_inventory",
658 estimate.memory_budget_bytes,
659 options.allow_degraded,
660 &mut measured_peak_rss_bytes,
661 &mut memory_budget_exceeded_checkpoints,
662 )?;
663 let encoder = configured_context_encoder(&loaded_config).map_err(|error| {
664 ClassifiedError::new(
665 ErrorKind::Contract,
666 "unsupported_tokenizer",
667 format!("{error:#}"),
668 )
669 .at("/tokenization/context_tokenizer_name")
670 })?;
671 let mut token_counts = BTreeMap::new();
672 let mut line_counts = BTreeMap::new();
673 let mut token_data = HashMap::new();
674 let tokenizer = config::pointer_str(&loaded_config, "/tokenization/context_tokenizer_name")
675 .unwrap_or("cl100k_base")
676 .to_string();
677 let large_file_bytes =
678 config::pointer_u64(&loaded_config, "/resources/large_file_bytes", 2_097_152) as usize;
679 let cache_path = state_root.join("cache").join("token-v4.sqlite3");
680 let mut cache_cleanup_warnings = Vec::new();
681 if !options.no_cache {
682 for version in ["token-v1", "token-v2", "token-v3"] {
683 let legacy_cache = state_root.join("cache").join(version);
684 if legacy_cache.exists() {
685 if let Err(error) = fs::remove_dir_all(&legacy_cache) {
686 cache_cleanup_warnings.push(format!(
687 "failed to remove legacy cache {}: {error}",
688 legacy_cache.display()
689 ));
690 }
691 }
692 }
693 }
694 let cache = if options.no_cache {
695 None
696 } else {
697 Some(TokenCache::open(&cache_path)?)
698 };
699 let mut cache_hits = 0usize;
700 let mut cache_misses = 0usize;
701 let mut structurally_skipped_large_files = 0usize;
702 for file in &inventory_files {
703 if file.analysis_status != "analyzed" || file.bytes > large_file_bytes {
704 structurally_skipped_large_files += 1;
705 }
706 if file.analysis_status != "analyzed" {
707 token_counts.insert(file.path.clone(), 0);
708 line_counts.insert(file.path.clone(), 0);
709 token_data.insert(
710 file.path.clone(),
711 (0, Vec::new(), Vec::new(), String::new()),
712 );
713 continue;
714 }
715 let mode = structural_mode(&file.path);
716 let cache_key = token_cache_key(&file.text, &tokenizer, large_file_bytes, mode);
717 let cached_value = cache
718 .as_ref()
719 .map(|cache| cache.get(&cache_key))
720 .transpose()?
721 .flatten();
722 let cached = if let Some(cached) = cached_value {
723 cache_hits += 1;
724 cached
725 } else {
726 cache_misses += 1;
727 let cached = CachedTokenData {
728 token_count: encoder.encode_ordinary(&file.text).len(),
729 structural_tokens: if file.bytes > large_file_bytes {
730 Vec::new()
731 } else {
732 structural_content_tokens(mode, &file.text)
733 },
734 content_fingerprint: content_fingerprint(&file.text),
735 };
736 if let Some(cache) = &cache {
737 cache.put(&cache_key, &cached)?;
738 }
739 cached
740 };
741 let count = cached.token_count;
742 let mut structural = cached.structural_tokens;
743 if file.bytes <= large_file_bytes {
744 structural.extend(structural_path_tokens(&file.path));
745 }
746 let top_term_limit =
747 config::pointer_u64(&loaded_config, "/semantic_drift/top_term_limit", 25) as usize;
748 let fingerprint = cached.content_fingerprint;
749 token_counts.insert(file.path.clone(), count);
750 line_counts.insert(file.path.clone(), file.lines);
751 token_data.insert(
752 file.path.clone(),
753 (
754 structural.len(),
755 top_terms(&structural, top_term_limit),
756 structural,
757 fingerprint,
758 ),
759 );
760 }
761 let cache_stats = if let Some(cache) = &cache {
762 cache.enforce_limits(
763 config::pointer_u64(&loaded_config, "/resources/cache_max_entries", 10_000) as usize,
764 config::pointer_u64(&loaded_config, "/resources/cache_max_bytes", 536_870_912),
765 )?
766 } else {
767 CacheStats::default()
768 };
769 phase("tokenization");
770 measure_rss_checkpoint(
771 "post_tokenization",
772 estimate.memory_budget_bytes,
773 options.allow_degraded,
774 &mut measured_peak_rss_bytes,
775 &mut memory_budget_exceeded_checkpoints,
776 )?;
777 repo.analyzed_content_digest = Some(starting_content_digest.clone());
778 let analyzed_paths: Vec<String> = inventory_files
779 .iter()
780 .map(|file| file.path.clone())
781 .collect();
782 let now = Utc::now();
783 let (history_by_path, commits, history_diagnostics) = history::analyze_history(
784 repo_root,
785 &analyzed_paths,
786 &token_counts,
787 &line_counts,
788 &loaded_config,
789 now,
790 )?;
791 phase("history");
792 measure_rss_checkpoint(
793 "post_history",
794 estimate.memory_budget_bytes,
795 options.allow_degraded,
796 &mut measured_peak_rss_bytes,
797 &mut memory_budget_exceeded_checkpoints,
798 )?;
799 let mut files = Vec::with_capacity(inventory_files.len());
800 for file in inventory_files {
801 let tokens = token_counts.get(&file.path).copied().unwrap_or_default();
802 let inline_tests = has_inline_tests(&file.language, &file.text);
803 let (structural_token_count, top_structural_terms, structural_tokens, content_fingerprint) =
804 token_data
805 .remove(&file.path)
806 .unwrap_or_else(|| (0, Vec::new(), Vec::new(), String::new()));
807 let history = history_by_path.get(&file.path).cloned().unwrap_or_default();
808 let categories = structural_categories(structural_mode(&file.path), &file.text);
809 files.push(FileAnalysis {
810 path: file.path,
811 bytes: file.bytes,
812 lines: file.lines,
813 blank_lines: file.blank_lines,
814 code_lines: file.code_lines,
815 comment_lines: file.comment_lines,
816 language: file.language,
817 profile: file.profile,
818 classification: file.classification,
819 analysis_status: file.analysis_status,
820 skipped_reason: file.skipped_reason,
821 symlink_metadata: file.symlink_metadata,
822 has_inline_tests: inline_tests,
823 tokens,
824 context_band: scoring::context_band_for_tokens(tokens, &loaded_config),
825 context_pressure: scoring::context_pressure_for_tokens(tokens, &loaded_config),
826 content_fingerprint,
827 structural_tokens,
828 structural_token_count,
829 top_structural_terms,
830 structural_categories: categories,
831 age_days: history.age_days,
832 revisions_window: history.revisions_window,
833 recency_weighted_commits: history.recency_weighted_commits,
834 added_window: history.added_window,
835 deleted_window: history.deleted_window,
836 churn_lines_window: history.line_churn_window,
837 line_churn_window: history.line_churn_window,
838 token_churn_window: history.token_churn_window,
839 relative_churn_window: history.relative_churn_window,
840 late_churn_spike: history.late_churn_spike,
841 author_count_window: history.author_count_window,
842 author_entropy: history.author_entropy,
843 top_author_share: history.top_author_share,
844 days_since_non_bot_edit: history.days_since_non_bot_edit,
845 recent_maintainer_diversity: history.recent_maintainer_diversity,
846 age_pressure: 0.0,
847 revision_norm: 0.0,
848 relative_churn_norm: 0.0,
849 churn_pressure: 0.0,
850 slop_score: 0.0,
851 slop_band: "low".to_string(),
852 reason_codes: Vec::new(),
853 costs: json!({}),
854 overlays: json!({}),
855 });
856 }
857 scoring::apply_scoring(&mut files, &loaded_config);
858 let organization = overlays::analyze(&mut files, &commits, &loaded_config)?;
859 phase("relationships");
860 measure_rss_checkpoint(
861 "post_relationships",
862 estimate.memory_budget_bytes,
863 options.allow_degraded,
864 &mut measured_peak_rss_bytes,
865 &mut memory_budget_exceeded_checkpoints,
866 )?;
867 let folders = scoring::build_folder_analyses(&files, &loaded_config);
868 let queue = action_queue(&files);
869 let generated_at = now.to_rfc3339_opts(SecondsFormat::Secs, true);
870 let analyzed_revision_at = repo.head_commit_timestamp.clone();
871 let ending_worktree = git::worktree_state_excluding(repo_root, &runtime_exclusions)?;
872 if ending_worktree.digest != repo.worktree_state_digest {
873 bail!("repository changed during analysis; no mixed-snapshot report was published");
874 }
875 if selected_content_digest(repo_root, &tracked_paths)? != starting_content_digest {
876 bail!(
877 "selected file content changed during analysis; no mixed-snapshot report was published"
878 );
879 }
880 let analysis = Analysis {
881 output_root,
882 report_profile: if options.report_profile.is_empty() {
883 "standard".to_string()
884 } else {
885 options.report_profile.clone()
886 },
887 compression: if options.compression.is_empty() {
888 "none".to_string()
889 } else {
890 options.compression.clone()
891 },
892 repo,
893 config: loaded_config,
894 generated_at,
895 analyzed_revision_at,
896 skipped,
897 tracked_file_count: all_tracked_paths.len(),
898 scope: scope_identity,
899 files,
900 folders,
901 organization,
902 action_queue: queue,
903 diagnostics: json!({
904 "analysis_elapsed_ms_before_report": started.elapsed().as_millis(),
905 "estimate": estimate,
906 "measured_peak_rss_bytes": measured_peak_rss_bytes,
907 "memory_budget_exceeded_checkpoints": memory_budget_exceeded_checkpoints,
908 "memory_measurement_status": if measured_peak_rss_bytes.is_some() { "measured" } else { "unsupported" },
909 "cache_hits": cache_hits,
910 "cache_misses": cache_misses,
911 "cache_entries": cache_stats.entries,
912 "cache_bytes": cache_stats.bytes,
913 "cache_failed_evictions": cache_stats.failed_evictions,
914 "cache_cleanup_warnings": cache_cleanup_warnings,
915 "cache_status": if options.no_cache { "disabled" } else { "enabled" },
916 "structurally_skipped_large_files": structurally_skipped_large_files,
917 "analysis_status": if tracked_paths.len() < original_selected_path_count || !memory_budget_exceeded_checkpoints.is_empty() { "degraded_resource_budget" } else if structurally_skipped_large_files > 0 { "degraded_large_files" } else { "complete" },
918 "resource_mode": if tracked_paths.len() < original_selected_path_count { "degraded_path_prefix" } else if !memory_budget_exceeded_checkpoints.is_empty() { "degraded_measured_rss" } else { "complete" },
919 "original_selected_path_count": original_selected_path_count,
920 "degraded_omitted_path_count": original_selected_path_count.saturating_sub(tracked_paths.len()),
921 "history": history_diagnostics,
922 "scope": scope
923 }),
924 };
925 let rollup = health::build_health_rollup(&analysis);
926 let result = report::write_report_bundle(&analysis, &rollup)?;
927 phase("report writing");
928 if result.report.get("schema_version").and_then(Value::as_u64) != Some(5) {
929 bail!("internal error: report writer did not produce schema 5");
930 }
931 Ok(result)
932}
933
934#[cfg(test)]
935mod tests {
936 use serde_json::json;
937 use tiktoken_rs::{cl100k_base, r50k_base};
938
939 use super::{
940 action_queue, configured_context_encoder, replace_quoted_strings, structural_tokens,
941 };
942 use crate::model::FileAnalysis;
943 use crate::scoring;
944
945 fn file(path: &str, relative_churn: f64) -> FileAnalysis {
946 FileAnalysis {
947 path: path.to_string(),
948 bytes: 400,
949 lines: 100,
950 blank_lines: 0,
951 code_lines: 100,
952 comment_lines: 0,
953 language: "Rust".to_string(),
954 profile: "agent_context".to_string(),
955 classification: "source".to_string(),
956 analysis_status: "analyzed".to_string(),
957 skipped_reason: None,
958 symlink_metadata: None,
959 has_inline_tests: false,
960 tokens: 100,
961 context_band: "compact".to_string(),
962 context_pressure: 0.0,
963 content_fingerprint: String::new(),
964 structural_tokens: Vec::new(),
965 structural_token_count: 0,
966 top_structural_terms: Vec::new(),
967 structural_categories: json!({"mode": "code"}),
968 age_days: 0,
969 revisions_window: 1,
970 recency_weighted_commits: 0.0,
971 added_window: 0,
972 deleted_window: 0,
973 churn_lines_window: 0,
974 line_churn_window: 0,
975 token_churn_window: 0,
976 relative_churn_window: relative_churn,
977 late_churn_spike: 0.0,
978 author_count_window: 0,
979 author_entropy: 0.0,
980 top_author_share: 0.0,
981 days_since_non_bot_edit: None,
982 recent_maintainer_diversity: 0,
983 age_pressure: 0.0,
984 revision_norm: 0.0,
985 relative_churn_norm: 0.0,
986 churn_pressure: 0.0,
987 slop_score: 0.0,
988 slop_band: String::new(),
989 reason_codes: Vec::new(),
990 costs: json!({}),
991 overlays: json!({}),
992 }
993 }
994
995 #[test]
996 fn structural_normalization_is_deterministic() {
997 let tokens = structural_tokens(
998 "src/my_file.rs",
999 "let camelCase = \"secret 123\"; // hello-world",
1000 );
1001 assert!(tokens.contains(&"camel".to_string()));
1002 assert!(tokens.contains(&"case".to_string()));
1003 assert!(tokens.contains(&"str".to_string()));
1004 assert!(tokens.contains(&"my".to_string()));
1005 assert_eq!(
1006 replace_quoted_strings("'one' \"two\" `three`"),
1007 " str str str "
1008 );
1009 }
1010
1011 #[test]
1012 fn structural_normalization_preserves_unicode_and_apostrophe_words() {
1013 let tokens = structural_tokens("docs/café.md", "L’équipe can’t rename HTTPServer_value");
1014 assert!(tokens.contains(&"équipe".to_string()));
1015 assert!(tokens.contains(&"can't".to_string()));
1016 assert!(tokens.contains(&"http".to_string()));
1017 assert!(tokens.contains(&"server".to_string()));
1018 assert!(tokens.contains(&"value".to_string()));
1019 assert!(tokens.iter().any(|token| token.contains("café")));
1020 }
1021
1022 #[test]
1023 fn configured_tokenizer_is_used_exactly_and_unknown_names_fail_closed() {
1024 let text = "お誕生日おめでとう";
1025 let configured = configured_context_encoder(&json!({"tokenization": {
1026 "context_tokenizer_name": "r50k_base"
1027 }}))
1028 .unwrap();
1029 assert_eq!(
1030 configured.encode_ordinary(text).len(),
1031 r50k_base().unwrap().encode_ordinary(text).len()
1032 );
1033 assert_ne!(
1034 configured.encode_ordinary(text).len(),
1035 cl100k_base().unwrap().encode_ordinary(text).len()
1036 );
1037
1038 let result = configured_context_encoder(&json!({"tokenization": {
1039 "context_tokenizer_name": "not-a-real-encoding"
1040 }}));
1041 let error = match result {
1042 Ok(_) => panic!("unsupported tokenizer must fail closed"),
1043 Err(error) => error,
1044 };
1045 assert!(
1046 error
1047 .to_string()
1048 .contains("unsupported tokenization.context_tokenizer_name")
1049 );
1050 }
1051
1052 #[test]
1053 fn action_queue_prioritizes_line_relative_churn_signal() {
1054 let mut files = vec![file("src/quiet.rs", 0.1), file("src/volatile.rs", 2.0)];
1055 scoring::apply_scoring(&mut files, &json!({}));
1056 let queue = action_queue(&files);
1057
1058 assert_eq!(queue[0]["path"], "src/volatile.rs");
1059 assert_eq!(queue[0]["reason_codes"][1], "high_relative_churn");
1060 assert_eq!(queue[0]["is_pure_context_hotspot"], false);
1061 }
1062}