Skip to main content

git_slop/
analyze.rs

1use std::collections::{BTreeMap, HashMap};
2use std::fs;
3use std::path::{Component, Path, PathBuf};
4use std::sync::LazyLock;
5use std::time::Instant;
6
7use anyhow::{Context, Result, bail};
8use chrono::{SecondsFormat, Utc};
9use regex::Regex;
10use rusqlite::{Connection, OptionalExtension, params};
11use serde::{Deserialize, Serialize};
12use serde_json::{Value, json};
13use sha2::{Digest, Sha256};
14use tiktoken_rs::{
15    CoreBPE, cl100k_base, o200k_base, o200k_harmony, p50k_base, p50k_edit, r50k_base,
16};
17use unicode_normalization::UnicodeNormalization;
18use unicode_segmentation::UnicodeSegmentation;
19
20use crate::config;
21use crate::error::{ClassifiedError, ErrorKind};
22use crate::estimate;
23use crate::git;
24use crate::health;
25use crate::history;
26use crate::inventory;
27use crate::model::{Analysis, FileAnalysis, FindResult, ScopeIdentity};
28use crate::overlays;
29use crate::report;
30use crate::scoring;
31
32static CAMEL_CASE_RE: LazyLock<Regex> =
33    LazyLock::new(|| Regex::new(r"([a-z0-9])([A-Z])").expect("valid camel-case regex"));
34static ACRONYM_BOUNDARY_RE: LazyLock<Regex> =
35    LazyLock::new(|| Regex::new(r"([A-Z]+)([A-Z][a-z])").expect("valid acronym regex"));
36static NUMBER_RE: LazyLock<Regex> =
37    LazyLock::new(|| Regex::new(r"\b\d+(?:\.\d+)?\b").expect("valid number regex"));
38#[derive(Debug, Serialize, Deserialize)]
39struct CachedTokenData {
40    token_count: usize,
41    structural_tokens: Vec<String>,
42    content_fingerprint: String,
43}
44
45fn token_cache_key(text: &str, tokenizer: &str, large_file_bytes: usize, mode: &str) -> String {
46    let mut digest = Sha256::new();
47    digest.update(b"git-slop-token-cache-v3\0");
48    digest.update(tokenizer.as_bytes());
49    digest.update([0]);
50    digest.update(large_file_bytes.to_le_bytes());
51    digest.update(mode.as_bytes());
52    digest.update([0]);
53    digest.update(text.as_bytes());
54    hex::encode(digest.finalize())
55}
56
57struct TokenCache {
58    connection: Connection,
59}
60
61#[derive(Default)]
62struct CacheStats {
63    entries: usize,
64    bytes: u64,
65    failed_evictions: usize,
66}
67
68impl TokenCache {
69    fn open(path: &Path) -> Result<Self> {
70        if let Some(parent) = path.parent() {
71            fs::create_dir_all(parent)?;
72        }
73        let connection = Connection::open(path)
74            .with_context(|| format!("failed to open packed cache {}", path.display()))?;
75        connection.execute_batch(
76            "PRAGMA journal_mode=WAL;
77             PRAGMA synchronous=NORMAL;
78             CREATE TABLE IF NOT EXISTS token_cache (
79               cache_key TEXT PRIMARY KEY,
80               payload BLOB NOT NULL,
81               payload_bytes INTEGER NOT NULL,
82               accessed_at INTEGER NOT NULL
83             );
84             CREATE INDEX IF NOT EXISTS token_cache_accessed ON token_cache(accessed_at, cache_key);",
85        )?;
86        Ok(Self { connection })
87    }
88
89    fn get(&self, key: &str) -> Result<Option<CachedTokenData>> {
90        let payload: Option<Vec<u8>> = self
91            .connection
92            .query_row(
93                "SELECT payload FROM token_cache WHERE cache_key = ?1",
94                [key],
95                |row| row.get(0),
96            )
97            .optional()?;
98        let Some(payload) = payload else {
99            return Ok(None);
100        };
101        self.connection.execute(
102            "UPDATE token_cache SET accessed_at = unixepoch() WHERE cache_key = ?1",
103            [key],
104        )?;
105        Ok(serde_json::from_slice(&payload).ok())
106    }
107
108    fn put(&self, key: &str, value: &CachedTokenData) -> Result<()> {
109        let payload = serde_json::to_vec(value)?;
110        let payload_bytes = payload.len() as u64;
111        self.connection.execute(
112            "INSERT INTO token_cache(cache_key, payload, payload_bytes, accessed_at)
113             VALUES(?1, ?2, ?3, unixepoch())
114             ON CONFLICT(cache_key) DO UPDATE SET
115               payload = excluded.payload,
116               payload_bytes = excluded.payload_bytes,
117               accessed_at = excluded.accessed_at",
118            params![key, payload, payload_bytes],
119        )?;
120        Ok(())
121    }
122
123    fn enforce_limits(&self, max_entries: usize, max_bytes: u64) -> Result<CacheStats> {
124        let mut stats = self.stats()?;
125        while stats.entries > max_entries || stats.bytes > max_bytes {
126            let candidate: Option<(String, u64)> = self
127                .connection
128                .query_row(
129                    "SELECT cache_key, payload_bytes FROM token_cache ORDER BY accessed_at, cache_key LIMIT 1",
130                    [],
131                    |row| Ok((row.get(0)?, row.get(1)?)),
132                )
133                .optional()?;
134            let Some((key, bytes)) = candidate else {
135                break;
136            };
137            match self
138                .connection
139                .execute("DELETE FROM token_cache WHERE cache_key = ?1", [&key])
140            {
141                Ok(1) => {
142                    stats.entries = stats.entries.saturating_sub(1);
143                    stats.bytes = stats.bytes.saturating_sub(bytes);
144                }
145                _ => {
146                    stats.failed_evictions += 1;
147                    break;
148                }
149            }
150        }
151        Ok(stats)
152    }
153
154    fn stats(&self) -> Result<CacheStats> {
155        let (entries, bytes): (u64, u64) = self.connection.query_row(
156            "SELECT COUNT(*), COALESCE(SUM(payload_bytes), 0) FROM token_cache",
157            [],
158            |row| Ok((row.get(0)?, row.get(1)?)),
159        )?;
160        Ok(CacheStats {
161            entries: entries as usize,
162            bytes,
163            failed_evictions: 0,
164        })
165    }
166}
167
168fn replace_quoted_strings(text: &str) -> String {
169    let mut result = String::with_capacity(text.len());
170    let mut chars = text.chars().peekable();
171    let mut previous = None;
172    while let Some(character) = chars.next() {
173        if !matches!(character, '\'' | '"' | '`') {
174            result.push(character);
175            previous = Some(character);
176            continue;
177        }
178        if character == '\''
179            && previous.is_some_and(char::is_alphanumeric)
180            && chars.peek().is_some_and(|next| next.is_alphanumeric())
181        {
182            result.push(character);
183            previous = Some(character);
184            continue;
185        }
186        result.push_str(" str ");
187        let quote = character;
188        let mut escaped = false;
189        for next in chars.by_ref() {
190            if escaped {
191                escaped = false;
192                continue;
193            }
194            if next == '\\' {
195                escaped = true;
196            } else if next == quote {
197                break;
198            }
199        }
200        previous = Some(' ');
201    }
202    result
203}
204
205fn structural_mode(path: &str) -> &'static str {
206    match Path::new(path).extension().and_then(|value| value.to_str()) {
207        Some("md" | "mdx") => "markdown",
208        Some("txt") => "prose",
209        Some("sql") => "sql",
210        Some("html" | "htm" | "xml" | "svg") => "markup",
211        _ => "code",
212    }
213}
214
215fn structural_categories(mode: &str, text: &str) -> Value {
216    match mode {
217        "markdown" => {
218            let mut fenced = false;
219            let mut prose_lines = 0usize;
220            let mut fenced_code_lines = 0usize;
221            for line in text.lines() {
222                if line.trim_start().starts_with("```") {
223                    fenced = !fenced;
224                } else if fenced {
225                    fenced_code_lines += 1;
226                } else {
227                    prose_lines += 1;
228                }
229            }
230            json!({"mode": mode, "prose_lines": prose_lines, "fenced_code_lines": fenced_code_lines})
231        }
232        "sql" => {
233            json!({"mode": mode, "query_lines": text.lines().count(), "string_literals_normalized": true})
234        }
235        "markup" => {
236            json!({"mode": mode, "markup_lines": text.lines().count(), "tag_and_text_categories": true})
237        }
238        "prose" => json!({"mode": mode, "prose_lines": text.lines().count()}),
239        _ => {
240            json!({"mode": "code", "code_lines": text.lines().count(), "string_literals_normalized": true})
241        }
242    }
243}
244
245fn structural_content_tokens(mode: &str, text: &str) -> Vec<String> {
246    let normalized: String = text.nfkc().collect();
247    let normalized = normalized.replace(['\u{2018}', '\u{2019}'], "'");
248    let normalized = ACRONYM_BOUNDARY_RE.replace_all(&normalized, "$1 $2");
249    let normalized = CAMEL_CASE_RE.replace_all(&normalized, "$1 $2");
250    let normalized = normalized.replace(['-', '/'], " ");
251    let normalized = if matches!(mode, "prose" | "markdown") {
252        normalized
253    } else {
254        replace_quoted_strings(&normalized)
255    };
256    let normalized = NUMBER_RE.replace_all(&normalized, " 0 ");
257    let lower = normalized.to_lowercase();
258    lower
259        .unicode_words()
260        .flat_map(|word| word.split('_'))
261        .map(|word| {
262            word.split_once('\'')
263                .filter(|(prefix, suffix)| prefix.chars().count() == 1 && !suffix.is_empty())
264                .map_or(word, |(_, suffix)| suffix)
265        })
266        .filter(|item| item.chars().count() > 1)
267        .map(ToOwned::to_owned)
268        .collect()
269}
270
271fn structural_path_tokens(path: &str) -> Vec<String> {
272    path.replace(['-', '_', '.'], "/")
273        .to_ascii_lowercase()
274        .split('/')
275        .filter(|item| !item.is_empty())
276        .map(ToOwned::to_owned)
277        .collect()
278}
279
280#[cfg(test)]
281fn structural_tokens(path: &str, text: &str) -> Vec<String> {
282    let mut tokens = structural_content_tokens(structural_mode(path), text);
283    tokens.extend(structural_path_tokens(path));
284    tokens
285}
286
287fn content_fingerprint(text: &str) -> String {
288    hex::encode(Sha256::digest(text.as_bytes()))
289}
290
291fn top_terms(tokens: &[String], limit: usize) -> Vec<String> {
292    let mut counts: BTreeMap<&str, usize> = BTreeMap::new();
293    for token in tokens {
294        *counts.entry(token).or_default() += 1;
295    }
296    let mut ranked: Vec<(&str, usize)> = counts.into_iter().collect();
297    ranked.sort_by(|left, right| right.1.cmp(&left.1).then_with(|| left.0.cmp(right.0)));
298    ranked
299        .into_iter()
300        .take(limit)
301        .map(|(term, _)| term.to_string())
302        .collect()
303}
304
305fn has_inline_tests(language: &str, text: &str) -> bool {
306    match language {
307        "Rust" => text.contains("#[cfg(test)]") || text.contains("#[test]"),
308        "Go" => text.contains("func Test") || text.contains("func Benchmark"),
309        "Python" => text.contains("def test_") || text.contains("class Test"),
310        "JavaScript" | "JSX" | "TypeScript" | "TSX" => {
311            text.contains("describe(") || text.contains("test(") || text.contains("it(")
312        }
313        "Swift" => text.contains("XCTestCase") || text.contains("@Test"),
314        _ => false,
315    }
316}
317
318fn configured_context_encoder(config: &Value) -> Result<CoreBPE> {
319    let tokenizer_name = match config.pointer("/tokenization/context_tokenizer_name") {
320        Some(Value::String(name)) if !name.trim().is_empty() => name.as_str(),
321        Some(Value::String(_)) => {
322            bail!("tokenization.context_tokenizer_name must not be empty")
323        }
324        Some(_) => bail!("tokenization.context_tokenizer_name must be a string"),
325        None => "cl100k_base",
326    };
327    let encoder = match tokenizer_name {
328        "cl100k_base" => cl100k_base(),
329        "o200k_base" => o200k_base(),
330        "o200k_harmony" => o200k_harmony(),
331        "p50k_base" => p50k_base(),
332        "p50k_edit" => p50k_edit(),
333        "r50k_base" => r50k_base(),
334        unsupported => {
335            bail!(
336                "unsupported tokenization.context_tokenizer_name {unsupported:?}; \
337                 supported encodings: cl100k_base, o200k_base, o200k_harmony, \
338                 p50k_base, p50k_edit, r50k_base"
339            )
340        }
341    };
342    encoder.with_context(|| format!("failed to initialize {tokenizer_name} tokenizer"))
343}
344
345fn action_queue(files: &[FileAnalysis]) -> Vec<Value> {
346    let mut files: Vec<&FileAnalysis> = files.iter().collect();
347    files.sort_by(|left, right| {
348        right
349            .slop_score
350            .total_cmp(&left.slop_score)
351            .then_with(|| right.tokens.cmp(&left.tokens))
352            .then_with(|| left.path.cmp(&right.path))
353    });
354    files
355        .into_iter()
356        .filter(|file| {
357            !file.reason_codes.is_empty()
358                || matches!(file.context_band.as_str(), "warning" | "critical")
359                || matches!(file.slop_band.as_str(), "high" | "critical")
360        })
361        .map(|file| {
362            let non_context_reasons = file.reason_codes.iter().any(|reason| {
363                !matches!(reason.as_str(), "critical_token_cost" | "high_token_cost")
364            });
365            json!({
366                "path": file.path,
367                "slop_score": file.slop_score,
368                "slop_band": file.slop_band,
369                "context_band": file.context_band,
370                "tokens": file.tokens,
371                "age_days": file.age_days,
372                "revisions_window": file.revisions_window,
373                "churn_pressure": file.churn_pressure,
374                "reason_codes": file.reason_codes,
375                "is_pure_context_hotspot": !file.reason_codes.is_empty() && !non_context_reasons
376            })
377        })
378        .collect()
379}
380
381pub fn run_find() -> Result<FindResult> {
382    let repo_root = git::resolve_repo_root()?;
383    run_find_in(&repo_root)
384}
385
386pub fn run_find_in(repo_root: &Path) -> Result<FindResult> {
387    run_find_scoped(repo_root, false, None, false)
388}
389
390pub fn run_find_in_with_options(repo_root: &Path, allow_shallow: bool) -> Result<FindResult> {
391    run_find_scoped(repo_root, allow_shallow, None, false)
392}
393
394#[derive(Debug, Clone, Default)]
395pub struct FindOptions {
396    pub allow_shallow: bool,
397    pub scope: Option<String>,
398    pub progress: bool,
399    pub allow_empty_scope: bool,
400    pub state_dir: Option<PathBuf>,
401    pub output_dir: Option<PathBuf>,
402    pub no_cache: bool,
403    pub allow_degraded: bool,
404    pub report_profile: String,
405    pub compression: String,
406}
407
408fn normalize_scope(value: Option<&str>) -> Result<Option<String>> {
409    let Some(raw) = value.map(str::trim) else {
410        return Ok(None);
411    };
412    if raw.is_empty() || raw == "." {
413        return Ok(None);
414    }
415    let path = Path::new(raw);
416    if path.is_absolute() {
417        bail!("--scope must be repo-relative, received {raw:?}");
418    }
419    let mut parts = Vec::new();
420    for component in path.components() {
421        match component {
422            Component::Normal(part) => parts.push(
423                part.to_str()
424                    .ok_or_else(|| anyhow::anyhow!("--scope must be valid UTF-8"))?,
425            ),
426            Component::CurDir => {}
427            Component::ParentDir | Component::RootDir | Component::Prefix(_) => {
428                bail!("--scope must not escape the repository, received {raw:?}");
429            }
430        }
431    }
432    let normalized = parts.join("/");
433    Ok((!normalized.is_empty()).then_some(normalized))
434}
435
436fn selected_path_digest(paths: &[String]) -> String {
437    let mut digest = Sha256::new();
438    for path in paths {
439        digest.update(path.as_bytes());
440        digest.update([0]);
441    }
442    hex::encode(digest.finalize())
443}
444
445fn measure_rss_checkpoint(
446    checkpoint: &'static str,
447    memory_budget_bytes: u128,
448    allow_degraded: bool,
449    peak_rss_bytes: &mut Option<u64>,
450    exceeded_checkpoints: &mut Vec<&'static str>,
451) -> Result<()> {
452    let Some(rss_bytes) = estimate::current_rss_bytes() else {
453        return Ok(());
454    };
455    *peak_rss_bytes = Some(peak_rss_bytes.unwrap_or_default().max(rss_bytes));
456    if u128::from(rss_bytes) <= memory_budget_bytes {
457        return Ok(());
458    }
459    exceeded_checkpoints.push(checkpoint);
460    if allow_degraded {
461        return Ok(());
462    }
463    Err(ClassifiedError::new(
464        ErrorKind::ResourceLimit,
465        "measured_memory_budget_exceeded",
466        format!(
467            "analysis stopped at {checkpoint}: measured RSS {} MiB exceeds resources.memory_budget_mb={}; narrow --scope, use --allow-degraded, or raise the explicit budget",
468            rss_bytes.div_ceil(1024 * 1024),
469            memory_budget_bytes / 1024 / 1024
470        ),
471    )
472    .at("/resources/memory_budget_mb")
473    .into())
474}
475
476fn selected_content_digest(repo_root: &Path, paths: &[String]) -> Result<String> {
477    let mut digest = Sha256::new();
478    for path in paths {
479        digest.update(path.as_bytes());
480        digest.update([0]);
481        let absolute = repo_root.join(path);
482        let metadata = fs::symlink_metadata(&absolute)
483            .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?;
484        let bytes = if metadata.file_type().is_symlink() {
485            fs::read_link(&absolute)
486                .with_context(|| format!("selected tracked link changed or disappeared: {path}"))?
487                .to_string_lossy()
488                .into_owned()
489                .into_bytes()
490        } else if metadata.is_dir() {
491            b"<gitlink>".to_vec()
492        } else {
493            fs::read(&absolute)
494                .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?
495        };
496        digest.update(bytes);
497        digest.update([0]);
498    }
499    Ok(hex::encode(digest.finalize()))
500}
501
502pub fn run_find_scoped(
503    repo_root: &Path,
504    allow_shallow: bool,
505    scope: Option<&str>,
506    progress: bool,
507) -> Result<FindResult> {
508    run_find_with_options(
509        repo_root,
510        &FindOptions {
511            allow_shallow,
512            scope: scope.map(ToOwned::to_owned),
513            progress,
514            allow_empty_scope: false,
515            ..FindOptions::default()
516        },
517    )
518}
519
520pub fn run_find_with_options(repo_root: &Path, options: &FindOptions) -> Result<FindResult> {
521    let allow_shallow = options.allow_shallow;
522    let scope = options.scope.as_deref();
523    let progress = options.progress;
524    let allow_empty_scope = options.allow_empty_scope;
525    let resolve_root = |value: Option<&Path>, fallback: PathBuf| {
526        value.map_or(fallback, |path| {
527            if path.is_absolute() {
528                path.to_path_buf()
529            } else {
530                repo_root.join(path)
531            }
532        })
533    };
534    let state_root = resolve_root(options.state_dir.as_deref(), config::slop_dir(repo_root));
535    let output_root = resolve_root(options.output_dir.as_deref(), config::slop_dir(repo_root));
536    let started = Instant::now();
537    let phase = |name: &str| {
538        if progress {
539            eprintln!("git-slop: {name} ({:.1}s)", started.elapsed().as_secs_f64());
540        }
541    };
542    phase("preflight");
543    let _scan_lock = config::acquire_scan_lock(repo_root)?;
544    let loaded_config = config::load(repo_root).map_err(|error| {
545        ClassifiedError::new(
546            ErrorKind::Contract,
547            "invalid_configuration",
548            format!("{error:#}"),
549        )
550        .at("/.slop/config.yaml")
551    })?;
552    let mut repo = git::repo_metadata(repo_root)?;
553    let runtime_exclusions = [
554        state_root.join("cache"),
555        output_root.join("latest"),
556        output_root.join("runs"),
557    ]
558    .into_iter()
559    .filter_map(|path| {
560        path.strip_prefix(repo_root)
561            .ok()
562            .map(|value| value.to_string_lossy().replace('\\', "/"))
563    })
564    .collect::<Vec<_>>();
565    let starting_worktree = git::worktree_state_excluding(repo_root, &runtime_exclusions)?;
566    repo.worktree_clean = starting_worktree.clean;
567    repo.staged_change_count = starting_worktree.staged_change_count;
568    repo.modified_tracked_file_count = starting_worktree.modified_tracked_file_count;
569    repo.untracked_file_count = starting_worktree.untracked_file_count;
570    repo.worktree_state_digest = starting_worktree.digest;
571    if repo.is_shallow && !allow_shallow {
572        bail!(
573            "repository history is shallow; rerun with git slop find --allow-shallow to acknowledge incomplete history"
574        );
575    }
576    let all_tracked_paths = git::list_tracked_files(repo_root)?;
577    let scope = normalize_scope(scope)?;
578    if let Some(scope) = scope.as_deref() {
579        if fs::symlink_metadata(repo_root.join(scope)).is_err() {
580            bail!("--scope does not exist in the repository: {scope}");
581        }
582    }
583    let mut tracked_paths = all_tracked_paths
584        .iter()
585        .filter(|path| {
586            scope
587                .as_deref()
588                .is_none_or(|scope| *path == scope || path.starts_with(&format!("{scope}/")))
589        })
590        .cloned()
591        .collect::<Vec<_>>();
592    if tracked_paths.is_empty() && !allow_empty_scope {
593        bail!(
594            "{} selected no tracked paths; pass --allow-empty-scope only when an empty report is intentional",
595            scope.as_deref().map_or_else(
596                || "repository".to_string(),
597                |scope| format!("--scope {scope:?}")
598            )
599        );
600    }
601    let original_selected_path_count = tracked_paths.len();
602    let initial_estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
603    if initial_estimate.estimated_peak_memory_bytes > initial_estimate.memory_budget_bytes
604        && options.allow_degraded
605    {
606        let mut low = 0usize;
607        let mut high = tracked_paths.len();
608        while low < high {
609            let middle = (low + high).div_ceil(2);
610            let candidate = estimate::build(repo_root, &tracked_paths[..middle], &loaded_config);
611            if candidate.estimated_peak_memory_bytes <= candidate.memory_budget_bytes {
612                low = middle;
613            } else {
614                high = middle.saturating_sub(1);
615            }
616        }
617        tracked_paths.truncate(low);
618    }
619    let scope_identity = ScopeIdentity {
620        mode: if scope.is_some() {
621            "scoped"
622        } else {
623            "repository"
624        }
625        .to_string(),
626        path: scope.clone(),
627        selected_path_count: tracked_paths.len(),
628        selected_path_digest: selected_path_digest(&tracked_paths),
629    };
630    let starting_content_digest = selected_content_digest(repo_root, &tracked_paths)?;
631    let estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
632    if estimate.estimated_peak_memory_bytes > estimate.memory_budget_bytes {
633        return Err(ClassifiedError::new(
634            ErrorKind::ResourceLimit,
635            "estimated_memory_budget_exceeded",
636            format!(
637                "analysis bounded before inventory: estimated {} MiB exceeds resources.memory_budget_mb={}; narrow --scope, use --allow-degraded, or raise the explicit budget",
638                estimate.estimated_peak_memory_bytes.div_ceil(1024 * 1024),
639                estimate.memory_budget_bytes / 1024 / 1024
640            ),
641        )
642        .at("/resources/memory_budget_mb")
643        .into());
644    }
645    let mut measured_peak_rss_bytes = None;
646    let mut memory_budget_exceeded_checkpoints = Vec::new();
647    measure_rss_checkpoint(
648        "pre_inventory",
649        estimate.memory_budget_bytes,
650        options.allow_degraded,
651        &mut measured_peak_rss_bytes,
652        &mut memory_budget_exceeded_checkpoints,
653    )?;
654    let (inventory_files, skipped) = inventory::build(repo_root, &tracked_paths, &loaded_config)?;
655    phase("inventory");
656    measure_rss_checkpoint(
657        "post_inventory",
658        estimate.memory_budget_bytes,
659        options.allow_degraded,
660        &mut measured_peak_rss_bytes,
661        &mut memory_budget_exceeded_checkpoints,
662    )?;
663    let encoder = configured_context_encoder(&loaded_config).map_err(|error| {
664        ClassifiedError::new(
665            ErrorKind::Contract,
666            "unsupported_tokenizer",
667            format!("{error:#}"),
668        )
669        .at("/tokenization/context_tokenizer_name")
670    })?;
671    let mut token_counts = BTreeMap::new();
672    let mut line_counts = BTreeMap::new();
673    let mut token_data = HashMap::new();
674    let tokenizer = config::pointer_str(&loaded_config, "/tokenization/context_tokenizer_name")
675        .unwrap_or("cl100k_base")
676        .to_string();
677    let large_file_bytes =
678        config::pointer_u64(&loaded_config, "/resources/large_file_bytes", 2_097_152) as usize;
679    let cache_path = state_root.join("cache").join("token-v4.sqlite3");
680    let mut cache_cleanup_warnings = Vec::new();
681    if !options.no_cache {
682        for version in ["token-v1", "token-v2", "token-v3"] {
683            let legacy_cache = state_root.join("cache").join(version);
684            if legacy_cache.exists() {
685                if let Err(error) = fs::remove_dir_all(&legacy_cache) {
686                    cache_cleanup_warnings.push(format!(
687                        "failed to remove legacy cache {}: {error}",
688                        legacy_cache.display()
689                    ));
690                }
691            }
692        }
693    }
694    let cache = if options.no_cache {
695        None
696    } else {
697        Some(TokenCache::open(&cache_path)?)
698    };
699    let mut cache_hits = 0usize;
700    let mut cache_misses = 0usize;
701    let mut structurally_skipped_large_files = 0usize;
702    for file in &inventory_files {
703        if file.analysis_status != "analyzed" || file.bytes > large_file_bytes {
704            structurally_skipped_large_files += 1;
705        }
706        if file.analysis_status != "analyzed" {
707            token_counts.insert(file.path.clone(), 0);
708            line_counts.insert(file.path.clone(), 0);
709            token_data.insert(
710                file.path.clone(),
711                (0, Vec::new(), Vec::new(), String::new()),
712            );
713            continue;
714        }
715        let mode = structural_mode(&file.path);
716        let cache_key = token_cache_key(&file.text, &tokenizer, large_file_bytes, mode);
717        let cached_value = cache
718            .as_ref()
719            .map(|cache| cache.get(&cache_key))
720            .transpose()?
721            .flatten();
722        let cached = if let Some(cached) = cached_value {
723            cache_hits += 1;
724            cached
725        } else {
726            cache_misses += 1;
727            let cached = CachedTokenData {
728                token_count: encoder.encode_ordinary(&file.text).len(),
729                structural_tokens: if file.bytes > large_file_bytes {
730                    Vec::new()
731                } else {
732                    structural_content_tokens(mode, &file.text)
733                },
734                content_fingerprint: content_fingerprint(&file.text),
735            };
736            if let Some(cache) = &cache {
737                cache.put(&cache_key, &cached)?;
738            }
739            cached
740        };
741        let count = cached.token_count;
742        let mut structural = cached.structural_tokens;
743        if file.bytes <= large_file_bytes {
744            structural.extend(structural_path_tokens(&file.path));
745        }
746        let top_term_limit =
747            config::pointer_u64(&loaded_config, "/semantic_drift/top_term_limit", 25) as usize;
748        let fingerprint = cached.content_fingerprint;
749        token_counts.insert(file.path.clone(), count);
750        line_counts.insert(file.path.clone(), file.lines);
751        token_data.insert(
752            file.path.clone(),
753            (
754                structural.len(),
755                top_terms(&structural, top_term_limit),
756                structural,
757                fingerprint,
758            ),
759        );
760    }
761    let cache_stats = if let Some(cache) = &cache {
762        cache.enforce_limits(
763            config::pointer_u64(&loaded_config, "/resources/cache_max_entries", 10_000) as usize,
764            config::pointer_u64(&loaded_config, "/resources/cache_max_bytes", 536_870_912),
765        )?
766    } else {
767        CacheStats::default()
768    };
769    phase("tokenization");
770    measure_rss_checkpoint(
771        "post_tokenization",
772        estimate.memory_budget_bytes,
773        options.allow_degraded,
774        &mut measured_peak_rss_bytes,
775        &mut memory_budget_exceeded_checkpoints,
776    )?;
777    repo.analyzed_content_digest = Some(starting_content_digest.clone());
778    let analyzed_paths: Vec<String> = inventory_files
779        .iter()
780        .map(|file| file.path.clone())
781        .collect();
782    let now = Utc::now();
783    let (history_by_path, commits, history_diagnostics) = history::analyze_history(
784        repo_root,
785        &analyzed_paths,
786        &token_counts,
787        &line_counts,
788        &loaded_config,
789        now,
790    )?;
791    phase("history");
792    measure_rss_checkpoint(
793        "post_history",
794        estimate.memory_budget_bytes,
795        options.allow_degraded,
796        &mut measured_peak_rss_bytes,
797        &mut memory_budget_exceeded_checkpoints,
798    )?;
799    let mut files = Vec::with_capacity(inventory_files.len());
800    for file in inventory_files {
801        let tokens = token_counts.get(&file.path).copied().unwrap_or_default();
802        let inline_tests = has_inline_tests(&file.language, &file.text);
803        let (structural_token_count, top_structural_terms, structural_tokens, content_fingerprint) =
804            token_data
805                .remove(&file.path)
806                .unwrap_or_else(|| (0, Vec::new(), Vec::new(), String::new()));
807        let history = history_by_path.get(&file.path).cloned().unwrap_or_default();
808        let categories = structural_categories(structural_mode(&file.path), &file.text);
809        files.push(FileAnalysis {
810            path: file.path,
811            bytes: file.bytes,
812            lines: file.lines,
813            blank_lines: file.blank_lines,
814            code_lines: file.code_lines,
815            comment_lines: file.comment_lines,
816            language: file.language,
817            profile: file.profile,
818            classification: file.classification,
819            analysis_status: file.analysis_status,
820            skipped_reason: file.skipped_reason,
821            symlink_metadata: file.symlink_metadata,
822            has_inline_tests: inline_tests,
823            tokens,
824            context_band: scoring::context_band_for_tokens(tokens, &loaded_config),
825            context_pressure: scoring::context_pressure_for_tokens(tokens, &loaded_config),
826            content_fingerprint,
827            structural_tokens,
828            structural_token_count,
829            top_structural_terms,
830            structural_categories: categories,
831            age_days: history.age_days,
832            revisions_window: history.revisions_window,
833            recency_weighted_commits: history.recency_weighted_commits,
834            added_window: history.added_window,
835            deleted_window: history.deleted_window,
836            churn_lines_window: history.line_churn_window,
837            line_churn_window: history.line_churn_window,
838            token_churn_window: history.token_churn_window,
839            relative_churn_window: history.relative_churn_window,
840            late_churn_spike: history.late_churn_spike,
841            author_count_window: history.author_count_window,
842            author_entropy: history.author_entropy,
843            top_author_share: history.top_author_share,
844            days_since_non_bot_edit: history.days_since_non_bot_edit,
845            recent_maintainer_diversity: history.recent_maintainer_diversity,
846            age_pressure: 0.0,
847            revision_norm: 0.0,
848            relative_churn_norm: 0.0,
849            churn_pressure: 0.0,
850            slop_score: 0.0,
851            slop_band: "low".to_string(),
852            reason_codes: Vec::new(),
853            costs: json!({}),
854            overlays: json!({}),
855        });
856    }
857    scoring::apply_scoring(&mut files, &loaded_config);
858    let organization = overlays::analyze(&mut files, &commits, &loaded_config)?;
859    phase("relationships");
860    measure_rss_checkpoint(
861        "post_relationships",
862        estimate.memory_budget_bytes,
863        options.allow_degraded,
864        &mut measured_peak_rss_bytes,
865        &mut memory_budget_exceeded_checkpoints,
866    )?;
867    let folders = scoring::build_folder_analyses(&files, &loaded_config);
868    let queue = action_queue(&files);
869    let generated_at = now.to_rfc3339_opts(SecondsFormat::Secs, true);
870    let analyzed_revision_at = repo.head_commit_timestamp.clone();
871    let ending_worktree = git::worktree_state_excluding(repo_root, &runtime_exclusions)?;
872    if ending_worktree.digest != repo.worktree_state_digest {
873        bail!("repository changed during analysis; no mixed-snapshot report was published");
874    }
875    if selected_content_digest(repo_root, &tracked_paths)? != starting_content_digest {
876        bail!(
877            "selected file content changed during analysis; no mixed-snapshot report was published"
878        );
879    }
880    let analysis = Analysis {
881        output_root,
882        report_profile: if options.report_profile.is_empty() {
883            "standard".to_string()
884        } else {
885            options.report_profile.clone()
886        },
887        compression: if options.compression.is_empty() {
888            "none".to_string()
889        } else {
890            options.compression.clone()
891        },
892        repo,
893        config: loaded_config,
894        generated_at,
895        analyzed_revision_at,
896        skipped,
897        tracked_file_count: all_tracked_paths.len(),
898        scope: scope_identity,
899        files,
900        folders,
901        organization,
902        action_queue: queue,
903        diagnostics: json!({
904            "analysis_elapsed_ms_before_report": started.elapsed().as_millis(),
905            "estimate": estimate,
906            "measured_peak_rss_bytes": measured_peak_rss_bytes,
907            "memory_budget_exceeded_checkpoints": memory_budget_exceeded_checkpoints,
908            "memory_measurement_status": if measured_peak_rss_bytes.is_some() { "measured" } else { "unsupported" },
909            "cache_hits": cache_hits,
910            "cache_misses": cache_misses,
911            "cache_entries": cache_stats.entries,
912            "cache_bytes": cache_stats.bytes,
913            "cache_failed_evictions": cache_stats.failed_evictions,
914            "cache_cleanup_warnings": cache_cleanup_warnings,
915            "cache_status": if options.no_cache { "disabled" } else { "enabled" },
916            "structurally_skipped_large_files": structurally_skipped_large_files,
917            "analysis_status": if tracked_paths.len() < original_selected_path_count || !memory_budget_exceeded_checkpoints.is_empty() { "degraded_resource_budget" } else if structurally_skipped_large_files > 0 { "degraded_large_files" } else { "complete" },
918            "resource_mode": if tracked_paths.len() < original_selected_path_count { "degraded_path_prefix" } else if !memory_budget_exceeded_checkpoints.is_empty() { "degraded_measured_rss" } else { "complete" },
919            "original_selected_path_count": original_selected_path_count,
920            "degraded_omitted_path_count": original_selected_path_count.saturating_sub(tracked_paths.len()),
921            "history": history_diagnostics,
922            "scope": scope
923        }),
924    };
925    let rollup = health::build_health_rollup(&analysis);
926    let result = report::write_report_bundle(&analysis, &rollup)?;
927    phase("report writing");
928    if result.report.get("schema_version").and_then(Value::as_u64) != Some(5) {
929        bail!("internal error: report writer did not produce schema 5");
930    }
931    Ok(result)
932}
933
934#[cfg(test)]
935mod tests {
936    use serde_json::json;
937    use tiktoken_rs::{cl100k_base, r50k_base};
938
939    use super::{
940        action_queue, configured_context_encoder, replace_quoted_strings, structural_tokens,
941    };
942    use crate::model::FileAnalysis;
943    use crate::scoring;
944
945    fn file(path: &str, relative_churn: f64) -> FileAnalysis {
946        FileAnalysis {
947            path: path.to_string(),
948            bytes: 400,
949            lines: 100,
950            blank_lines: 0,
951            code_lines: 100,
952            comment_lines: 0,
953            language: "Rust".to_string(),
954            profile: "agent_context".to_string(),
955            classification: "source".to_string(),
956            analysis_status: "analyzed".to_string(),
957            skipped_reason: None,
958            symlink_metadata: None,
959            has_inline_tests: false,
960            tokens: 100,
961            context_band: "compact".to_string(),
962            context_pressure: 0.0,
963            content_fingerprint: String::new(),
964            structural_tokens: Vec::new(),
965            structural_token_count: 0,
966            top_structural_terms: Vec::new(),
967            structural_categories: json!({"mode": "code"}),
968            age_days: 0,
969            revisions_window: 1,
970            recency_weighted_commits: 0.0,
971            added_window: 0,
972            deleted_window: 0,
973            churn_lines_window: 0,
974            line_churn_window: 0,
975            token_churn_window: 0,
976            relative_churn_window: relative_churn,
977            late_churn_spike: 0.0,
978            author_count_window: 0,
979            author_entropy: 0.0,
980            top_author_share: 0.0,
981            days_since_non_bot_edit: None,
982            recent_maintainer_diversity: 0,
983            age_pressure: 0.0,
984            revision_norm: 0.0,
985            relative_churn_norm: 0.0,
986            churn_pressure: 0.0,
987            slop_score: 0.0,
988            slop_band: String::new(),
989            reason_codes: Vec::new(),
990            costs: json!({}),
991            overlays: json!({}),
992        }
993    }
994
995    #[test]
996    fn structural_normalization_is_deterministic() {
997        let tokens = structural_tokens(
998            "src/my_file.rs",
999            "let camelCase = \"secret 123\"; // hello-world",
1000        );
1001        assert!(tokens.contains(&"camel".to_string()));
1002        assert!(tokens.contains(&"case".to_string()));
1003        assert!(tokens.contains(&"str".to_string()));
1004        assert!(tokens.contains(&"my".to_string()));
1005        assert_eq!(
1006            replace_quoted_strings("'one' \"two\" `three`"),
1007            " str   str   str "
1008        );
1009    }
1010
1011    #[test]
1012    fn structural_normalization_preserves_unicode_and_apostrophe_words() {
1013        let tokens = structural_tokens("docs/café.md", "L’équipe can’t rename HTTPServer_value");
1014        assert!(tokens.contains(&"équipe".to_string()));
1015        assert!(tokens.contains(&"can't".to_string()));
1016        assert!(tokens.contains(&"http".to_string()));
1017        assert!(tokens.contains(&"server".to_string()));
1018        assert!(tokens.contains(&"value".to_string()));
1019        assert!(tokens.iter().any(|token| token.contains("café")));
1020    }
1021
1022    #[test]
1023    fn configured_tokenizer_is_used_exactly_and_unknown_names_fail_closed() {
1024        let text = "お誕生日おめでとう";
1025        let configured = configured_context_encoder(&json!({"tokenization": {
1026            "context_tokenizer_name": "r50k_base"
1027        }}))
1028        .unwrap();
1029        assert_eq!(
1030            configured.encode_ordinary(text).len(),
1031            r50k_base().unwrap().encode_ordinary(text).len()
1032        );
1033        assert_ne!(
1034            configured.encode_ordinary(text).len(),
1035            cl100k_base().unwrap().encode_ordinary(text).len()
1036        );
1037
1038        let result = configured_context_encoder(&json!({"tokenization": {
1039            "context_tokenizer_name": "not-a-real-encoding"
1040        }}));
1041        let error = match result {
1042            Ok(_) => panic!("unsupported tokenizer must fail closed"),
1043            Err(error) => error,
1044        };
1045        assert!(
1046            error
1047                .to_string()
1048                .contains("unsupported tokenization.context_tokenizer_name")
1049        );
1050    }
1051
1052    #[test]
1053    fn action_queue_prioritizes_line_relative_churn_signal() {
1054        let mut files = vec![file("src/quiet.rs", 0.1), file("src/volatile.rs", 2.0)];
1055        scoring::apply_scoring(&mut files, &json!({}));
1056        let queue = action_queue(&files);
1057
1058        assert_eq!(queue[0]["path"], "src/volatile.rs");
1059        assert_eq!(queue[0]["reason_codes"][1], "high_relative_churn");
1060        assert_eq!(queue[0]["is_pure_context_hotspot"], false);
1061    }
1062}