Skip to main content

git_slop/
analyze.rs

1use std::collections::{BTreeMap, HashMap};
2use std::fs;
3use std::path::{Component, Path, PathBuf};
4use std::sync::LazyLock;
5use std::time::{Duration, Instant};
6
7use anyhow::{Context, Result, bail};
8use chrono::{DateTime, SecondsFormat, Utc};
9use regex::Regex;
10use rusqlite::{Connection, OptionalExtension, params};
11use serde::{Deserialize, Serialize};
12use serde_json::{Value, json};
13use sha2::{Digest, Sha256};
14use tiktoken_rs::{
15    CoreBPE, cl100k_base, o200k_base, o200k_harmony, p50k_base, p50k_edit, r50k_base,
16};
17use unicode_normalization::UnicodeNormalization;
18use unicode_segmentation::UnicodeSegmentation;
19
20use crate::config;
21use crate::error::{ClassifiedError, ErrorKind};
22use crate::estimate;
23use crate::git;
24use crate::health;
25use crate::history;
26use crate::inventory;
27use crate::model::{Analysis, FileAnalysis, FindResult, ScopeIdentity};
28use crate::overlays;
29use crate::report;
30use crate::scoring;
31
32static CAMEL_CASE_RE: LazyLock<Regex> =
33    LazyLock::new(|| Regex::new(r"([a-z0-9])([A-Z])").expect("valid camel-case regex"));
34static ACRONYM_BOUNDARY_RE: LazyLock<Regex> =
35    LazyLock::new(|| Regex::new(r"([A-Z]+)([A-Z][a-z])").expect("valid acronym regex"));
36static NUMBER_RE: LazyLock<Regex> =
37    LazyLock::new(|| Regex::new(r"\b\d+(?:\.\d+)?\b").expect("valid number regex"));
38static RUST_SYMBOL_RE: LazyLock<Regex> = LazyLock::new(|| {
39    Regex::new(r"(?m)^\s*(?:pub(?:\([^)]*\))?\s+)?(?:(?:async|unsafe)\s+)?(?:fn|struct|enum|trait|mod|const|static)\s+([A-Za-z_][A-Za-z0-9_]*)")
40        .expect("valid Rust symbol regex")
41});
42static PYTHON_SYMBOL_RE: LazyLock<Regex> = LazyLock::new(|| {
43    Regex::new(r"(?m)^\s*(?:async\s+)?(?:def|class)\s+([A-Za-z_][A-Za-z0-9_]*)")
44        .expect("valid Python symbol regex")
45});
46static GO_SYMBOL_RE: LazyLock<Regex> = LazyLock::new(|| {
47    Regex::new(r"(?m)^\s*(?:func(?:\s+\([^)]*\))?|type)\s+([A-Za-z_][A-Za-z0-9_]*)")
48        .expect("valid Go symbol regex")
49});
50static JS_SYMBOL_RE: LazyLock<Regex> = LazyLock::new(|| {
51    Regex::new(r"(?m)^\s*(?:export\s+)?(?:default\s+)?(?:async\s+)?(?:function|class|interface|type)\s+([A-Za-z_$][A-Za-z0-9_$]*)")
52        .expect("valid JavaScript symbol regex")
53});
54static MARKDOWN_HEADING_RE: LazyLock<Regex> = LazyLock::new(|| {
55    Regex::new(r"(?m)^\s{0,3}#{1,6}\s+([^#\r\n]+?)\s*#*\s*$").expect("valid Markdown heading regex")
56});
57include!("analyze/cache.rs");
58include!("analyze/structural.rs");
59pub fn run_find() -> Result<FindResult> {
60    let repo_root = git::resolve_repo_root()?;
61    run_find_in(&repo_root)
62}
63
64pub fn run_find_in(repo_root: &Path) -> Result<FindResult> {
65    run_find_scoped(repo_root, false, None, false)
66}
67
68pub fn run_find_in_with_options(repo_root: &Path, allow_shallow: bool) -> Result<FindResult> {
69    run_find_scoped(repo_root, allow_shallow, None, false)
70}
71
72#[derive(Debug, Clone, Default)]
73pub struct FindOptions {
74    pub allow_shallow: bool,
75    pub scope: Option<String>,
76    pub progress: bool,
77    pub allow_empty_scope: bool,
78    pub state_dir: Option<PathBuf>,
79    pub output_dir: Option<PathBuf>,
80    pub no_cache: bool,
81    pub allow_degraded: bool,
82    pub as_of: Option<DateTime<Utc>>,
83    pub report_profile: String,
84    pub compression: String,
85}
86
87pub(crate) fn normalize_scope(value: Option<&str>) -> Result<Option<String>> {
88    let Some(raw) = value.map(str::trim) else {
89        return Ok(None);
90    };
91    if raw.is_empty() || raw == "." {
92        return Ok(None);
93    }
94    let path = Path::new(raw);
95    if path.is_absolute() {
96        bail!("--scope must be repo-relative, received {raw:?}");
97    }
98    let mut parts = Vec::new();
99    for component in path.components() {
100        match component {
101            Component::Normal(part) => parts.push(
102                part.to_str()
103                    .ok_or_else(|| anyhow::anyhow!("--scope must be valid UTF-8"))?,
104            ),
105            Component::CurDir => {}
106            Component::ParentDir | Component::RootDir | Component::Prefix(_) => {
107                bail!("--scope must not escape the repository, received {raw:?}");
108            }
109        }
110    }
111    let normalized = parts.join("/");
112    Ok((!normalized.is_empty()).then_some(normalized))
113}
114
115pub(crate) fn selected_path_digest(paths: &[String]) -> String {
116    let mut digest = Sha256::new();
117    for path in paths {
118        digest.update(path.as_bytes());
119        digest.update([0]);
120    }
121    hex::encode(digest.finalize())
122}
123
124fn measure_rss_checkpoint(
125    checkpoint: &'static str,
126    memory_budget_bytes: u128,
127    allow_degraded: bool,
128    peak_rss_bytes: &mut Option<u64>,
129    exceeded_checkpoints: &mut Vec<&'static str>,
130) -> Result<()> {
131    let Some(rss_bytes) = estimate::current_rss_bytes() else {
132        return Ok(());
133    };
134    *peak_rss_bytes = Some(peak_rss_bytes.unwrap_or_default().max(rss_bytes));
135    if u128::from(rss_bytes) <= memory_budget_bytes {
136        return Ok(());
137    }
138    exceeded_checkpoints.push(checkpoint);
139    if allow_degraded {
140        return Err(ClassifiedError::new(
141            ErrorKind::ResourceLimit,
142            "degraded_memory_recovery_unavailable",
143            format!(
144                "analysis stopped at {checkpoint}: measured RSS {} MiB still exceeds resources.memory_budget_mb={} after deterministic degraded sampling; continuing would violate the memory contract",
145                rss_bytes.div_ceil(1024 * 1024),
146                memory_budget_bytes / 1024 / 1024
147            ),
148        )
149        .at("/resources/memory_budget_mb")
150        .into());
151    }
152    Err(ClassifiedError::new(
153        ErrorKind::ResourceLimit,
154        "measured_memory_budget_exceeded",
155        format!(
156            "analysis stopped at {checkpoint}: measured RSS {} MiB exceeds resources.memory_budget_mb={}; narrow --scope, use --allow-degraded, or raise the explicit budget",
157            rss_bytes.div_ceil(1024 * 1024),
158            memory_budget_bytes / 1024 / 1024
159        ),
160    )
161    .at("/resources/memory_budget_mb")
162    .into())
163}
164
165fn selected_content_digest(repo_root: &Path, paths: &[String]) -> Result<String> {
166    let mut digest = Sha256::new();
167    for path in paths {
168        digest.update(path.as_bytes());
169        digest.update([0]);
170        let absolute = repo_root.join(path);
171        let metadata = fs::symlink_metadata(&absolute)
172            .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?;
173        let bytes = if metadata.file_type().is_symlink() {
174            fs::read_link(&absolute)
175                .with_context(|| format!("selected tracked link changed or disappeared: {path}"))?
176                .to_string_lossy()
177                .into_owned()
178                .into_bytes()
179        } else if metadata.is_dir() {
180            b"<gitlink>".to_vec()
181        } else {
182            fs::read(&absolute)
183                .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?
184        };
185        digest.update(bytes);
186        digest.update([0]);
187    }
188    Ok(hex::encode(digest.finalize()))
189}
190
191fn balanced_path_sample(paths: &[String], limit: usize) -> Vec<String> {
192    let mut roots = BTreeMap::<&str, Vec<&String>>::new();
193    for path in paths {
194        roots
195            .entry(path.split('/').next().unwrap_or("."))
196            .or_default()
197            .push(path);
198    }
199    let mut selected = Vec::with_capacity(limit.min(paths.len()));
200    let mut offset = 0usize;
201    while selected.len() < limit {
202        let mut added = false;
203        for values in roots.values() {
204            if let Some(path) = values.get(offset) {
205                selected.push((*path).clone());
206                added = true;
207                if selected.len() == limit {
208                    break;
209                }
210            }
211        }
212        if !added {
213            break;
214        }
215        offset += 1;
216    }
217    selected.sort();
218    selected
219}
220
221pub fn run_find_scoped(
222    repo_root: &Path,
223    allow_shallow: bool,
224    scope: Option<&str>,
225    progress: bool,
226) -> Result<FindResult> {
227    run_find_with_options(
228        repo_root,
229        &FindOptions {
230            allow_shallow,
231            scope: scope.map(ToOwned::to_owned),
232            progress,
233            allow_empty_scope: false,
234            ..FindOptions::default()
235        },
236    )
237}
238
239pub fn run_find_with_options(repo_root: &Path, options: &FindOptions) -> Result<FindResult> {
240    let allow_shallow = options.allow_shallow;
241    let scope = options.scope.as_deref();
242    let progress = options.progress;
243    let allow_empty_scope = options.allow_empty_scope;
244    let resolve_root = |value: Option<&Path>, fallback: PathBuf| {
245        value.map_or(fallback, |path| {
246            if path.is_absolute() {
247                path.to_path_buf()
248            } else {
249                repo_root.join(path)
250            }
251        })
252    };
253    let state_root = resolve_root(options.state_dir.as_deref(), config::slop_dir(repo_root));
254    let output_root = resolve_root(options.output_dir.as_deref(), config::slop_dir(repo_root));
255    let started = Instant::now();
256    let phase = |name: &str| {
257        if progress {
258            eprintln!("git-slop: {name} ({:.1}s)", started.elapsed().as_secs_f64());
259        }
260    };
261    phase("preflight");
262    let _scan_lock = config::acquire_scan_lock(&state_root)?;
263    let loaded_config = config::load(repo_root).map_err(|error| {
264        ClassifiedError::new(
265            ErrorKind::Contract,
266            "invalid_configuration",
267            format!("{error:#}"),
268        )
269        .at("/.slop/config.yaml")
270    })?;
271    let mut repo = git::repo_metadata(repo_root)?;
272    let runtime_exclusions = [
273        state_root.join("cache"),
274        state_root.join("scan.lock"),
275        state_root.join("scan.lock.owner"),
276        output_root.join("latest"),
277        output_root.join("runs"),
278    ]
279    .into_iter()
280    .filter_map(|path| {
281        path.strip_prefix(repo_root)
282            .ok()
283            .map(|value| value.to_string_lossy().replace('\\', "/"))
284    })
285    .collect::<Vec<_>>();
286    let starting_worktree = git::worktree_state_excluding(repo_root, &runtime_exclusions)?;
287    repo.worktree_clean = starting_worktree.clean;
288    repo.staged_change_count = starting_worktree.staged_change_count;
289    repo.modified_tracked_file_count = starting_worktree.modified_tracked_file_count;
290    repo.untracked_file_count = starting_worktree.untracked_file_count;
291    repo.worktree_state_digest = starting_worktree.digest;
292    if repo.is_shallow && !allow_shallow {
293        bail!(
294            "repository history is shallow; rerun with git slop find --allow-shallow to acknowledge incomplete history"
295        );
296    }
297    let all_tracked_paths = git::list_tracked_files(repo_root)?;
298    let scope = normalize_scope(scope)?;
299    if let Some(scope) = scope.as_deref() {
300        if fs::symlink_metadata(repo_root.join(scope)).is_err() {
301            bail!("--scope does not exist in the repository: {scope}");
302        }
303    }
304    let mut tracked_paths = all_tracked_paths
305        .iter()
306        .filter(|path| {
307            scope
308                .as_deref()
309                .is_none_or(|scope| *path == scope || path.starts_with(&format!("{scope}/")))
310        })
311        .cloned()
312        .collect::<Vec<_>>();
313    if tracked_paths.is_empty() && !allow_empty_scope {
314        bail!(
315            "{} selected no tracked paths; pass --allow-empty-scope only when an empty report is intentional",
316            scope.as_deref().map_or_else(
317                || "repository".to_string(),
318                |scope| format!("--scope {scope:?}")
319            )
320        );
321    }
322    let original_selected_path_count = tracked_paths.len();
323    let initial_estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
324    if initial_estimate.estimated_peak_memory_bytes > initial_estimate.memory_budget_bytes
325        && options.allow_degraded
326    {
327        let mut low = 0usize;
328        let mut high = tracked_paths.len();
329        while low < high {
330            let middle = (low + high).div_ceil(2);
331            let candidate = estimate::build(repo_root, &tracked_paths[..middle], &loaded_config);
332            if candidate.estimated_peak_memory_bytes <= candidate.memory_budget_bytes {
333                low = middle;
334            } else {
335                high = middle.saturating_sub(1);
336            }
337        }
338        tracked_paths = balanced_path_sample(&tracked_paths, low);
339    }
340    let scope_identity = ScopeIdentity {
341        mode: if scope.is_some() {
342            "scoped"
343        } else {
344            "repository"
345        }
346        .to_string(),
347        path: scope.clone(),
348        selected_path_count: tracked_paths.len(),
349        selected_path_digest: selected_path_digest(&tracked_paths),
350    };
351    let starting_content_digest = selected_content_digest(repo_root, &tracked_paths)?;
352    let estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
353    if estimate.estimated_peak_memory_bytes > estimate.memory_budget_bytes {
354        return Err(ClassifiedError::new(
355            ErrorKind::ResourceLimit,
356            "estimated_memory_budget_exceeded",
357            format!(
358                "analysis bounded before inventory: estimated {} MiB exceeds resources.memory_budget_mb={}; narrow --scope, use --allow-degraded, or raise the explicit budget",
359                estimate.estimated_peak_memory_bytes.div_ceil(1024 * 1024),
360                estimate.memory_budget_bytes / 1024 / 1024
361            ),
362        )
363        .at("/resources/memory_budget_mb")
364        .into());
365    }
366    let mut measured_peak_rss_bytes = None;
367    let mut memory_budget_exceeded_checkpoints = Vec::new();
368    measure_rss_checkpoint(
369        "pre_inventory",
370        estimate.memory_budget_bytes,
371        options.allow_degraded,
372        &mut measured_peak_rss_bytes,
373        &mut memory_budget_exceeded_checkpoints,
374    )?;
375    let (inventory_files, skipped) = inventory::build(repo_root, &tracked_paths, &loaded_config)?;
376    phase("inventory");
377    measure_rss_checkpoint(
378        "post_inventory",
379        estimate.memory_budget_bytes,
380        options.allow_degraded,
381        &mut measured_peak_rss_bytes,
382        &mut memory_budget_exceeded_checkpoints,
383    )?;
384    let encoder = configured_context_encoder(&loaded_config).map_err(|error| {
385        ClassifiedError::new(
386            ErrorKind::Contract,
387            "unsupported_tokenizer",
388            format!("{error:#}"),
389        )
390        .at("/tokenization/context_tokenizer_name")
391    })?;
392    let mut token_counts = BTreeMap::new();
393    let mut line_counts = BTreeMap::new();
394    let mut token_data = HashMap::new();
395    let tokenizer = config::pointer_str(&loaded_config, "/tokenization/context_tokenizer_name")
396        .unwrap_or("cl100k_base")
397        .to_string();
398    let large_file_bytes =
399        config::pointer_u64(&loaded_config, "/resources/large_file_bytes", 2_097_152) as usize;
400    let cache_path = state_root.join("cache").join("token-v4.sqlite3");
401    let mut cache_cleanup_warnings = Vec::new();
402    if !options.no_cache {
403        for version in ["token-v1", "token-v2", "token-v3"] {
404            let legacy_cache = state_root.join("cache").join(version);
405            if legacy_cache.exists() {
406                if let Err(error) = fs::remove_dir_all(&legacy_cache) {
407                    cache_cleanup_warnings.push(format!(
408                        "failed to remove legacy cache {}: {error}",
409                        legacy_cache.display()
410                    ));
411                }
412            }
413        }
414    }
415    let mut cache = if options.no_cache {
416        None
417    } else {
418        match TokenCache::open(&cache_path) {
419            Ok(cache) => Some(cache),
420            Err(error) => {
421                cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
422                TokenCache::open(&cache_path).ok()
423            }
424        }
425    };
426    let mut cache_hits = 0usize;
427    let mut cache_misses = 0usize;
428    let mut structurally_skipped_large_files = 0usize;
429    let mut intentionally_skipped_non_text_files = 0usize;
430    let mut incomplete_inventory_files = 0usize;
431    for file in &inventory_files {
432        if file.skipped_reason.as_deref() == Some("large_file_limit") {
433            structurally_skipped_large_files += 1;
434        }
435        if matches!(
436            file.skipped_reason.as_deref(),
437            Some("binary" | "gitlink" | "undecodable")
438        ) {
439            intentionally_skipped_non_text_files += 1;
440        } else if file.analysis_status != "analyzed"
441            && file.skipped_reason.as_deref() != Some("large_file_limit")
442        {
443            incomplete_inventory_files += 1;
444        }
445        if file.analysis_status != "analyzed" {
446            let conservative_tokens = if file.skipped_reason.as_deref() == Some("large_file_limit")
447            {
448                file.bytes.div_ceil(4)
449            } else {
450                0
451            };
452            token_counts.insert(file.path.clone(), conservative_tokens);
453            line_counts.insert(file.path.clone(), 0);
454            token_data.insert(
455                file.path.clone(),
456                (
457                    0,
458                    Vec::new(),
459                    Vec::new(),
460                    format!(
461                        "incomplete:{}:{}",
462                        file.skipped_reason.as_deref().unwrap_or("unknown"),
463                        file.bytes
464                    ),
465                ),
466            );
467            continue;
468        }
469        let mode = structural_mode(&file.path);
470        let cache_key = token_cache_key(&file.text, &tokenizer, large_file_bytes, mode);
471        let cached_value = if let Some(active_cache) = cache.as_ref() {
472            match active_cache.get(&cache_key) {
473                Ok(value) => value,
474                Err(error) => {
475                    cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
476                    cache = TokenCache::open(&cache_path).ok();
477                    None
478                }
479            }
480        } else {
481            None
482        };
483        let cached = if let Some(cached) = cached_value {
484            cache_hits += 1;
485            cached
486        } else {
487            cache_misses += 1;
488            let cached = CachedTokenData {
489                token_count: encoder.encode_ordinary(&file.text).len(),
490                structural_tokens: if file.bytes > large_file_bytes {
491                    Vec::new()
492                } else {
493                    structural_content_tokens(mode, &file.text)
494                },
495                content_fingerprint: content_fingerprint(&file.text),
496            };
497            let put_error = cache
498                .as_ref()
499                .and_then(|active_cache| active_cache.put(&cache_key, &cached).err());
500            if let Some(error) = put_error {
501                cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
502                cache = TokenCache::open(&cache_path).ok();
503            }
504            cached
505        };
506        let count = cached.token_count;
507        let mut structural = cached.structural_tokens;
508        if file.bytes <= large_file_bytes {
509            structural.extend(structural_path_tokens(&file.path));
510        }
511        let top_term_limit =
512            config::pointer_u64(&loaded_config, "/semantic_drift/top_term_limit", 25) as usize;
513        let fingerprint = cached.content_fingerprint;
514        token_counts.insert(file.path.clone(), count);
515        line_counts.insert(file.path.clone(), file.lines);
516        token_data.insert(
517            file.path.clone(),
518            (
519                structural.len(),
520                top_terms(&structural, &file.language, &file.text, top_term_limit),
521                structural,
522                fingerprint,
523            ),
524        );
525    }
526    let cache_stats = if let Some(cache) = &cache {
527        match cache.enforce_limits(
528            config::pointer_u64(&loaded_config, "/resources/cache_max_entries", 10_000) as usize,
529            config::pointer_u64(&loaded_config, "/resources/cache_max_bytes", 536_870_912),
530        ) {
531            Ok(stats) => stats,
532            Err(error) => {
533                cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
534                CacheStats::default()
535            }
536        }
537    } else {
538        CacheStats::default()
539    };
540    phase("tokenization");
541    measure_rss_checkpoint(
542        "post_tokenization",
543        estimate.memory_budget_bytes,
544        options.allow_degraded,
545        &mut measured_peak_rss_bytes,
546        &mut memory_budget_exceeded_checkpoints,
547    )?;
548    repo.analyzed_content_digest = Some(starting_content_digest.clone());
549    let analyzed_paths: Vec<String> = inventory_files
550        .iter()
551        .map(|file| file.path.clone())
552        .collect();
553    let now = options.as_of.unwrap_or_else(Utc::now);
554    let (history_by_path, commits, history_diagnostics) = history::analyze_history(
555        repo_root,
556        &analyzed_paths,
557        &token_counts,
558        &line_counts,
559        &loaded_config,
560        now,
561    )?;
562    phase("history");
563    measure_rss_checkpoint(
564        "post_history",
565        estimate.memory_budget_bytes,
566        options.allow_degraded,
567        &mut measured_peak_rss_bytes,
568        &mut memory_budget_exceeded_checkpoints,
569    )?;
570    let mut files = Vec::with_capacity(inventory_files.len());
571    for file in inventory_files {
572        let tokens = token_counts.get(&file.path).copied().unwrap_or_default();
573        let inline_tests = has_inline_tests(&file.language, &file.text);
574        let (structural_token_count, top_structural_terms, structural_tokens, content_fingerprint) =
575            token_data
576                .remove(&file.path)
577                .unwrap_or_else(|| (0, Vec::new(), Vec::new(), String::new()));
578        let history = history_by_path.get(&file.path).cloned().unwrap_or_default();
579        let categories = structural_categories(structural_mode(&file.path), &file.text);
580        files.push(FileAnalysis {
581            path: file.path,
582            bytes: file.bytes,
583            lines: file.lines,
584            blank_lines: file.blank_lines,
585            code_lines: file.code_lines,
586            comment_lines: file.comment_lines,
587            language: file.language,
588            profile: file.profile.clone(),
589            classification: file.classification,
590            generated_from: file.generated_from,
591            generated_provenance: file.generated_provenance,
592            analysis_status: file.analysis_status,
593            skipped_reason: file.skipped_reason,
594            symlink_metadata: file.symlink_metadata,
595            has_inline_tests: inline_tests,
596            tokens,
597            context_band: scoring::context_band_for_profile(tokens, &file.profile, &loaded_config),
598            context_pressure: scoring::context_pressure_for_profile(
599                tokens,
600                &file.profile,
601                &loaded_config,
602            ),
603            content_fingerprint,
604            content_sha256: file.content_sha256,
605            structural_tokens,
606            structural_token_count,
607            top_structural_terms,
608            structural_categories: categories,
609            age_days: history.age_days,
610            revisions_window: history.revisions_window,
611            recency_weighted_commits: history.recency_weighted_commits,
612            added_window: history.added_window,
613            deleted_window: history.deleted_window,
614            churn_lines_window: history.line_churn_window,
615            line_churn_window: history.line_churn_window,
616            token_churn_window: history.token_churn_window,
617            relative_churn_window: history.relative_churn_window,
618            late_churn_spike: history.late_churn_spike,
619            author_count_window: history.author_count_window,
620            author_entropy: history.author_entropy,
621            top_author_share: history.top_author_share,
622            days_since_non_bot_edit: history.days_since_non_bot_edit,
623            recent_maintainer_diversity: history.recent_maintainer_diversity,
624            age_pressure: 0.0,
625            revision_norm: 0.0,
626            relative_churn_norm: 0.0,
627            churn_pressure: 0.0,
628            slop_score: 0.0,
629            slop_band: "low".to_string(),
630            reason_codes: Vec::new(),
631            costs: json!({}),
632            overlays: json!({}),
633        });
634    }
635    let history_evidence_reliable = !repo.is_shallow
636        && !history_diagnostics
637            .get("history_cap_reached")
638            .and_then(Value::as_bool)
639            .unwrap_or(false)
640        && ![
641            "full_history_cap_status",
642            "window_status_cap_status",
643            "window_numstat_cap_status",
644        ]
645        .into_iter()
646        .any(|field| history_diagnostics.get(field).and_then(Value::as_str) == Some("truncated"));
647    scoring::apply_scoring_with_evidence(&mut files, &loaded_config, history_evidence_reliable);
648    let organization = overlays::analyze(&mut files, &commits, &loaded_config)?;
649    phase("relationships");
650    measure_rss_checkpoint(
651        "post_relationships",
652        estimate.memory_budget_bytes,
653        options.allow_degraded,
654        &mut measured_peak_rss_bytes,
655        &mut memory_budget_exceeded_checkpoints,
656    )?;
657    let folders = scoring::build_folder_analyses(&files, &loaded_config);
658    let candidates = action_queue(&files, history_evidence_reliable, &loaded_config);
659    let (queue, observation_feed): (Vec<_>, Vec<_>) = candidates.into_iter().partition(|item| {
660        let classification = item
661            .get("classification")
662            .and_then(Value::as_str)
663            .unwrap_or("other");
664        let actionable = classification == "source";
665        let supported = item.get("evidence_status").and_then(Value::as_str) == Some("supported")
666            || item.get("is_pure_context_hotspot").and_then(Value::as_bool) == Some(true);
667        actionable
668            && supported
669            && matches!(
670                item.get("severity").and_then(Value::as_str),
671                Some("warning" | "error")
672            )
673    });
674    let generated_at = now.to_rfc3339_opts(SecondsFormat::Secs, true);
675    let analyzed_revision_at = repo.head_commit_timestamp.clone();
676    let ending_worktree = git::worktree_state_excluding(repo_root, &runtime_exclusions)?;
677    if ending_worktree.digest != repo.worktree_state_digest {
678        bail!("repository changed during analysis; no mixed-snapshot report was published");
679    }
680    if selected_content_digest(repo_root, &tracked_paths)? != starting_content_digest {
681        bail!(
682            "selected file content changed during analysis; no mixed-snapshot report was published"
683        );
684    }
685    let estimator_error_ratio = measured_peak_rss_bytes.map(|measured| {
686        let estimated = estimate.estimated_peak_memory_bytes.max(1) as f64;
687        ((measured as f64 - estimated) / estimated * 1_000_000.0).round() / 1_000_000.0
688    });
689    let estimate_range_contains_measurement = measured_peak_rss_bytes.map(|measured| {
690        let measured = u128::from(measured);
691        measured >= estimate.estimated_peak_memory_low_bytes
692            && measured <= estimate.estimated_peak_memory_high_bytes
693    });
694    let history_evidence_status = if repo.head_commit.is_none() {
695        "not_applicable_unborn"
696    } else if history_evidence_reliable {
697        "supported_with_per_file_shrinkage"
698    } else {
699        "incomplete_suppressed"
700    };
701    let analysis = Analysis {
702        output_root,
703        report_profile: if options.report_profile.is_empty() {
704            "standard".to_string()
705        } else {
706            options.report_profile.clone()
707        },
708        compression: if options.compression.is_empty() {
709            "none".to_string()
710        } else {
711            options.compression.clone()
712        },
713        repo,
714        config: loaded_config,
715        generated_at,
716        analyzed_revision_at,
717        skipped,
718        tracked_file_count: all_tracked_paths.len(),
719        scope: scope_identity,
720        files,
721        folders,
722        organization,
723        action_queue: queue,
724        observation_feed,
725        diagnostics: json!({
726            "analysis_elapsed_ms_before_report": started.elapsed().as_millis(),
727            "estimate": estimate,
728            "measured_peak_rss_bytes": measured_peak_rss_bytes,
729            "estimator_error_ratio": estimator_error_ratio,
730            "estimate_range_contains_measurement": estimate_range_contains_measurement,
731            "memory_budget_exceeded_checkpoints": memory_budget_exceeded_checkpoints,
732            "memory_measurement_status": if measured_peak_rss_bytes.is_some() { "measured" } else { "unsupported" },
733            "cache_hits": cache_hits,
734            "cache_misses": cache_misses,
735            "cache_entries": cache_stats.entries,
736            "cache_bytes": cache_stats.bytes,
737            "cache_failed_evictions": cache_stats.failed_evictions,
738            "cache_cleanup_warnings": cache_cleanup_warnings,
739            "cache_status": if options.no_cache { "disabled" } else { "enabled" },
740            "structurally_skipped_large_files": structurally_skipped_large_files,
741            "intentionally_skipped_non_text_files": intentionally_skipped_non_text_files,
742            "incomplete_inventory_files": incomplete_inventory_files,
743            "analysis_status": if tracked_paths.len() < original_selected_path_count || !memory_budget_exceeded_checkpoints.is_empty() { "degraded_resource_budget" } else if structurally_skipped_large_files > 0 { "degraded_large_files" } else if incomplete_inventory_files > 0 { "degraded_incomplete_inventory" } else { "complete" },
744            "resource_mode": if tracked_paths.len() < original_selected_path_count { "degraded_path_prefix" } else if !memory_budget_exceeded_checkpoints.is_empty() { "degraded_measured_rss" } else { "complete" },
745            "original_selected_path_count": original_selected_path_count,
746            "degraded_omitted_path_count": original_selected_path_count.saturating_sub(tracked_paths.len()),
747            "history": history_diagnostics,
748            "history_evidence_status": history_evidence_status,
749            "scope": scope
750        }),
751    };
752    let rollup = health::build_health_rollup(&analysis);
753    let mut result = report::write_report_bundle(&analysis, &rollup)?;
754    phase("report writing");
755    result.elapsed_ms = started.elapsed().as_millis();
756    if result.report.get("schema_version").and_then(Value::as_u64) != Some(5) {
757        bail!("internal error: report writer did not produce schema 5");
758    }
759    Ok(result)
760}
761
762include!("analyze/tests.rs");