Skip to main content

git_slop/
analyze.rs

1use std::collections::{BTreeMap, HashMap};
2use std::fs;
3use std::path::{Component, Path, PathBuf};
4use std::sync::LazyLock;
5use std::time::{Duration, Instant};
6
7use anyhow::{Context, Result, bail};
8use chrono::{DateTime, SecondsFormat, Utc};
9use regex::Regex;
10use rusqlite::{Connection, OptionalExtension, params};
11use serde::{Deserialize, Serialize};
12use serde_json::{Value, json};
13use sha2::{Digest, Sha256};
14use tiktoken_rs::{
15    CoreBPE, cl100k_base, o200k_base, o200k_harmony, p50k_base, p50k_edit, r50k_base,
16};
17use unicode_normalization::UnicodeNormalization;
18use unicode_segmentation::UnicodeSegmentation;
19
20use crate::config;
21use crate::error::{ClassifiedError, ErrorKind};
22use crate::estimate;
23use crate::git;
24use crate::health;
25use crate::history;
26use crate::inventory;
27use crate::model::{Analysis, FileAnalysis, FindResult, ScopeIdentity};
28use crate::overlays;
29use crate::report;
30use crate::scoring;
31
32static CAMEL_CASE_RE: LazyLock<Regex> =
33    LazyLock::new(|| Regex::new(r"([a-z0-9])([A-Z])").expect("valid camel-case regex"));
34static ACRONYM_BOUNDARY_RE: LazyLock<Regex> =
35    LazyLock::new(|| Regex::new(r"([A-Z]+)([A-Z][a-z])").expect("valid acronym regex"));
36static NUMBER_RE: LazyLock<Regex> =
37    LazyLock::new(|| Regex::new(r"\b\d+(?:\.\d+)?\b").expect("valid number regex"));
38static RUST_SYMBOL_RE: LazyLock<Regex> = LazyLock::new(|| {
39    Regex::new(r"(?m)^\s*(?:pub(?:\([^)]*\))?\s+)?(?:(?:async|unsafe)\s+)?(?:fn|struct|enum|trait|mod|const|static)\s+([A-Za-z_][A-Za-z0-9_]*)")
40        .expect("valid Rust symbol regex")
41});
42static PYTHON_SYMBOL_RE: LazyLock<Regex> = LazyLock::new(|| {
43    Regex::new(r"(?m)^\s*(?:async\s+)?(?:def|class)\s+([A-Za-z_][A-Za-z0-9_]*)")
44        .expect("valid Python symbol regex")
45});
46static GO_SYMBOL_RE: LazyLock<Regex> = LazyLock::new(|| {
47    Regex::new(r"(?m)^\s*(?:func(?:\s+\([^)]*\))?|type)\s+([A-Za-z_][A-Za-z0-9_]*)")
48        .expect("valid Go symbol regex")
49});
50static JS_SYMBOL_RE: LazyLock<Regex> = LazyLock::new(|| {
51    Regex::new(r"(?m)^\s*(?:export\s+)?(?:default\s+)?(?:async\s+)?(?:function|class|interface|type)\s+([A-Za-z_$][A-Za-z0-9_$]*)")
52        .expect("valid JavaScript symbol regex")
53});
54static MARKDOWN_HEADING_RE: LazyLock<Regex> = LazyLock::new(|| {
55    Regex::new(r"(?m)^\s{0,3}#{1,6}\s+([^#\r\n]+?)\s*#*\s*$").expect("valid Markdown heading regex")
56});
57include!("analyze/cache.rs");
58include!("analyze/structural.rs");
59pub fn run_find() -> Result<FindResult> {
60    let repo_root = git::resolve_repo_root()?;
61    run_find_in(&repo_root)
62}
63
64pub fn run_find_in(repo_root: &Path) -> Result<FindResult> {
65    run_find_scoped(repo_root, false, None, false)
66}
67
68pub fn run_find_in_with_options(repo_root: &Path, allow_shallow: bool) -> Result<FindResult> {
69    run_find_scoped(repo_root, allow_shallow, None, false)
70}
71
72#[derive(Debug, Clone, Default)]
73pub struct FindOptions {
74    pub allow_shallow: bool,
75    pub scope: Option<String>,
76    pub progress: bool,
77    pub allow_empty_scope: bool,
78    pub state_dir: Option<PathBuf>,
79    pub output_dir: Option<PathBuf>,
80    pub no_cache: bool,
81    pub allow_degraded: bool,
82    pub as_of: Option<DateTime<Utc>>,
83    pub report_profile: String,
84    pub compression: String,
85}
86
87pub(crate) fn normalize_scope(value: Option<&str>) -> Result<Option<String>> {
88    let Some(raw) = value.map(str::trim) else {
89        return Ok(None);
90    };
91    if raw.is_empty() || raw == "." {
92        return Ok(None);
93    }
94    let path = Path::new(raw);
95    if path.is_absolute() {
96        bail!("--scope must be repo-relative, received {raw:?}");
97    }
98    let mut parts = Vec::new();
99    for component in path.components() {
100        match component {
101            Component::Normal(part) => parts.push(
102                part.to_str()
103                    .ok_or_else(|| anyhow::anyhow!("--scope must be valid UTF-8"))?,
104            ),
105            Component::CurDir => {}
106            Component::ParentDir | Component::RootDir | Component::Prefix(_) => {
107                bail!("--scope must not escape the repository, received {raw:?}");
108            }
109        }
110    }
111    let normalized = parts.join("/");
112    Ok((!normalized.is_empty()).then_some(normalized))
113}
114
115fn selected_path_digest(paths: &[String]) -> String {
116    let mut digest = Sha256::new();
117    for path in paths {
118        digest.update(path.as_bytes());
119        digest.update([0]);
120    }
121    hex::encode(digest.finalize())
122}
123
124fn measure_rss_checkpoint(
125    checkpoint: &'static str,
126    memory_budget_bytes: u128,
127    allow_degraded: bool,
128    peak_rss_bytes: &mut Option<u64>,
129    exceeded_checkpoints: &mut Vec<&'static str>,
130) -> Result<()> {
131    let Some(rss_bytes) = estimate::current_rss_bytes() else {
132        return Ok(());
133    };
134    *peak_rss_bytes = Some(peak_rss_bytes.unwrap_or_default().max(rss_bytes));
135    if u128::from(rss_bytes) <= memory_budget_bytes {
136        return Ok(());
137    }
138    exceeded_checkpoints.push(checkpoint);
139    if allow_degraded {
140        return Err(ClassifiedError::new(
141            ErrorKind::ResourceLimit,
142            "degraded_memory_recovery_unavailable",
143            format!(
144                "analysis stopped at {checkpoint}: measured RSS {} MiB still exceeds resources.memory_budget_mb={} after deterministic degraded sampling; continuing would violate the memory contract",
145                rss_bytes.div_ceil(1024 * 1024),
146                memory_budget_bytes / 1024 / 1024
147            ),
148        )
149        .at("/resources/memory_budget_mb")
150        .into());
151    }
152    Err(ClassifiedError::new(
153        ErrorKind::ResourceLimit,
154        "measured_memory_budget_exceeded",
155        format!(
156            "analysis stopped at {checkpoint}: measured RSS {} MiB exceeds resources.memory_budget_mb={}; narrow --scope, use --allow-degraded, or raise the explicit budget",
157            rss_bytes.div_ceil(1024 * 1024),
158            memory_budget_bytes / 1024 / 1024
159        ),
160    )
161    .at("/resources/memory_budget_mb")
162    .into())
163}
164
165fn selected_content_digest(repo_root: &Path, paths: &[String]) -> Result<String> {
166    let mut digest = Sha256::new();
167    for path in paths {
168        digest.update(path.as_bytes());
169        digest.update([0]);
170        let absolute = repo_root.join(path);
171        let metadata = fs::symlink_metadata(&absolute)
172            .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?;
173        let bytes = if metadata.file_type().is_symlink() {
174            fs::read_link(&absolute)
175                .with_context(|| format!("selected tracked link changed or disappeared: {path}"))?
176                .to_string_lossy()
177                .into_owned()
178                .into_bytes()
179        } else if metadata.is_dir() {
180            b"<gitlink>".to_vec()
181        } else {
182            fs::read(&absolute)
183                .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?
184        };
185        digest.update(bytes);
186        digest.update([0]);
187    }
188    Ok(hex::encode(digest.finalize()))
189}
190
191fn balanced_path_sample(paths: &[String], limit: usize) -> Vec<String> {
192    let mut roots = BTreeMap::<&str, Vec<&String>>::new();
193    for path in paths {
194        roots
195            .entry(path.split('/').next().unwrap_or("."))
196            .or_default()
197            .push(path);
198    }
199    let mut selected = Vec::with_capacity(limit.min(paths.len()));
200    let mut offset = 0usize;
201    while selected.len() < limit {
202        let mut added = false;
203        for values in roots.values() {
204            if let Some(path) = values.get(offset) {
205                selected.push((*path).clone());
206                added = true;
207                if selected.len() == limit {
208                    break;
209                }
210            }
211        }
212        if !added {
213            break;
214        }
215        offset += 1;
216    }
217    selected.sort();
218    selected
219}
220
221pub fn run_find_scoped(
222    repo_root: &Path,
223    allow_shallow: bool,
224    scope: Option<&str>,
225    progress: bool,
226) -> Result<FindResult> {
227    run_find_with_options(
228        repo_root,
229        &FindOptions {
230            allow_shallow,
231            scope: scope.map(ToOwned::to_owned),
232            progress,
233            allow_empty_scope: false,
234            ..FindOptions::default()
235        },
236    )
237}
238
239pub fn run_find_with_options(repo_root: &Path, options: &FindOptions) -> Result<FindResult> {
240    let allow_shallow = options.allow_shallow;
241    let scope = options.scope.as_deref();
242    let progress = options.progress;
243    let allow_empty_scope = options.allow_empty_scope;
244    let resolve_root = |value: Option<&Path>, fallback: PathBuf| {
245        value.map_or(fallback, |path| {
246            if path.is_absolute() {
247                path.to_path_buf()
248            } else {
249                repo_root.join(path)
250            }
251        })
252    };
253    let state_root = resolve_root(options.state_dir.as_deref(), config::slop_dir(repo_root));
254    let output_root = resolve_root(options.output_dir.as_deref(), config::slop_dir(repo_root));
255    let started = Instant::now();
256    let phase = |name: &str| {
257        if progress {
258            eprintln!("git-slop: {name} ({:.1}s)", started.elapsed().as_secs_f64());
259        }
260    };
261    phase("preflight");
262    let _scan_lock = config::acquire_scan_lock(&state_root)?;
263    let loaded_config = config::load(repo_root).map_err(|error| {
264        ClassifiedError::new(
265            ErrorKind::Contract,
266            "invalid_configuration",
267            format!("{error:#}"),
268        )
269        .at("/.slop/config.yaml")
270    })?;
271    let mut repo = git::repo_metadata(repo_root)?;
272    let runtime_exclusions = [
273        state_root.join("cache"),
274        output_root.join("latest"),
275        output_root.join("runs"),
276    ]
277    .into_iter()
278    .filter_map(|path| {
279        path.strip_prefix(repo_root)
280            .ok()
281            .map(|value| value.to_string_lossy().replace('\\', "/"))
282    })
283    .collect::<Vec<_>>();
284    let starting_worktree = git::worktree_state_excluding(repo_root, &runtime_exclusions)?;
285    repo.worktree_clean = starting_worktree.clean;
286    repo.staged_change_count = starting_worktree.staged_change_count;
287    repo.modified_tracked_file_count = starting_worktree.modified_tracked_file_count;
288    repo.untracked_file_count = starting_worktree.untracked_file_count;
289    repo.worktree_state_digest = starting_worktree.digest;
290    if repo.is_shallow && !allow_shallow {
291        bail!(
292            "repository history is shallow; rerun with git slop find --allow-shallow to acknowledge incomplete history"
293        );
294    }
295    let all_tracked_paths = git::list_tracked_files(repo_root)?;
296    let scope = normalize_scope(scope)?;
297    if let Some(scope) = scope.as_deref() {
298        if fs::symlink_metadata(repo_root.join(scope)).is_err() {
299            bail!("--scope does not exist in the repository: {scope}");
300        }
301    }
302    let mut tracked_paths = all_tracked_paths
303        .iter()
304        .filter(|path| {
305            scope
306                .as_deref()
307                .is_none_or(|scope| *path == scope || path.starts_with(&format!("{scope}/")))
308        })
309        .cloned()
310        .collect::<Vec<_>>();
311    if tracked_paths.is_empty() && !allow_empty_scope {
312        bail!(
313            "{} selected no tracked paths; pass --allow-empty-scope only when an empty report is intentional",
314            scope.as_deref().map_or_else(
315                || "repository".to_string(),
316                |scope| format!("--scope {scope:?}")
317            )
318        );
319    }
320    let original_selected_path_count = tracked_paths.len();
321    let initial_estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
322    if initial_estimate.estimated_peak_memory_bytes > initial_estimate.memory_budget_bytes
323        && options.allow_degraded
324    {
325        let mut low = 0usize;
326        let mut high = tracked_paths.len();
327        while low < high {
328            let middle = (low + high).div_ceil(2);
329            let candidate = estimate::build(repo_root, &tracked_paths[..middle], &loaded_config);
330            if candidate.estimated_peak_memory_bytes <= candidate.memory_budget_bytes {
331                low = middle;
332            } else {
333                high = middle.saturating_sub(1);
334            }
335        }
336        tracked_paths = balanced_path_sample(&tracked_paths, low);
337    }
338    let scope_identity = ScopeIdentity {
339        mode: if scope.is_some() {
340            "scoped"
341        } else {
342            "repository"
343        }
344        .to_string(),
345        path: scope.clone(),
346        selected_path_count: tracked_paths.len(),
347        selected_path_digest: selected_path_digest(&tracked_paths),
348    };
349    let starting_content_digest = selected_content_digest(repo_root, &tracked_paths)?;
350    let estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
351    if estimate.estimated_peak_memory_bytes > estimate.memory_budget_bytes {
352        return Err(ClassifiedError::new(
353            ErrorKind::ResourceLimit,
354            "estimated_memory_budget_exceeded",
355            format!(
356                "analysis bounded before inventory: estimated {} MiB exceeds resources.memory_budget_mb={}; narrow --scope, use --allow-degraded, or raise the explicit budget",
357                estimate.estimated_peak_memory_bytes.div_ceil(1024 * 1024),
358                estimate.memory_budget_bytes / 1024 / 1024
359            ),
360        )
361        .at("/resources/memory_budget_mb")
362        .into());
363    }
364    let mut measured_peak_rss_bytes = None;
365    let mut memory_budget_exceeded_checkpoints = Vec::new();
366    measure_rss_checkpoint(
367        "pre_inventory",
368        estimate.memory_budget_bytes,
369        options.allow_degraded,
370        &mut measured_peak_rss_bytes,
371        &mut memory_budget_exceeded_checkpoints,
372    )?;
373    let (inventory_files, skipped) = inventory::build(repo_root, &tracked_paths, &loaded_config)?;
374    phase("inventory");
375    measure_rss_checkpoint(
376        "post_inventory",
377        estimate.memory_budget_bytes,
378        options.allow_degraded,
379        &mut measured_peak_rss_bytes,
380        &mut memory_budget_exceeded_checkpoints,
381    )?;
382    let encoder = configured_context_encoder(&loaded_config).map_err(|error| {
383        ClassifiedError::new(
384            ErrorKind::Contract,
385            "unsupported_tokenizer",
386            format!("{error:#}"),
387        )
388        .at("/tokenization/context_tokenizer_name")
389    })?;
390    let mut token_counts = BTreeMap::new();
391    let mut line_counts = BTreeMap::new();
392    let mut token_data = HashMap::new();
393    let tokenizer = config::pointer_str(&loaded_config, "/tokenization/context_tokenizer_name")
394        .unwrap_or("cl100k_base")
395        .to_string();
396    let large_file_bytes =
397        config::pointer_u64(&loaded_config, "/resources/large_file_bytes", 2_097_152) as usize;
398    let cache_path = state_root.join("cache").join("token-v4.sqlite3");
399    let mut cache_cleanup_warnings = Vec::new();
400    if !options.no_cache {
401        for version in ["token-v1", "token-v2", "token-v3"] {
402            let legacy_cache = state_root.join("cache").join(version);
403            if legacy_cache.exists() {
404                if let Err(error) = fs::remove_dir_all(&legacy_cache) {
405                    cache_cleanup_warnings.push(format!(
406                        "failed to remove legacy cache {}: {error}",
407                        legacy_cache.display()
408                    ));
409                }
410            }
411        }
412    }
413    let mut cache = if options.no_cache {
414        None
415    } else {
416        match TokenCache::open(&cache_path) {
417            Ok(cache) => Some(cache),
418            Err(error) => {
419                cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
420                TokenCache::open(&cache_path).ok()
421            }
422        }
423    };
424    let mut cache_hits = 0usize;
425    let mut cache_misses = 0usize;
426    let mut structurally_skipped_large_files = 0usize;
427    let mut intentionally_skipped_non_text_files = 0usize;
428    let mut incomplete_inventory_files = 0usize;
429    for file in &inventory_files {
430        if file.skipped_reason.as_deref() == Some("large_file_limit") {
431            structurally_skipped_large_files += 1;
432        }
433        if matches!(
434            file.skipped_reason.as_deref(),
435            Some("binary" | "gitlink" | "undecodable")
436        ) {
437            intentionally_skipped_non_text_files += 1;
438        } else if file.analysis_status != "analyzed"
439            && file.skipped_reason.as_deref() != Some("large_file_limit")
440        {
441            incomplete_inventory_files += 1;
442        }
443        if file.analysis_status != "analyzed" {
444            let conservative_tokens = if file.skipped_reason.as_deref() == Some("large_file_limit")
445            {
446                file.bytes.div_ceil(4)
447            } else {
448                0
449            };
450            token_counts.insert(file.path.clone(), conservative_tokens);
451            line_counts.insert(file.path.clone(), 0);
452            token_data.insert(
453                file.path.clone(),
454                (
455                    0,
456                    Vec::new(),
457                    Vec::new(),
458                    format!(
459                        "incomplete:{}:{}",
460                        file.skipped_reason.as_deref().unwrap_or("unknown"),
461                        file.bytes
462                    ),
463                ),
464            );
465            continue;
466        }
467        let mode = structural_mode(&file.path);
468        let cache_key = token_cache_key(&file.text, &tokenizer, large_file_bytes, mode);
469        let cached_value = if let Some(active_cache) = cache.as_ref() {
470            match active_cache.get(&cache_key) {
471                Ok(value) => value,
472                Err(error) => {
473                    cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
474                    cache = TokenCache::open(&cache_path).ok();
475                    None
476                }
477            }
478        } else {
479            None
480        };
481        let cached = if let Some(cached) = cached_value {
482            cache_hits += 1;
483            cached
484        } else {
485            cache_misses += 1;
486            let cached = CachedTokenData {
487                token_count: encoder.encode_ordinary(&file.text).len(),
488                structural_tokens: if file.bytes > large_file_bytes {
489                    Vec::new()
490                } else {
491                    structural_content_tokens(mode, &file.text)
492                },
493                content_fingerprint: content_fingerprint(&file.text),
494            };
495            let put_error = cache
496                .as_ref()
497                .and_then(|active_cache| active_cache.put(&cache_key, &cached).err());
498            if let Some(error) = put_error {
499                cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
500                cache = TokenCache::open(&cache_path).ok();
501            }
502            cached
503        };
504        let count = cached.token_count;
505        let mut structural = cached.structural_tokens;
506        if file.bytes <= large_file_bytes {
507            structural.extend(structural_path_tokens(&file.path));
508        }
509        let top_term_limit =
510            config::pointer_u64(&loaded_config, "/semantic_drift/top_term_limit", 25) as usize;
511        let fingerprint = cached.content_fingerprint;
512        token_counts.insert(file.path.clone(), count);
513        line_counts.insert(file.path.clone(), file.lines);
514        token_data.insert(
515            file.path.clone(),
516            (
517                structural.len(),
518                top_terms(&structural, &file.language, &file.text, top_term_limit),
519                structural,
520                fingerprint,
521            ),
522        );
523    }
524    let cache_stats = if let Some(cache) = &cache {
525        match cache.enforce_limits(
526            config::pointer_u64(&loaded_config, "/resources/cache_max_entries", 10_000) as usize,
527            config::pointer_u64(&loaded_config, "/resources/cache_max_bytes", 536_870_912),
528        ) {
529            Ok(stats) => stats,
530            Err(error) => {
531                cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
532                CacheStats::default()
533            }
534        }
535    } else {
536        CacheStats::default()
537    };
538    phase("tokenization");
539    measure_rss_checkpoint(
540        "post_tokenization",
541        estimate.memory_budget_bytes,
542        options.allow_degraded,
543        &mut measured_peak_rss_bytes,
544        &mut memory_budget_exceeded_checkpoints,
545    )?;
546    repo.analyzed_content_digest = Some(starting_content_digest.clone());
547    let analyzed_paths: Vec<String> = inventory_files
548        .iter()
549        .map(|file| file.path.clone())
550        .collect();
551    let now = options.as_of.unwrap_or_else(Utc::now);
552    let (history_by_path, commits, history_diagnostics) = history::analyze_history(
553        repo_root,
554        &analyzed_paths,
555        &token_counts,
556        &line_counts,
557        &loaded_config,
558        now,
559    )?;
560    phase("history");
561    measure_rss_checkpoint(
562        "post_history",
563        estimate.memory_budget_bytes,
564        options.allow_degraded,
565        &mut measured_peak_rss_bytes,
566        &mut memory_budget_exceeded_checkpoints,
567    )?;
568    let mut files = Vec::with_capacity(inventory_files.len());
569    for file in inventory_files {
570        let tokens = token_counts.get(&file.path).copied().unwrap_or_default();
571        let inline_tests = has_inline_tests(&file.language, &file.text);
572        let (structural_token_count, top_structural_terms, structural_tokens, content_fingerprint) =
573            token_data
574                .remove(&file.path)
575                .unwrap_or_else(|| (0, Vec::new(), Vec::new(), String::new()));
576        let history = history_by_path.get(&file.path).cloned().unwrap_or_default();
577        let categories = structural_categories(structural_mode(&file.path), &file.text);
578        files.push(FileAnalysis {
579            path: file.path,
580            bytes: file.bytes,
581            lines: file.lines,
582            blank_lines: file.blank_lines,
583            code_lines: file.code_lines,
584            comment_lines: file.comment_lines,
585            language: file.language,
586            profile: file.profile.clone(),
587            classification: file.classification,
588            generated_from: file.generated_from,
589            generated_provenance: file.generated_provenance,
590            analysis_status: file.analysis_status,
591            skipped_reason: file.skipped_reason,
592            symlink_metadata: file.symlink_metadata,
593            has_inline_tests: inline_tests,
594            tokens,
595            context_band: scoring::context_band_for_profile(tokens, &file.profile, &loaded_config),
596            context_pressure: scoring::context_pressure_for_profile(
597                tokens,
598                &file.profile,
599                &loaded_config,
600            ),
601            content_fingerprint,
602            content_sha256: file.content_sha256,
603            structural_tokens,
604            structural_token_count,
605            top_structural_terms,
606            structural_categories: categories,
607            age_days: history.age_days,
608            revisions_window: history.revisions_window,
609            recency_weighted_commits: history.recency_weighted_commits,
610            added_window: history.added_window,
611            deleted_window: history.deleted_window,
612            churn_lines_window: history.line_churn_window,
613            line_churn_window: history.line_churn_window,
614            token_churn_window: history.token_churn_window,
615            relative_churn_window: history.relative_churn_window,
616            late_churn_spike: history.late_churn_spike,
617            author_count_window: history.author_count_window,
618            author_entropy: history.author_entropy,
619            top_author_share: history.top_author_share,
620            days_since_non_bot_edit: history.days_since_non_bot_edit,
621            recent_maintainer_diversity: history.recent_maintainer_diversity,
622            age_pressure: 0.0,
623            revision_norm: 0.0,
624            relative_churn_norm: 0.0,
625            churn_pressure: 0.0,
626            slop_score: 0.0,
627            slop_band: "low".to_string(),
628            reason_codes: Vec::new(),
629            costs: json!({}),
630            overlays: json!({}),
631        });
632    }
633    let history_evidence_reliable = !repo.is_shallow
634        && !history_diagnostics
635            .get("history_cap_reached")
636            .and_then(Value::as_bool)
637            .unwrap_or(false)
638        && ![
639            "full_history_cap_status",
640            "window_status_cap_status",
641            "window_numstat_cap_status",
642        ]
643        .into_iter()
644        .any(|field| history_diagnostics.get(field).and_then(Value::as_str) == Some("truncated"));
645    scoring::apply_scoring_with_evidence(&mut files, &loaded_config, history_evidence_reliable);
646    let organization = overlays::analyze(&mut files, &commits, &loaded_config)?;
647    phase("relationships");
648    measure_rss_checkpoint(
649        "post_relationships",
650        estimate.memory_budget_bytes,
651        options.allow_degraded,
652        &mut measured_peak_rss_bytes,
653        &mut memory_budget_exceeded_checkpoints,
654    )?;
655    let folders = scoring::build_folder_analyses(&files, &loaded_config);
656    let candidates = action_queue(&files, history_evidence_reliable, &loaded_config);
657    let (queue, observation_feed): (Vec<_>, Vec<_>) = candidates.into_iter().partition(|item| {
658        let classification = item
659            .get("classification")
660            .and_then(Value::as_str)
661            .unwrap_or("other");
662        let actionable = !matches!(
663            classification,
664            "generated" | "vendored" | "snapshot" | "fixture" | "migration_fixture"
665        );
666        let supported = item.get("evidence_status").and_then(Value::as_str) == Some("supported")
667            || item.get("is_pure_context_hotspot").and_then(Value::as_bool) == Some(true);
668        actionable
669            && supported
670            && matches!(
671                item.get("severity").and_then(Value::as_str),
672                Some("warning" | "error")
673            )
674    });
675    let generated_at = now.to_rfc3339_opts(SecondsFormat::Secs, true);
676    let analyzed_revision_at = repo.head_commit_timestamp.clone();
677    let ending_worktree = git::worktree_state_excluding(repo_root, &runtime_exclusions)?;
678    if ending_worktree.digest != repo.worktree_state_digest {
679        bail!("repository changed during analysis; no mixed-snapshot report was published");
680    }
681    if selected_content_digest(repo_root, &tracked_paths)? != starting_content_digest {
682        bail!(
683            "selected file content changed during analysis; no mixed-snapshot report was published"
684        );
685    }
686    let estimator_error_ratio = measured_peak_rss_bytes.map(|measured| {
687        let estimated = estimate.estimated_peak_memory_bytes.max(1) as f64;
688        ((measured as f64 - estimated) / estimated * 1_000_000.0).round() / 1_000_000.0
689    });
690    let estimate_range_contains_measurement = measured_peak_rss_bytes.map(|measured| {
691        let measured = u128::from(measured);
692        measured >= estimate.estimated_peak_memory_low_bytes
693            && measured <= estimate.estimated_peak_memory_high_bytes
694    });
695    let history_evidence_status = if repo.head_commit.is_none() {
696        "not_applicable_unborn"
697    } else if history_evidence_reliable {
698        "supported_with_per_file_shrinkage"
699    } else {
700        "incomplete_suppressed"
701    };
702    let analysis = Analysis {
703        output_root,
704        report_profile: if options.report_profile.is_empty() {
705            "standard".to_string()
706        } else {
707            options.report_profile.clone()
708        },
709        compression: if options.compression.is_empty() {
710            "none".to_string()
711        } else {
712            options.compression.clone()
713        },
714        repo,
715        config: loaded_config,
716        generated_at,
717        analyzed_revision_at,
718        skipped,
719        tracked_file_count: all_tracked_paths.len(),
720        scope: scope_identity,
721        files,
722        folders,
723        organization,
724        action_queue: queue,
725        observation_feed,
726        diagnostics: json!({
727            "analysis_elapsed_ms_before_report": started.elapsed().as_millis(),
728            "estimate": estimate,
729            "measured_peak_rss_bytes": measured_peak_rss_bytes,
730            "estimator_error_ratio": estimator_error_ratio,
731            "estimate_range_contains_measurement": estimate_range_contains_measurement,
732            "memory_budget_exceeded_checkpoints": memory_budget_exceeded_checkpoints,
733            "memory_measurement_status": if measured_peak_rss_bytes.is_some() { "measured" } else { "unsupported" },
734            "cache_hits": cache_hits,
735            "cache_misses": cache_misses,
736            "cache_entries": cache_stats.entries,
737            "cache_bytes": cache_stats.bytes,
738            "cache_failed_evictions": cache_stats.failed_evictions,
739            "cache_cleanup_warnings": cache_cleanup_warnings,
740            "cache_status": if options.no_cache { "disabled" } else { "enabled" },
741            "structurally_skipped_large_files": structurally_skipped_large_files,
742            "intentionally_skipped_non_text_files": intentionally_skipped_non_text_files,
743            "incomplete_inventory_files": incomplete_inventory_files,
744            "analysis_status": if tracked_paths.len() < original_selected_path_count || !memory_budget_exceeded_checkpoints.is_empty() { "degraded_resource_budget" } else if structurally_skipped_large_files > 0 { "degraded_large_files" } else if incomplete_inventory_files > 0 { "degraded_incomplete_inventory" } else { "complete" },
745            "resource_mode": if tracked_paths.len() < original_selected_path_count { "degraded_path_prefix" } else if !memory_budget_exceeded_checkpoints.is_empty() { "degraded_measured_rss" } else { "complete" },
746            "original_selected_path_count": original_selected_path_count,
747            "degraded_omitted_path_count": original_selected_path_count.saturating_sub(tracked_paths.len()),
748            "history": history_diagnostics,
749            "history_evidence_status": history_evidence_status,
750            "scope": scope
751        }),
752    };
753    let rollup = health::build_health_rollup(&analysis);
754    let result = report::write_report_bundle(&analysis, &rollup)?;
755    phase("report writing");
756    if result.report.get("schema_version").and_then(Value::as_u64) != Some(5) {
757        bail!("internal error: report writer did not produce schema 5");
758    }
759    Ok(result)
760}
761
762include!("analyze/tests.rs");