Skip to main content

git_slop/
analyze.rs

1use std::collections::{BTreeMap, HashMap};
2use std::fs;
3use std::path::{Component, Path, PathBuf};
4use std::sync::LazyLock;
5use std::time::{Duration, Instant};
6
7use anyhow::{Context, Result, bail};
8use chrono::{DateTime, SecondsFormat, Utc};
9use regex::Regex;
10use rusqlite::{Connection, OptionalExtension, params};
11use serde::{Deserialize, Serialize};
12use serde_json::{Value, json};
13use sha2::{Digest, Sha256};
14use tiktoken_rs::{
15    CoreBPE, cl100k_base, o200k_base, o200k_harmony, p50k_base, p50k_edit, r50k_base,
16};
17use unicode_normalization::UnicodeNormalization;
18use unicode_segmentation::UnicodeSegmentation;
19
20use crate::config;
21use crate::error::{ClassifiedError, ErrorKind};
22use crate::estimate;
23use crate::git;
24use crate::health;
25use crate::history;
26use crate::inventory;
27use crate::model::{Analysis, FileAnalysis, FindResult, ScopeIdentity};
28use crate::overlays;
29use crate::report;
30use crate::scoring;
31
32static CAMEL_CASE_RE: LazyLock<Regex> =
33    LazyLock::new(|| Regex::new(r"([a-z0-9])([A-Z])").expect("valid camel-case regex"));
34static ACRONYM_BOUNDARY_RE: LazyLock<Regex> =
35    LazyLock::new(|| Regex::new(r"([A-Z]+)([A-Z][a-z])").expect("valid acronym regex"));
36static NUMBER_RE: LazyLock<Regex> =
37    LazyLock::new(|| Regex::new(r"\b\d+(?:\.\d+)?\b").expect("valid number regex"));
38include!("analyze/cache.rs");
39include!("analyze/structural.rs");
40pub fn run_find() -> Result<FindResult> {
41    let repo_root = git::resolve_repo_root()?;
42    run_find_in(&repo_root)
43}
44
45pub fn run_find_in(repo_root: &Path) -> Result<FindResult> {
46    run_find_scoped(repo_root, false, None, false)
47}
48
49pub fn run_find_in_with_options(repo_root: &Path, allow_shallow: bool) -> Result<FindResult> {
50    run_find_scoped(repo_root, allow_shallow, None, false)
51}
52
53#[derive(Debug, Clone, Default)]
54pub struct FindOptions {
55    pub allow_shallow: bool,
56    pub scope: Option<String>,
57    pub progress: bool,
58    pub allow_empty_scope: bool,
59    pub state_dir: Option<PathBuf>,
60    pub output_dir: Option<PathBuf>,
61    pub no_cache: bool,
62    pub allow_degraded: bool,
63    pub as_of: Option<DateTime<Utc>>,
64    pub report_profile: String,
65    pub compression: String,
66}
67
68pub(crate) fn normalize_scope(value: Option<&str>) -> Result<Option<String>> {
69    let Some(raw) = value.map(str::trim) else {
70        return Ok(None);
71    };
72    if raw.is_empty() || raw == "." {
73        return Ok(None);
74    }
75    let path = Path::new(raw);
76    if path.is_absolute() {
77        bail!("--scope must be repo-relative, received {raw:?}");
78    }
79    let mut parts = Vec::new();
80    for component in path.components() {
81        match component {
82            Component::Normal(part) => parts.push(
83                part.to_str()
84                    .ok_or_else(|| anyhow::anyhow!("--scope must be valid UTF-8"))?,
85            ),
86            Component::CurDir => {}
87            Component::ParentDir | Component::RootDir | Component::Prefix(_) => {
88                bail!("--scope must not escape the repository, received {raw:?}");
89            }
90        }
91    }
92    let normalized = parts.join("/");
93    Ok((!normalized.is_empty()).then_some(normalized))
94}
95
96fn selected_path_digest(paths: &[String]) -> String {
97    let mut digest = Sha256::new();
98    for path in paths {
99        digest.update(path.as_bytes());
100        digest.update([0]);
101    }
102    hex::encode(digest.finalize())
103}
104
105fn measure_rss_checkpoint(
106    checkpoint: &'static str,
107    memory_budget_bytes: u128,
108    allow_degraded: bool,
109    peak_rss_bytes: &mut Option<u64>,
110    exceeded_checkpoints: &mut Vec<&'static str>,
111) -> Result<()> {
112    let Some(rss_bytes) = estimate::current_rss_bytes() else {
113        return Ok(());
114    };
115    *peak_rss_bytes = Some(peak_rss_bytes.unwrap_or_default().max(rss_bytes));
116    if u128::from(rss_bytes) <= memory_budget_bytes {
117        return Ok(());
118    }
119    exceeded_checkpoints.push(checkpoint);
120    if allow_degraded {
121        return Err(ClassifiedError::new(
122            ErrorKind::ResourceLimit,
123            "degraded_memory_recovery_unavailable",
124            format!(
125                "analysis stopped at {checkpoint}: measured RSS {} MiB still exceeds resources.memory_budget_mb={} after deterministic degraded sampling; continuing would violate the memory contract",
126                rss_bytes.div_ceil(1024 * 1024),
127                memory_budget_bytes / 1024 / 1024
128            ),
129        )
130        .at("/resources/memory_budget_mb")
131        .into());
132    }
133    Err(ClassifiedError::new(
134        ErrorKind::ResourceLimit,
135        "measured_memory_budget_exceeded",
136        format!(
137            "analysis stopped at {checkpoint}: measured RSS {} MiB exceeds resources.memory_budget_mb={}; narrow --scope, use --allow-degraded, or raise the explicit budget",
138            rss_bytes.div_ceil(1024 * 1024),
139            memory_budget_bytes / 1024 / 1024
140        ),
141    )
142    .at("/resources/memory_budget_mb")
143    .into())
144}
145
146fn selected_content_digest(repo_root: &Path, paths: &[String]) -> Result<String> {
147    let mut digest = Sha256::new();
148    for path in paths {
149        digest.update(path.as_bytes());
150        digest.update([0]);
151        let absolute = repo_root.join(path);
152        let metadata = fs::symlink_metadata(&absolute)
153            .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?;
154        let bytes = if metadata.file_type().is_symlink() {
155            fs::read_link(&absolute)
156                .with_context(|| format!("selected tracked link changed or disappeared: {path}"))?
157                .to_string_lossy()
158                .into_owned()
159                .into_bytes()
160        } else if metadata.is_dir() {
161            b"<gitlink>".to_vec()
162        } else {
163            fs::read(&absolute)
164                .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?
165        };
166        digest.update(bytes);
167        digest.update([0]);
168    }
169    Ok(hex::encode(digest.finalize()))
170}
171
172fn balanced_path_sample(paths: &[String], limit: usize) -> Vec<String> {
173    let mut roots = BTreeMap::<&str, Vec<&String>>::new();
174    for path in paths {
175        roots
176            .entry(path.split('/').next().unwrap_or("."))
177            .or_default()
178            .push(path);
179    }
180    let mut selected = Vec::with_capacity(limit.min(paths.len()));
181    let mut offset = 0usize;
182    while selected.len() < limit {
183        let mut added = false;
184        for values in roots.values() {
185            if let Some(path) = values.get(offset) {
186                selected.push((*path).clone());
187                added = true;
188                if selected.len() == limit {
189                    break;
190                }
191            }
192        }
193        if !added {
194            break;
195        }
196        offset += 1;
197    }
198    selected.sort();
199    selected
200}
201
202pub fn run_find_scoped(
203    repo_root: &Path,
204    allow_shallow: bool,
205    scope: Option<&str>,
206    progress: bool,
207) -> Result<FindResult> {
208    run_find_with_options(
209        repo_root,
210        &FindOptions {
211            allow_shallow,
212            scope: scope.map(ToOwned::to_owned),
213            progress,
214            allow_empty_scope: false,
215            ..FindOptions::default()
216        },
217    )
218}
219
220pub fn run_find_with_options(repo_root: &Path, options: &FindOptions) -> Result<FindResult> {
221    let allow_shallow = options.allow_shallow;
222    let scope = options.scope.as_deref();
223    let progress = options.progress;
224    let allow_empty_scope = options.allow_empty_scope;
225    let resolve_root = |value: Option<&Path>, fallback: PathBuf| {
226        value.map_or(fallback, |path| {
227            if path.is_absolute() {
228                path.to_path_buf()
229            } else {
230                repo_root.join(path)
231            }
232        })
233    };
234    let state_root = resolve_root(options.state_dir.as_deref(), config::slop_dir(repo_root));
235    let output_root = resolve_root(options.output_dir.as_deref(), config::slop_dir(repo_root));
236    let started = Instant::now();
237    let phase = |name: &str| {
238        if progress {
239            eprintln!("git-slop: {name} ({:.1}s)", started.elapsed().as_secs_f64());
240        }
241    };
242    phase("preflight");
243    let _scan_lock = config::acquire_scan_lock(repo_root)?;
244    let loaded_config = config::load(repo_root).map_err(|error| {
245        ClassifiedError::new(
246            ErrorKind::Contract,
247            "invalid_configuration",
248            format!("{error:#}"),
249        )
250        .at("/.slop/config.yaml")
251    })?;
252    let mut repo = git::repo_metadata(repo_root)?;
253    let runtime_exclusions = [
254        state_root.join("cache"),
255        output_root.join("latest"),
256        output_root.join("runs"),
257    ]
258    .into_iter()
259    .filter_map(|path| {
260        path.strip_prefix(repo_root)
261            .ok()
262            .map(|value| value.to_string_lossy().replace('\\', "/"))
263    })
264    .collect::<Vec<_>>();
265    let starting_worktree = git::worktree_state_excluding(repo_root, &runtime_exclusions)?;
266    repo.worktree_clean = starting_worktree.clean;
267    repo.staged_change_count = starting_worktree.staged_change_count;
268    repo.modified_tracked_file_count = starting_worktree.modified_tracked_file_count;
269    repo.untracked_file_count = starting_worktree.untracked_file_count;
270    repo.worktree_state_digest = starting_worktree.digest;
271    if repo.is_shallow && !allow_shallow {
272        bail!(
273            "repository history is shallow; rerun with git slop find --allow-shallow to acknowledge incomplete history"
274        );
275    }
276    let all_tracked_paths = git::list_tracked_files(repo_root)?;
277    let scope = normalize_scope(scope)?;
278    if let Some(scope) = scope.as_deref() {
279        if fs::symlink_metadata(repo_root.join(scope)).is_err() {
280            bail!("--scope does not exist in the repository: {scope}");
281        }
282    }
283    let mut tracked_paths = all_tracked_paths
284        .iter()
285        .filter(|path| {
286            scope
287                .as_deref()
288                .is_none_or(|scope| *path == scope || path.starts_with(&format!("{scope}/")))
289        })
290        .cloned()
291        .collect::<Vec<_>>();
292    if tracked_paths.is_empty() && !allow_empty_scope {
293        bail!(
294            "{} selected no tracked paths; pass --allow-empty-scope only when an empty report is intentional",
295            scope.as_deref().map_or_else(
296                || "repository".to_string(),
297                |scope| format!("--scope {scope:?}")
298            )
299        );
300    }
301    let original_selected_path_count = tracked_paths.len();
302    let initial_estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
303    if initial_estimate.estimated_peak_memory_bytes > initial_estimate.memory_budget_bytes
304        && options.allow_degraded
305    {
306        let mut low = 0usize;
307        let mut high = tracked_paths.len();
308        while low < high {
309            let middle = (low + high).div_ceil(2);
310            let candidate = estimate::build(repo_root, &tracked_paths[..middle], &loaded_config);
311            if candidate.estimated_peak_memory_bytes <= candidate.memory_budget_bytes {
312                low = middle;
313            } else {
314                high = middle.saturating_sub(1);
315            }
316        }
317        tracked_paths = balanced_path_sample(&tracked_paths, low);
318    }
319    let scope_identity = ScopeIdentity {
320        mode: if scope.is_some() {
321            "scoped"
322        } else {
323            "repository"
324        }
325        .to_string(),
326        path: scope.clone(),
327        selected_path_count: tracked_paths.len(),
328        selected_path_digest: selected_path_digest(&tracked_paths),
329    };
330    let starting_content_digest = selected_content_digest(repo_root, &tracked_paths)?;
331    let estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
332    if estimate.estimated_peak_memory_bytes > estimate.memory_budget_bytes {
333        return Err(ClassifiedError::new(
334            ErrorKind::ResourceLimit,
335            "estimated_memory_budget_exceeded",
336            format!(
337                "analysis bounded before inventory: estimated {} MiB exceeds resources.memory_budget_mb={}; narrow --scope, use --allow-degraded, or raise the explicit budget",
338                estimate.estimated_peak_memory_bytes.div_ceil(1024 * 1024),
339                estimate.memory_budget_bytes / 1024 / 1024
340            ),
341        )
342        .at("/resources/memory_budget_mb")
343        .into());
344    }
345    let mut measured_peak_rss_bytes = None;
346    let mut memory_budget_exceeded_checkpoints = Vec::new();
347    measure_rss_checkpoint(
348        "pre_inventory",
349        estimate.memory_budget_bytes,
350        options.allow_degraded,
351        &mut measured_peak_rss_bytes,
352        &mut memory_budget_exceeded_checkpoints,
353    )?;
354    let (inventory_files, skipped) = inventory::build(repo_root, &tracked_paths, &loaded_config)?;
355    phase("inventory");
356    measure_rss_checkpoint(
357        "post_inventory",
358        estimate.memory_budget_bytes,
359        options.allow_degraded,
360        &mut measured_peak_rss_bytes,
361        &mut memory_budget_exceeded_checkpoints,
362    )?;
363    let encoder = configured_context_encoder(&loaded_config).map_err(|error| {
364        ClassifiedError::new(
365            ErrorKind::Contract,
366            "unsupported_tokenizer",
367            format!("{error:#}"),
368        )
369        .at("/tokenization/context_tokenizer_name")
370    })?;
371    let mut token_counts = BTreeMap::new();
372    let mut line_counts = BTreeMap::new();
373    let mut token_data = HashMap::new();
374    let tokenizer = config::pointer_str(&loaded_config, "/tokenization/context_tokenizer_name")
375        .unwrap_or("cl100k_base")
376        .to_string();
377    let large_file_bytes =
378        config::pointer_u64(&loaded_config, "/resources/large_file_bytes", 2_097_152) as usize;
379    let cache_path = state_root.join("cache").join("token-v4.sqlite3");
380    let mut cache_cleanup_warnings = Vec::new();
381    if !options.no_cache {
382        for version in ["token-v1", "token-v2", "token-v3"] {
383            let legacy_cache = state_root.join("cache").join(version);
384            if legacy_cache.exists() {
385                if let Err(error) = fs::remove_dir_all(&legacy_cache) {
386                    cache_cleanup_warnings.push(format!(
387                        "failed to remove legacy cache {}: {error}",
388                        legacy_cache.display()
389                    ));
390                }
391            }
392        }
393    }
394    let mut cache = if options.no_cache {
395        None
396    } else {
397        match TokenCache::open(&cache_path) {
398            Ok(cache) => Some(cache),
399            Err(error) => {
400                cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
401                TokenCache::open(&cache_path).ok()
402            }
403        }
404    };
405    let mut cache_hits = 0usize;
406    let mut cache_misses = 0usize;
407    let mut structurally_skipped_large_files = 0usize;
408    let mut intentionally_skipped_non_text_files = 0usize;
409    let mut incomplete_inventory_files = 0usize;
410    for file in &inventory_files {
411        if file.skipped_reason.as_deref() == Some("large_file_limit") {
412            structurally_skipped_large_files += 1;
413        }
414        if matches!(
415            file.skipped_reason.as_deref(),
416            Some("binary" | "gitlink" | "undecodable")
417        ) {
418            intentionally_skipped_non_text_files += 1;
419        } else if file.analysis_status != "analyzed"
420            && file.skipped_reason.as_deref() != Some("large_file_limit")
421        {
422            incomplete_inventory_files += 1;
423        }
424        if file.analysis_status != "analyzed" {
425            let conservative_tokens = if file.skipped_reason.as_deref() == Some("large_file_limit")
426            {
427                file.bytes.div_ceil(4)
428            } else {
429                0
430            };
431            token_counts.insert(file.path.clone(), conservative_tokens);
432            line_counts.insert(file.path.clone(), 0);
433            token_data.insert(
434                file.path.clone(),
435                (
436                    0,
437                    Vec::new(),
438                    Vec::new(),
439                    format!(
440                        "incomplete:{}:{}",
441                        file.skipped_reason.as_deref().unwrap_or("unknown"),
442                        file.bytes
443                    ),
444                ),
445            );
446            continue;
447        }
448        let mode = structural_mode(&file.path);
449        let cache_key = token_cache_key(&file.text, &tokenizer, large_file_bytes, mode);
450        let cached_value = if let Some(active_cache) = cache.as_ref() {
451            match active_cache.get(&cache_key) {
452                Ok(value) => value,
453                Err(error) => {
454                    cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
455                    cache = TokenCache::open(&cache_path).ok();
456                    None
457                }
458            }
459        } else {
460            None
461        };
462        let cached = if let Some(cached) = cached_value {
463            cache_hits += 1;
464            cached
465        } else {
466            cache_misses += 1;
467            let cached = CachedTokenData {
468                token_count: encoder.encode_ordinary(&file.text).len(),
469                structural_tokens: if file.bytes > large_file_bytes {
470                    Vec::new()
471                } else {
472                    structural_content_tokens(mode, &file.text)
473                },
474                content_fingerprint: content_fingerprint(&file.text),
475            };
476            let put_error = cache
477                .as_ref()
478                .and_then(|active_cache| active_cache.put(&cache_key, &cached).err());
479            if let Some(error) = put_error {
480                cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
481                cache = TokenCache::open(&cache_path).ok();
482            }
483            cached
484        };
485        let count = cached.token_count;
486        let mut structural = cached.structural_tokens;
487        if file.bytes <= large_file_bytes {
488            structural.extend(structural_path_tokens(&file.path));
489        }
490        let top_term_limit =
491            config::pointer_u64(&loaded_config, "/semantic_drift/top_term_limit", 25) as usize;
492        let fingerprint = cached.content_fingerprint;
493        token_counts.insert(file.path.clone(), count);
494        line_counts.insert(file.path.clone(), file.lines);
495        token_data.insert(
496            file.path.clone(),
497            (
498                structural.len(),
499                top_terms(&structural, top_term_limit),
500                structural,
501                fingerprint,
502            ),
503        );
504    }
505    let cache_stats = if let Some(cache) = &cache {
506        match cache.enforce_limits(
507            config::pointer_u64(&loaded_config, "/resources/cache_max_entries", 10_000) as usize,
508            config::pointer_u64(&loaded_config, "/resources/cache_max_bytes", 536_870_912),
509        ) {
510            Ok(stats) => stats,
511            Err(error) => {
512                cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
513                CacheStats::default()
514            }
515        }
516    } else {
517        CacheStats::default()
518    };
519    phase("tokenization");
520    measure_rss_checkpoint(
521        "post_tokenization",
522        estimate.memory_budget_bytes,
523        options.allow_degraded,
524        &mut measured_peak_rss_bytes,
525        &mut memory_budget_exceeded_checkpoints,
526    )?;
527    repo.analyzed_content_digest = Some(starting_content_digest.clone());
528    let analyzed_paths: Vec<String> = inventory_files
529        .iter()
530        .map(|file| file.path.clone())
531        .collect();
532    let now = options.as_of.unwrap_or_else(Utc::now);
533    let (history_by_path, commits, history_diagnostics) = history::analyze_history(
534        repo_root,
535        &analyzed_paths,
536        &token_counts,
537        &line_counts,
538        &loaded_config,
539        now,
540    )?;
541    phase("history");
542    measure_rss_checkpoint(
543        "post_history",
544        estimate.memory_budget_bytes,
545        options.allow_degraded,
546        &mut measured_peak_rss_bytes,
547        &mut memory_budget_exceeded_checkpoints,
548    )?;
549    let mut files = Vec::with_capacity(inventory_files.len());
550    for file in inventory_files {
551        let tokens = token_counts.get(&file.path).copied().unwrap_or_default();
552        let inline_tests = has_inline_tests(&file.language, &file.text);
553        let (structural_token_count, top_structural_terms, structural_tokens, content_fingerprint) =
554            token_data
555                .remove(&file.path)
556                .unwrap_or_else(|| (0, Vec::new(), Vec::new(), String::new()));
557        let history = history_by_path.get(&file.path).cloned().unwrap_or_default();
558        let categories = structural_categories(structural_mode(&file.path), &file.text);
559        files.push(FileAnalysis {
560            path: file.path,
561            bytes: file.bytes,
562            lines: file.lines,
563            blank_lines: file.blank_lines,
564            code_lines: file.code_lines,
565            comment_lines: file.comment_lines,
566            language: file.language,
567            profile: file.profile.clone(),
568            classification: file.classification,
569            generated_from: file.generated_from,
570            analysis_status: file.analysis_status,
571            skipped_reason: file.skipped_reason,
572            symlink_metadata: file.symlink_metadata,
573            has_inline_tests: inline_tests,
574            tokens,
575            context_band: scoring::context_band_for_profile(tokens, &file.profile, &loaded_config),
576            context_pressure: scoring::context_pressure_for_profile(
577                tokens,
578                &file.profile,
579                &loaded_config,
580            ),
581            content_fingerprint,
582            content_sha256: file.content_sha256,
583            structural_tokens,
584            structural_token_count,
585            top_structural_terms,
586            structural_categories: categories,
587            age_days: history.age_days,
588            revisions_window: history.revisions_window,
589            recency_weighted_commits: history.recency_weighted_commits,
590            added_window: history.added_window,
591            deleted_window: history.deleted_window,
592            churn_lines_window: history.line_churn_window,
593            line_churn_window: history.line_churn_window,
594            token_churn_window: history.token_churn_window,
595            relative_churn_window: history.relative_churn_window,
596            late_churn_spike: history.late_churn_spike,
597            author_count_window: history.author_count_window,
598            author_entropy: history.author_entropy,
599            top_author_share: history.top_author_share,
600            days_since_non_bot_edit: history.days_since_non_bot_edit,
601            recent_maintainer_diversity: history.recent_maintainer_diversity,
602            age_pressure: 0.0,
603            revision_norm: 0.0,
604            relative_churn_norm: 0.0,
605            churn_pressure: 0.0,
606            slop_score: 0.0,
607            slop_band: "low".to_string(),
608            reason_codes: Vec::new(),
609            costs: json!({}),
610            overlays: json!({}),
611        });
612    }
613    let history_evidence_reliable = !repo.is_shallow
614        && !history_diagnostics
615            .get("history_cap_reached")
616            .and_then(Value::as_bool)
617            .unwrap_or(false)
618        && ![
619            "full_history_cap_status",
620            "window_status_cap_status",
621            "window_numstat_cap_status",
622        ]
623        .into_iter()
624        .any(|field| history_diagnostics.get(field).and_then(Value::as_str) == Some("truncated"));
625    scoring::apply_scoring_with_evidence(&mut files, &loaded_config, history_evidence_reliable);
626    let organization = overlays::analyze(&mut files, &commits, &loaded_config)?;
627    phase("relationships");
628    measure_rss_checkpoint(
629        "post_relationships",
630        estimate.memory_budget_bytes,
631        options.allow_degraded,
632        &mut measured_peak_rss_bytes,
633        &mut memory_budget_exceeded_checkpoints,
634    )?;
635    let folders = scoring::build_folder_analyses(&files, &loaded_config);
636    let candidates = action_queue(&files, history_evidence_reliable, &loaded_config);
637    let (queue, observation_feed): (Vec<_>, Vec<_>) = candidates.into_iter().partition(|item| {
638        let classification = item
639            .get("classification")
640            .and_then(Value::as_str)
641            .unwrap_or("other");
642        let actionable = !matches!(
643            classification,
644            "generated" | "vendored" | "snapshot" | "fixture" | "migration_fixture"
645        );
646        let supported = item.get("evidence_status").and_then(Value::as_str) == Some("supported")
647            || item.get("is_pure_context_hotspot").and_then(Value::as_bool) == Some(true);
648        actionable
649            && supported
650            && matches!(
651                item.get("severity").and_then(Value::as_str),
652                Some("warning" | "error")
653            )
654    });
655    let generated_at = now.to_rfc3339_opts(SecondsFormat::Secs, true);
656    let analyzed_revision_at = repo.head_commit_timestamp.clone();
657    let ending_worktree = git::worktree_state_excluding(repo_root, &runtime_exclusions)?;
658    if ending_worktree.digest != repo.worktree_state_digest {
659        bail!("repository changed during analysis; no mixed-snapshot report was published");
660    }
661    if selected_content_digest(repo_root, &tracked_paths)? != starting_content_digest {
662        bail!(
663            "selected file content changed during analysis; no mixed-snapshot report was published"
664        );
665    }
666    let estimator_error_ratio = measured_peak_rss_bytes.map(|measured| {
667        let estimated = estimate.estimated_peak_memory_bytes.max(1) as f64;
668        ((measured as f64 - estimated) / estimated * 1_000_000.0).round() / 1_000_000.0
669    });
670    let estimate_range_contains_measurement = measured_peak_rss_bytes.map(|measured| {
671        let measured = u128::from(measured);
672        measured >= estimate.estimated_peak_memory_low_bytes
673            && measured <= estimate.estimated_peak_memory_high_bytes
674    });
675    let history_evidence_status = if repo.head_commit.is_none() {
676        "not_applicable_unborn"
677    } else if history_evidence_reliable {
678        "supported_with_per_file_shrinkage"
679    } else {
680        "incomplete_suppressed"
681    };
682    let analysis = Analysis {
683        output_root,
684        report_profile: if options.report_profile.is_empty() {
685            "standard".to_string()
686        } else {
687            options.report_profile.clone()
688        },
689        compression: if options.compression.is_empty() {
690            "none".to_string()
691        } else {
692            options.compression.clone()
693        },
694        repo,
695        config: loaded_config,
696        generated_at,
697        analyzed_revision_at,
698        skipped,
699        tracked_file_count: all_tracked_paths.len(),
700        scope: scope_identity,
701        files,
702        folders,
703        organization,
704        action_queue: queue,
705        observation_feed,
706        diagnostics: json!({
707            "analysis_elapsed_ms_before_report": started.elapsed().as_millis(),
708            "estimate": estimate,
709            "measured_peak_rss_bytes": measured_peak_rss_bytes,
710            "estimator_error_ratio": estimator_error_ratio,
711            "estimate_range_contains_measurement": estimate_range_contains_measurement,
712            "memory_budget_exceeded_checkpoints": memory_budget_exceeded_checkpoints,
713            "memory_measurement_status": if measured_peak_rss_bytes.is_some() { "measured" } else { "unsupported" },
714            "cache_hits": cache_hits,
715            "cache_misses": cache_misses,
716            "cache_entries": cache_stats.entries,
717            "cache_bytes": cache_stats.bytes,
718            "cache_failed_evictions": cache_stats.failed_evictions,
719            "cache_cleanup_warnings": cache_cleanup_warnings,
720            "cache_status": if options.no_cache { "disabled" } else { "enabled" },
721            "structurally_skipped_large_files": structurally_skipped_large_files,
722            "intentionally_skipped_non_text_files": intentionally_skipped_non_text_files,
723            "incomplete_inventory_files": incomplete_inventory_files,
724            "analysis_status": if tracked_paths.len() < original_selected_path_count || !memory_budget_exceeded_checkpoints.is_empty() { "degraded_resource_budget" } else if structurally_skipped_large_files > 0 { "degraded_large_files" } else if incomplete_inventory_files > 0 { "degraded_incomplete_inventory" } else { "complete" },
725            "resource_mode": if tracked_paths.len() < original_selected_path_count { "degraded_path_prefix" } else if !memory_budget_exceeded_checkpoints.is_empty() { "degraded_measured_rss" } else { "complete" },
726            "original_selected_path_count": original_selected_path_count,
727            "degraded_omitted_path_count": original_selected_path_count.saturating_sub(tracked_paths.len()),
728            "history": history_diagnostics,
729            "history_evidence_status": history_evidence_status,
730            "scope": scope
731        }),
732    };
733    let rollup = health::build_health_rollup(&analysis);
734    let result = report::write_report_bundle(&analysis, &rollup)?;
735    phase("report writing");
736    if result.report.get("schema_version").and_then(Value::as_u64) != Some(5) {
737        bail!("internal error: report writer did not produce schema 5");
738    }
739    Ok(result)
740}
741
742include!("analyze/tests.rs");