1use std::collections::{BTreeMap, HashMap};
2use std::fs;
3use std::path::{Component, Path, PathBuf};
4use std::sync::LazyLock;
5use std::time::{Duration, Instant};
6
7use anyhow::{Context, Result, bail};
8use chrono::{DateTime, SecondsFormat, Utc};
9use regex::Regex;
10use rusqlite::{Connection, OptionalExtension, params};
11use serde::{Deserialize, Serialize};
12use serde_json::{Value, json};
13use sha2::{Digest, Sha256};
14use tiktoken_rs::{
15 CoreBPE, cl100k_base, o200k_base, o200k_harmony, p50k_base, p50k_edit, r50k_base,
16};
17use unicode_normalization::UnicodeNormalization;
18use unicode_segmentation::UnicodeSegmentation;
19
20use crate::config;
21use crate::error::{ClassifiedError, ErrorKind};
22use crate::estimate;
23use crate::git;
24use crate::health;
25use crate::history;
26use crate::inventory;
27use crate::model::{Analysis, FileAnalysis, FindResult, ScopeIdentity};
28use crate::overlays;
29use crate::report;
30use crate::scoring;
31
32static CAMEL_CASE_RE: LazyLock<Regex> =
33 LazyLock::new(|| Regex::new(r"([a-z0-9])([A-Z])").expect("valid camel-case regex"));
34static ACRONYM_BOUNDARY_RE: LazyLock<Regex> =
35 LazyLock::new(|| Regex::new(r"([A-Z]+)([A-Z][a-z])").expect("valid acronym regex"));
36static NUMBER_RE: LazyLock<Regex> =
37 LazyLock::new(|| Regex::new(r"\b\d+(?:\.\d+)?\b").expect("valid number regex"));
38static RUST_SYMBOL_RE: LazyLock<Regex> = LazyLock::new(|| {
39 Regex::new(r"(?m)^\s*(?:pub(?:\([^)]*\))?\s+)?(?:(?:async|unsafe)\s+)?(?:fn|struct|enum|trait|mod|const|static)\s+([A-Za-z_][A-Za-z0-9_]*)")
40 .expect("valid Rust symbol regex")
41});
42static PYTHON_SYMBOL_RE: LazyLock<Regex> = LazyLock::new(|| {
43 Regex::new(r"(?m)^\s*(?:async\s+)?(?:def|class)\s+([A-Za-z_][A-Za-z0-9_]*)")
44 .expect("valid Python symbol regex")
45});
46static GO_SYMBOL_RE: LazyLock<Regex> = LazyLock::new(|| {
47 Regex::new(r"(?m)^\s*(?:func(?:\s+\([^)]*\))?|type)\s+([A-Za-z_][A-Za-z0-9_]*)")
48 .expect("valid Go symbol regex")
49});
50static JS_SYMBOL_RE: LazyLock<Regex> = LazyLock::new(|| {
51 Regex::new(r"(?m)^\s*(?:export\s+)?(?:default\s+)?(?:async\s+)?(?:function|class|interface|type)\s+([A-Za-z_$][A-Za-z0-9_$]*)")
52 .expect("valid JavaScript symbol regex")
53});
54static MARKDOWN_HEADING_RE: LazyLock<Regex> = LazyLock::new(|| {
55 Regex::new(r"(?m)^\s{0,3}#{1,6}\s+([^#\r\n]+?)\s*#*\s*$").expect("valid Markdown heading regex")
56});
57mod inline_tests;
58include!("analyze/cache.rs");
59include!("analyze/structural.rs");
60pub fn run_find() -> Result<FindResult> {
61 let repo_root = git::resolve_repo_root()?;
62 run_find_in(&repo_root)
63}
64
65pub fn run_find_in(repo_root: &Path) -> Result<FindResult> {
66 run_find_scoped(repo_root, false, None, false)
67}
68
69pub fn run_find_in_with_options(repo_root: &Path, allow_shallow: bool) -> Result<FindResult> {
70 run_find_scoped(repo_root, allow_shallow, None, false)
71}
72
73#[derive(Debug, Clone, Default)]
74pub struct FindOptions {
75 pub allow_shallow: bool,
76 pub scope: Option<String>,
77 pub progress: bool,
78 pub allow_empty_scope: bool,
79 pub state_dir: Option<PathBuf>,
80 pub output_dir: Option<PathBuf>,
81 pub no_cache: bool,
82 pub allow_degraded: bool,
83 pub as_of: Option<DateTime<Utc>>,
84 pub report_profile: String,
85 pub compression: String,
86}
87
88pub(crate) fn normalize_scope(value: Option<&str>) -> Result<Option<String>> {
89 let Some(raw) = value.map(str::trim) else {
90 return Ok(None);
91 };
92 if raw.is_empty() || raw == "." {
93 return Ok(None);
94 }
95 let path = Path::new(raw);
96 if path.is_absolute() {
97 bail!("--scope must be repo-relative, received {raw:?}");
98 }
99 let mut parts = Vec::new();
100 for component in path.components() {
101 match component {
102 Component::Normal(part) => parts.push(
103 part.to_str()
104 .ok_or_else(|| anyhow::anyhow!("--scope must be valid UTF-8"))?,
105 ),
106 Component::CurDir => {}
107 Component::ParentDir | Component::RootDir | Component::Prefix(_) => {
108 bail!("--scope must not escape the repository, received {raw:?}");
109 }
110 }
111 }
112 let normalized = parts.join("/");
113 Ok((!normalized.is_empty()).then_some(normalized))
114}
115
116pub(crate) fn selected_path_digest(paths: &[String]) -> String {
117 let mut digest = Sha256::new();
118 for path in paths {
119 digest.update(path.as_bytes());
120 digest.update([0]);
121 }
122 hex::encode(digest.finalize())
123}
124
125fn measure_rss_checkpoint(
126 checkpoint: &'static str,
127 memory_budget_bytes: u128,
128 allow_degraded: bool,
129 peak_rss_bytes: &mut Option<u64>,
130 exceeded_checkpoints: &mut Vec<&'static str>,
131) -> Result<()> {
132 let Some(rss_bytes) = estimate::current_rss_bytes() else {
133 return Ok(());
134 };
135 *peak_rss_bytes = Some(peak_rss_bytes.unwrap_or_default().max(rss_bytes));
136 if u128::from(rss_bytes) <= memory_budget_bytes {
137 return Ok(());
138 }
139 exceeded_checkpoints.push(checkpoint);
140 if allow_degraded {
141 return Err(ClassifiedError::new(
142 ErrorKind::ResourceLimit,
143 "degraded_memory_recovery_unavailable",
144 format!(
145 "analysis stopped at {checkpoint}: measured RSS {} MiB still exceeds resources.memory_budget_mb={} after deterministic degraded sampling; continuing would violate the memory contract",
146 rss_bytes.div_ceil(1024 * 1024),
147 memory_budget_bytes / 1024 / 1024
148 ),
149 )
150 .at("/resources/memory_budget_mb")
151 .into());
152 }
153 Err(ClassifiedError::new(
154 ErrorKind::ResourceLimit,
155 "measured_memory_budget_exceeded",
156 format!(
157 "analysis stopped at {checkpoint}: measured RSS {} MiB exceeds resources.memory_budget_mb={}; narrow --scope, use --allow-degraded, or raise the explicit budget",
158 rss_bytes.div_ceil(1024 * 1024),
159 memory_budget_bytes / 1024 / 1024
160 ),
161 )
162 .at("/resources/memory_budget_mb")
163 .into())
164}
165
166fn selected_content_digest(repo_root: &Path, paths: &[String]) -> Result<String> {
167 let mut digest = Sha256::new();
168 for path in paths {
169 digest.update(path.as_bytes());
170 digest.update([0]);
171 let absolute = repo_root.join(path);
172 let metadata = fs::symlink_metadata(&absolute)
173 .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?;
174 let bytes = if metadata.file_type().is_symlink() {
175 fs::read_link(&absolute)
176 .with_context(|| format!("selected tracked link changed or disappeared: {path}"))?
177 .to_string_lossy()
178 .into_owned()
179 .into_bytes()
180 } else if metadata.is_dir() {
181 b"<gitlink>".to_vec()
182 } else {
183 fs::read(&absolute)
184 .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?
185 };
186 digest.update(bytes);
187 digest.update([0]);
188 }
189 Ok(hex::encode(digest.finalize()))
190}
191
192fn balanced_path_sample(paths: &[String], limit: usize) -> Vec<String> {
193 let mut roots = BTreeMap::<&str, Vec<&String>>::new();
194 for path in paths {
195 roots
196 .entry(path.split('/').next().unwrap_or("."))
197 .or_default()
198 .push(path);
199 }
200 let mut selected = Vec::with_capacity(limit.min(paths.len()));
201 let mut offset = 0usize;
202 while selected.len() < limit {
203 let mut added = false;
204 for values in roots.values() {
205 if let Some(path) = values.get(offset) {
206 selected.push((*path).clone());
207 added = true;
208 if selected.len() == limit {
209 break;
210 }
211 }
212 }
213 if !added {
214 break;
215 }
216 offset += 1;
217 }
218 selected.sort();
219 selected
220}
221
222pub fn run_find_scoped(
223 repo_root: &Path,
224 allow_shallow: bool,
225 scope: Option<&str>,
226 progress: bool,
227) -> Result<FindResult> {
228 run_find_with_options(
229 repo_root,
230 &FindOptions {
231 allow_shallow,
232 scope: scope.map(ToOwned::to_owned),
233 progress,
234 allow_empty_scope: false,
235 ..FindOptions::default()
236 },
237 )
238}
239
240pub fn run_find_with_options(repo_root: &Path, options: &FindOptions) -> Result<FindResult> {
241 let allow_shallow = options.allow_shallow;
242 let scope = options.scope.as_deref();
243 let progress = options.progress;
244 let allow_empty_scope = options.allow_empty_scope;
245 let resolve_root = |value: Option<&Path>, fallback: PathBuf| {
246 value.map_or(fallback, |path| {
247 if path.is_absolute() {
248 path.to_path_buf()
249 } else {
250 repo_root.join(path)
251 }
252 })
253 };
254 let state_root = resolve_root(options.state_dir.as_deref(), config::slop_dir(repo_root));
255 let output_root = resolve_root(options.output_dir.as_deref(), config::slop_dir(repo_root));
256 let started = Instant::now();
257 let phase = |name: &str| {
258 if progress {
259 eprintln!("git-slop: {name} ({:.1}s)", started.elapsed().as_secs_f64());
260 }
261 };
262 phase("preflight");
263 let _scan_lock = config::acquire_scan_lock(&state_root)?;
264 let loaded_config = config::load(repo_root).map_err(|error| {
265 ClassifiedError::new(
266 ErrorKind::Contract,
267 "invalid_configuration",
268 format!("{error:#}"),
269 )
270 .at("/.slop/config.yaml")
271 })?;
272 let mut repo = git::repo_metadata(repo_root)?;
273 let runtime_exclusions = [
274 state_root.join("cache"),
275 state_root.join("scan.lock"),
276 state_root.join("scan.lock.owner"),
277 output_root.join("latest"),
278 output_root.join("runs"),
279 ]
280 .into_iter()
281 .filter_map(|path| {
282 path.strip_prefix(repo_root)
283 .ok()
284 .map(|value| value.to_string_lossy().replace('\\', "/"))
285 })
286 .collect::<Vec<_>>();
287 let starting_worktree = git::worktree_state_excluding(repo_root, &runtime_exclusions)?;
288 repo.worktree_clean = starting_worktree.clean;
289 repo.staged_change_count = starting_worktree.staged_change_count;
290 repo.modified_tracked_file_count = starting_worktree.modified_tracked_file_count;
291 repo.untracked_file_count = starting_worktree.untracked_file_count;
292 repo.worktree_state_digest = starting_worktree.digest;
293 if repo.is_shallow && !allow_shallow {
294 bail!(
295 "repository history is shallow; rerun with git slop find --allow-shallow to acknowledge incomplete history"
296 );
297 }
298 let all_tracked_paths = git::list_tracked_files(repo_root)?;
299 let scope = normalize_scope(scope)?;
300 if let Some(scope) = scope.as_deref() {
301 if fs::symlink_metadata(repo_root.join(scope)).is_err() {
302 bail!("--scope does not exist in the repository: {scope}");
303 }
304 }
305 let mut tracked_paths = all_tracked_paths
306 .iter()
307 .filter(|path| {
308 scope
309 .as_deref()
310 .is_none_or(|scope| *path == scope || path.starts_with(&format!("{scope}/")))
311 })
312 .cloned()
313 .collect::<Vec<_>>();
314 if tracked_paths.is_empty() && !allow_empty_scope {
315 bail!(
316 "{} selected no tracked paths; pass --allow-empty-scope only when an empty report is intentional",
317 scope.as_deref().map_or_else(
318 || "repository".to_string(),
319 |scope| format!("--scope {scope:?}")
320 )
321 );
322 }
323 let original_selected_path_count = tracked_paths.len();
324 let initial_estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
325 if initial_estimate.estimated_peak_memory_bytes > initial_estimate.memory_budget_bytes
326 && options.allow_degraded
327 {
328 let mut low = 0usize;
329 let mut high = tracked_paths.len();
330 while low < high {
331 let middle = (low + high).div_ceil(2);
332 let candidate = estimate::build(repo_root, &tracked_paths[..middle], &loaded_config);
333 if candidate.estimated_peak_memory_bytes <= candidate.memory_budget_bytes {
334 low = middle;
335 } else {
336 high = middle.saturating_sub(1);
337 }
338 }
339 tracked_paths = balanced_path_sample(&tracked_paths, low);
340 }
341 let scope_identity = ScopeIdentity {
342 mode: if scope.is_some() {
343 "scoped"
344 } else {
345 "repository"
346 }
347 .to_string(),
348 path: scope.clone(),
349 selected_path_count: tracked_paths.len(),
350 selected_path_digest: selected_path_digest(&tracked_paths),
351 };
352 let starting_content_digest = selected_content_digest(repo_root, &tracked_paths)?;
353 let estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
354 if estimate.estimated_peak_memory_bytes > estimate.memory_budget_bytes {
355 return Err(ClassifiedError::new(
356 ErrorKind::ResourceLimit,
357 "estimated_memory_budget_exceeded",
358 format!(
359 "analysis bounded before inventory: estimated {} MiB exceeds resources.memory_budget_mb={}; narrow --scope, use --allow-degraded, or raise the explicit budget",
360 estimate.estimated_peak_memory_bytes.div_ceil(1024 * 1024),
361 estimate.memory_budget_bytes / 1024 / 1024
362 ),
363 )
364 .at("/resources/memory_budget_mb")
365 .into());
366 }
367 let mut measured_peak_rss_bytes = None;
368 let mut memory_budget_exceeded_checkpoints = Vec::new();
369 measure_rss_checkpoint(
370 "pre_inventory",
371 estimate.memory_budget_bytes,
372 options.allow_degraded,
373 &mut measured_peak_rss_bytes,
374 &mut memory_budget_exceeded_checkpoints,
375 )?;
376 let (inventory_files, skipped) = inventory::build(repo_root, &tracked_paths, &loaded_config)?;
377 phase("inventory");
378 measure_rss_checkpoint(
379 "post_inventory",
380 estimate.memory_budget_bytes,
381 options.allow_degraded,
382 &mut measured_peak_rss_bytes,
383 &mut memory_budget_exceeded_checkpoints,
384 )?;
385 let encoder = configured_context_encoder(&loaded_config).map_err(|error| {
386 ClassifiedError::new(
387 ErrorKind::Contract,
388 "unsupported_tokenizer",
389 format!("{error:#}"),
390 )
391 .at("/tokenization/context_tokenizer_name")
392 })?;
393 let mut token_counts = BTreeMap::new();
394 let mut line_counts = BTreeMap::new();
395 let mut token_data = HashMap::new();
396 let tokenizer = config::pointer_str(&loaded_config, "/tokenization/context_tokenizer_name")
397 .unwrap_or("cl100k_base")
398 .to_string();
399 let large_file_bytes =
400 config::pointer_u64(&loaded_config, "/resources/large_file_bytes", 2_097_152) as usize;
401 let cache_path = state_root.join("cache").join("token-v4.sqlite3");
402 let mut cache_cleanup_warnings = Vec::new();
403 if !options.no_cache {
404 for version in ["token-v1", "token-v2", "token-v3"] {
405 let legacy_cache = state_root.join("cache").join(version);
406 if legacy_cache.exists() {
407 if let Err(error) = fs::remove_dir_all(&legacy_cache) {
408 cache_cleanup_warnings.push(format!(
409 "failed to remove legacy cache {}: {error}",
410 legacy_cache.display()
411 ));
412 }
413 }
414 }
415 }
416 let mut cache = if options.no_cache {
417 None
418 } else {
419 match TokenCache::open(&cache_path) {
420 Ok(cache) => Some(cache),
421 Err(error) => {
422 cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
423 TokenCache::open(&cache_path).ok()
424 }
425 }
426 };
427 let mut cache_hits = 0usize;
428 let mut cache_misses = 0usize;
429 let mut structurally_skipped_large_files = 0usize;
430 let mut intentionally_skipped_non_text_files = 0usize;
431 let mut incomplete_inventory_files = 0usize;
432 for file in &inventory_files {
433 if file.skipped_reason.as_deref() == Some("large_file_limit") {
434 structurally_skipped_large_files += 1;
435 }
436 if matches!(
437 file.skipped_reason.as_deref(),
438 Some("binary" | "gitlink" | "undecodable")
439 ) {
440 intentionally_skipped_non_text_files += 1;
441 } else if file.analysis_status != "analyzed"
442 && file.skipped_reason.as_deref() != Some("large_file_limit")
443 {
444 incomplete_inventory_files += 1;
445 }
446 if file.analysis_status != "analyzed" {
447 let conservative_tokens = if file.skipped_reason.as_deref() == Some("large_file_limit")
448 {
449 file.bytes.div_ceil(4)
450 } else {
451 0
452 };
453 token_counts.insert(file.path.clone(), conservative_tokens);
454 line_counts.insert(file.path.clone(), 0);
455 token_data.insert(
456 file.path.clone(),
457 (
458 0,
459 Vec::new(),
460 Vec::new(),
461 format!(
462 "incomplete:{}:{}",
463 file.skipped_reason.as_deref().unwrap_or("unknown"),
464 file.bytes
465 ),
466 ),
467 );
468 continue;
469 }
470 let mode = structural_mode(&file.path);
471 let cache_key = token_cache_key(&file.text, &tokenizer, large_file_bytes, mode);
472 let cached_value = if let Some(active_cache) = cache.as_ref() {
473 match active_cache.get(&cache_key) {
474 Ok(value) => value,
475 Err(error) => {
476 cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
477 cache = TokenCache::open(&cache_path).ok();
478 None
479 }
480 }
481 } else {
482 None
483 };
484 let cached = if let Some(cached) = cached_value {
485 cache_hits += 1;
486 cached
487 } else {
488 cache_misses += 1;
489 let cached = CachedTokenData {
490 token_count: encoder.encode_ordinary(&file.text).len(),
491 structural_tokens: if file.bytes > large_file_bytes {
492 Vec::new()
493 } else {
494 structural_content_tokens(mode, &file.text)
495 },
496 content_fingerprint: content_fingerprint(&file.text),
497 };
498 let put_error = cache
499 .as_ref()
500 .and_then(|active_cache| active_cache.put(&cache_key, &cached).err());
501 if let Some(error) = put_error {
502 cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
503 cache = TokenCache::open(&cache_path).ok();
504 }
505 cached
506 };
507 let count = cached.token_count;
508 let mut structural = cached.structural_tokens;
509 if file.bytes <= large_file_bytes {
510 structural.extend(structural_path_tokens(&file.path));
511 }
512 let top_term_limit =
513 config::pointer_u64(&loaded_config, "/semantic_drift/top_term_limit", 25) as usize;
514 let fingerprint = cached.content_fingerprint;
515 token_counts.insert(file.path.clone(), count);
516 line_counts.insert(file.path.clone(), file.lines);
517 token_data.insert(
518 file.path.clone(),
519 (
520 structural.len(),
521 top_terms(&structural, &file.language, &file.text, top_term_limit),
522 structural,
523 fingerprint,
524 ),
525 );
526 }
527 let cache_stats = if let Some(cache) = &cache {
528 match cache.enforce_limits(
529 config::pointer_u64(&loaded_config, "/resources/cache_max_entries", 10_000) as usize,
530 config::pointer_u64(&loaded_config, "/resources/cache_max_bytes", 536_870_912),
531 ) {
532 Ok(stats) => stats,
533 Err(error) => {
534 cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
535 CacheStats::default()
536 }
537 }
538 } else {
539 CacheStats::default()
540 };
541 phase("tokenization");
542 measure_rss_checkpoint(
543 "post_tokenization",
544 estimate.memory_budget_bytes,
545 options.allow_degraded,
546 &mut measured_peak_rss_bytes,
547 &mut memory_budget_exceeded_checkpoints,
548 )?;
549 repo.analyzed_content_digest = Some(starting_content_digest.clone());
550 let analyzed_paths: Vec<String> = inventory_files
551 .iter()
552 .map(|file| file.path.clone())
553 .collect();
554 let now = options.as_of.unwrap_or_else(Utc::now);
555 let (history_by_path, commits, history_diagnostics) = history::analyze_history(
556 repo_root,
557 &analyzed_paths,
558 &token_counts,
559 &line_counts,
560 &loaded_config,
561 now,
562 )?;
563 phase("history");
564 measure_rss_checkpoint(
565 "post_history",
566 estimate.memory_budget_bytes,
567 options.allow_degraded,
568 &mut measured_peak_rss_bytes,
569 &mut memory_budget_exceeded_checkpoints,
570 )?;
571 let mut files = Vec::with_capacity(inventory_files.len());
572 for file in inventory_files {
573 let tokens = token_counts.get(&file.path).copied().unwrap_or_default();
574 let inline_tests = has_inline_tests(&file.language, &file.text);
575 let (structural_token_count, top_structural_terms, structural_tokens, content_fingerprint) =
576 token_data
577 .remove(&file.path)
578 .unwrap_or_else(|| (0, Vec::new(), Vec::new(), String::new()));
579 let history = history_by_path.get(&file.path).cloned().unwrap_or_default();
580 let categories = structural_categories(structural_mode(&file.path), &file.text);
581 files.push(FileAnalysis {
582 path: file.path,
583 bytes: file.bytes,
584 lines: file.lines,
585 blank_lines: file.blank_lines,
586 code_lines: file.code_lines,
587 comment_lines: file.comment_lines,
588 language: file.language,
589 profile: file.profile.clone(),
590 classification: file.classification,
591 generated_from: file.generated_from,
592 generated_provenance: file.generated_provenance,
593 analysis_status: file.analysis_status,
594 skipped_reason: file.skipped_reason,
595 symlink_metadata: file.symlink_metadata,
596 has_inline_tests: inline_tests,
597 tokens,
598 context_band: scoring::context_band_for_profile(tokens, &file.profile, &loaded_config),
599 context_pressure: scoring::context_pressure_for_profile(
600 tokens,
601 &file.profile,
602 &loaded_config,
603 ),
604 content_fingerprint,
605 content_sha256: file.content_sha256,
606 structural_tokens,
607 structural_token_count,
608 top_structural_terms,
609 structural_categories: categories,
610 age_days: history.age_days,
611 revisions_window: history.revisions_window,
612 recency_weighted_commits: history.recency_weighted_commits,
613 added_window: history.added_window,
614 deleted_window: history.deleted_window,
615 churn_lines_window: history.line_churn_window,
616 line_churn_window: history.line_churn_window,
617 token_churn_window: history.token_churn_window,
618 relative_churn_window: history.relative_churn_window,
619 late_churn_spike: history.late_churn_spike,
620 author_count_window: history.author_count_window,
621 author_entropy: history.author_entropy,
622 top_author_share: history.top_author_share,
623 days_since_non_bot_edit: history.days_since_non_bot_edit,
624 recent_maintainer_diversity: history.recent_maintainer_diversity,
625 age_pressure: 0.0,
626 revision_norm: 0.0,
627 relative_churn_norm: 0.0,
628 churn_pressure: 0.0,
629 slop_score: 0.0,
630 slop_band: "low".to_string(),
631 reason_codes: Vec::new(),
632 costs: json!({}),
633 overlays: json!({}),
634 });
635 }
636 let history_evidence_reliable = !repo.is_shallow
637 && !history_diagnostics
638 .get("history_cap_reached")
639 .and_then(Value::as_bool)
640 .unwrap_or(false)
641 && ![
642 "full_history_cap_status",
643 "window_status_cap_status",
644 "window_numstat_cap_status",
645 ]
646 .into_iter()
647 .any(|field| history_diagnostics.get(field).and_then(Value::as_str) == Some("truncated"));
648 scoring::apply_scoring_with_evidence(&mut files, &loaded_config, history_evidence_reliable);
649 let organization = overlays::analyze(&mut files, &commits, &loaded_config)?;
650 phase("relationships");
651 measure_rss_checkpoint(
652 "post_relationships",
653 estimate.memory_budget_bytes,
654 options.allow_degraded,
655 &mut measured_peak_rss_bytes,
656 &mut memory_budget_exceeded_checkpoints,
657 )?;
658 let folders = scoring::build_folder_analyses(&files, &loaded_config);
659 let candidates = action_queue(&files, history_evidence_reliable, &loaded_config);
660 let (queue, observation_feed): (Vec<_>, Vec<_>) = candidates.into_iter().partition(|item| {
661 let classification = item
662 .get("classification")
663 .and_then(Value::as_str)
664 .unwrap_or("other");
665 let actionable = classification == "source";
666 let supported = item.get("evidence_status").and_then(Value::as_str) == Some("supported")
667 || item.get("is_pure_context_hotspot").and_then(Value::as_bool) == Some(true);
668 actionable
669 && supported
670 && matches!(
671 item.get("severity").and_then(Value::as_str),
672 Some("warning" | "error")
673 )
674 });
675 let generated_at = now.to_rfc3339_opts(SecondsFormat::Secs, true);
676 let analyzed_revision_at = repo.head_commit_timestamp.clone();
677 let ending_worktree = git::worktree_state_excluding(repo_root, &runtime_exclusions)?;
678 if ending_worktree.digest != repo.worktree_state_digest {
679 bail!("repository changed during analysis; no mixed-snapshot report was published");
680 }
681 if selected_content_digest(repo_root, &tracked_paths)? != starting_content_digest {
682 bail!(
683 "selected file content changed during analysis; no mixed-snapshot report was published"
684 );
685 }
686 let estimator_error_ratio = measured_peak_rss_bytes.map(|measured| {
687 let estimated = estimate.estimated_peak_memory_bytes.max(1) as f64;
688 ((measured as f64 - estimated) / estimated * 1_000_000.0).round() / 1_000_000.0
689 });
690 let estimate_range_contains_measurement = measured_peak_rss_bytes.map(|measured| {
691 let measured = u128::from(measured);
692 measured >= estimate.estimated_peak_memory_low_bytes
693 && measured <= estimate.estimated_peak_memory_high_bytes
694 });
695 let history_evidence_status = if repo.head_commit.is_none() {
696 "not_applicable_unborn"
697 } else if history_evidence_reliable {
698 "supported_with_per_file_shrinkage"
699 } else {
700 "incomplete_suppressed"
701 };
702 let analysis = Analysis {
703 output_root,
704 report_profile: if options.report_profile.is_empty() {
705 "standard".to_string()
706 } else {
707 options.report_profile.clone()
708 },
709 compression: if options.compression.is_empty() {
710 "none".to_string()
711 } else {
712 options.compression.clone()
713 },
714 repo,
715 config: loaded_config,
716 generated_at,
717 analyzed_revision_at,
718 skipped,
719 tracked_file_count: all_tracked_paths.len(),
720 scope: scope_identity,
721 files,
722 folders,
723 organization,
724 action_queue: queue,
725 observation_feed,
726 diagnostics: json!({
727 "analysis_elapsed_ms_before_report": started.elapsed().as_millis(),
728 "estimate": estimate,
729 "measured_peak_rss_bytes": measured_peak_rss_bytes,
730 "estimator_error_ratio": estimator_error_ratio,
731 "estimate_range_contains_measurement": estimate_range_contains_measurement,
732 "memory_budget_exceeded_checkpoints": memory_budget_exceeded_checkpoints,
733 "memory_measurement_status": if measured_peak_rss_bytes.is_some() { "measured" } else { "unsupported" },
734 "cache_hits": cache_hits,
735 "cache_misses": cache_misses,
736 "cache_entries": cache_stats.entries,
737 "cache_bytes": cache_stats.bytes,
738 "cache_failed_evictions": cache_stats.failed_evictions,
739 "cache_cleanup_warnings": cache_cleanup_warnings,
740 "cache_status": if options.no_cache { "disabled" } else { "enabled" },
741 "structurally_skipped_large_files": structurally_skipped_large_files,
742 "intentionally_skipped_non_text_files": intentionally_skipped_non_text_files,
743 "incomplete_inventory_files": incomplete_inventory_files,
744 "analysis_status": if tracked_paths.len() < original_selected_path_count || !memory_budget_exceeded_checkpoints.is_empty() { "degraded_resource_budget" } else if structurally_skipped_large_files > 0 { "degraded_large_files" } else if incomplete_inventory_files > 0 { "degraded_incomplete_inventory" } else { "complete" },
745 "resource_mode": if tracked_paths.len() < original_selected_path_count { "degraded_path_prefix" } else if !memory_budget_exceeded_checkpoints.is_empty() { "degraded_measured_rss" } else { "complete" },
746 "original_selected_path_count": original_selected_path_count,
747 "degraded_omitted_path_count": original_selected_path_count.saturating_sub(tracked_paths.len()),
748 "history": history_diagnostics,
749 "history_evidence_status": history_evidence_status,
750 "scope": scope
751 }),
752 };
753 let rollup = health::build_health_rollup(&analysis);
754 let mut result = report::write_report_bundle(&analysis, &rollup)?;
755 phase("report writing");
756 result.elapsed_ms = started.elapsed().as_millis();
757 if result.report.get("schema_version").and_then(Value::as_u64) != Some(5) {
758 bail!("internal error: report writer did not produce schema 5");
759 }
760 Ok(result)
761}
762
763include!("analyze/tests.rs");