1use std::collections::{BTreeMap, HashMap};
2use std::fs;
3use std::path::{Component, Path, PathBuf};
4use std::sync::LazyLock;
5use std::time::{Duration, Instant};
6
7use anyhow::{Context, Result, bail};
8use chrono::{DateTime, SecondsFormat, Utc};
9use regex::Regex;
10use rusqlite::{Connection, OptionalExtension, params};
11use serde::{Deserialize, Serialize};
12use serde_json::{Value, json};
13use sha2::{Digest, Sha256};
14use tiktoken_rs::{
15 CoreBPE, cl100k_base, o200k_base, o200k_harmony, p50k_base, p50k_edit, r50k_base,
16};
17use unicode_normalization::UnicodeNormalization;
18use unicode_segmentation::UnicodeSegmentation;
19
20use crate::config;
21use crate::error::{ClassifiedError, ErrorKind};
22use crate::estimate;
23use crate::git;
24use crate::health;
25use crate::history;
26use crate::inventory;
27use crate::model::{Analysis, FileAnalysis, FindResult, ScopeIdentity};
28use crate::overlays;
29use crate::report;
30use crate::scoring;
31
32static CAMEL_CASE_RE: LazyLock<Regex> =
33 LazyLock::new(|| Regex::new(r"([a-z0-9])([A-Z])").expect("valid camel-case regex"));
34static ACRONYM_BOUNDARY_RE: LazyLock<Regex> =
35 LazyLock::new(|| Regex::new(r"([A-Z]+)([A-Z][a-z])").expect("valid acronym regex"));
36static NUMBER_RE: LazyLock<Regex> =
37 LazyLock::new(|| Regex::new(r"\b\d+(?:\.\d+)?\b").expect("valid number regex"));
38include!("analyze/cache.rs");
39include!("analyze/structural.rs");
40pub fn run_find() -> Result<FindResult> {
41 let repo_root = git::resolve_repo_root()?;
42 run_find_in(&repo_root)
43}
44
45pub fn run_find_in(repo_root: &Path) -> Result<FindResult> {
46 run_find_scoped(repo_root, false, None, false)
47}
48
49pub fn run_find_in_with_options(repo_root: &Path, allow_shallow: bool) -> Result<FindResult> {
50 run_find_scoped(repo_root, allow_shallow, None, false)
51}
52
53#[derive(Debug, Clone, Default)]
54pub struct FindOptions {
55 pub allow_shallow: bool,
56 pub scope: Option<String>,
57 pub progress: bool,
58 pub allow_empty_scope: bool,
59 pub state_dir: Option<PathBuf>,
60 pub output_dir: Option<PathBuf>,
61 pub no_cache: bool,
62 pub allow_degraded: bool,
63 pub as_of: Option<DateTime<Utc>>,
64 pub report_profile: String,
65 pub compression: String,
66}
67
68pub(crate) fn normalize_scope(value: Option<&str>) -> Result<Option<String>> {
69 let Some(raw) = value.map(str::trim) else {
70 return Ok(None);
71 };
72 if raw.is_empty() || raw == "." {
73 return Ok(None);
74 }
75 let path = Path::new(raw);
76 if path.is_absolute() {
77 bail!("--scope must be repo-relative, received {raw:?}");
78 }
79 let mut parts = Vec::new();
80 for component in path.components() {
81 match component {
82 Component::Normal(part) => parts.push(
83 part.to_str()
84 .ok_or_else(|| anyhow::anyhow!("--scope must be valid UTF-8"))?,
85 ),
86 Component::CurDir => {}
87 Component::ParentDir | Component::RootDir | Component::Prefix(_) => {
88 bail!("--scope must not escape the repository, received {raw:?}");
89 }
90 }
91 }
92 let normalized = parts.join("/");
93 Ok((!normalized.is_empty()).then_some(normalized))
94}
95
96fn selected_path_digest(paths: &[String]) -> String {
97 let mut digest = Sha256::new();
98 for path in paths {
99 digest.update(path.as_bytes());
100 digest.update([0]);
101 }
102 hex::encode(digest.finalize())
103}
104
105fn measure_rss_checkpoint(
106 checkpoint: &'static str,
107 memory_budget_bytes: u128,
108 allow_degraded: bool,
109 peak_rss_bytes: &mut Option<u64>,
110 exceeded_checkpoints: &mut Vec<&'static str>,
111) -> Result<()> {
112 let Some(rss_bytes) = estimate::current_rss_bytes() else {
113 return Ok(());
114 };
115 *peak_rss_bytes = Some(peak_rss_bytes.unwrap_or_default().max(rss_bytes));
116 if u128::from(rss_bytes) <= memory_budget_bytes {
117 return Ok(());
118 }
119 exceeded_checkpoints.push(checkpoint);
120 if allow_degraded {
121 return Err(ClassifiedError::new(
122 ErrorKind::ResourceLimit,
123 "degraded_memory_recovery_unavailable",
124 format!(
125 "analysis stopped at {checkpoint}: measured RSS {} MiB still exceeds resources.memory_budget_mb={} after deterministic degraded sampling; continuing would violate the memory contract",
126 rss_bytes.div_ceil(1024 * 1024),
127 memory_budget_bytes / 1024 / 1024
128 ),
129 )
130 .at("/resources/memory_budget_mb")
131 .into());
132 }
133 Err(ClassifiedError::new(
134 ErrorKind::ResourceLimit,
135 "measured_memory_budget_exceeded",
136 format!(
137 "analysis stopped at {checkpoint}: measured RSS {} MiB exceeds resources.memory_budget_mb={}; narrow --scope, use --allow-degraded, or raise the explicit budget",
138 rss_bytes.div_ceil(1024 * 1024),
139 memory_budget_bytes / 1024 / 1024
140 ),
141 )
142 .at("/resources/memory_budget_mb")
143 .into())
144}
145
146fn selected_content_digest(repo_root: &Path, paths: &[String]) -> Result<String> {
147 let mut digest = Sha256::new();
148 for path in paths {
149 digest.update(path.as_bytes());
150 digest.update([0]);
151 let absolute = repo_root.join(path);
152 let metadata = fs::symlink_metadata(&absolute)
153 .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?;
154 let bytes = if metadata.file_type().is_symlink() {
155 fs::read_link(&absolute)
156 .with_context(|| format!("selected tracked link changed or disappeared: {path}"))?
157 .to_string_lossy()
158 .into_owned()
159 .into_bytes()
160 } else if metadata.is_dir() {
161 b"<gitlink>".to_vec()
162 } else {
163 fs::read(&absolute)
164 .with_context(|| format!("selected tracked path changed or disappeared: {path}"))?
165 };
166 digest.update(bytes);
167 digest.update([0]);
168 }
169 Ok(hex::encode(digest.finalize()))
170}
171
172fn balanced_path_sample(paths: &[String], limit: usize) -> Vec<String> {
173 let mut roots = BTreeMap::<&str, Vec<&String>>::new();
174 for path in paths {
175 roots
176 .entry(path.split('/').next().unwrap_or("."))
177 .or_default()
178 .push(path);
179 }
180 let mut selected = Vec::with_capacity(limit.min(paths.len()));
181 let mut offset = 0usize;
182 while selected.len() < limit {
183 let mut added = false;
184 for values in roots.values() {
185 if let Some(path) = values.get(offset) {
186 selected.push((*path).clone());
187 added = true;
188 if selected.len() == limit {
189 break;
190 }
191 }
192 }
193 if !added {
194 break;
195 }
196 offset += 1;
197 }
198 selected.sort();
199 selected
200}
201
202pub fn run_find_scoped(
203 repo_root: &Path,
204 allow_shallow: bool,
205 scope: Option<&str>,
206 progress: bool,
207) -> Result<FindResult> {
208 run_find_with_options(
209 repo_root,
210 &FindOptions {
211 allow_shallow,
212 scope: scope.map(ToOwned::to_owned),
213 progress,
214 allow_empty_scope: false,
215 ..FindOptions::default()
216 },
217 )
218}
219
220pub fn run_find_with_options(repo_root: &Path, options: &FindOptions) -> Result<FindResult> {
221 let allow_shallow = options.allow_shallow;
222 let scope = options.scope.as_deref();
223 let progress = options.progress;
224 let allow_empty_scope = options.allow_empty_scope;
225 let resolve_root = |value: Option<&Path>, fallback: PathBuf| {
226 value.map_or(fallback, |path| {
227 if path.is_absolute() {
228 path.to_path_buf()
229 } else {
230 repo_root.join(path)
231 }
232 })
233 };
234 let state_root = resolve_root(options.state_dir.as_deref(), config::slop_dir(repo_root));
235 let output_root = resolve_root(options.output_dir.as_deref(), config::slop_dir(repo_root));
236 let started = Instant::now();
237 let phase = |name: &str| {
238 if progress {
239 eprintln!("git-slop: {name} ({:.1}s)", started.elapsed().as_secs_f64());
240 }
241 };
242 phase("preflight");
243 let _scan_lock = config::acquire_scan_lock(repo_root)?;
244 let loaded_config = config::load(repo_root).map_err(|error| {
245 ClassifiedError::new(
246 ErrorKind::Contract,
247 "invalid_configuration",
248 format!("{error:#}"),
249 )
250 .at("/.slop/config.yaml")
251 })?;
252 let mut repo = git::repo_metadata(repo_root)?;
253 let runtime_exclusions = [
254 state_root.join("cache"),
255 output_root.join("latest"),
256 output_root.join("runs"),
257 ]
258 .into_iter()
259 .filter_map(|path| {
260 path.strip_prefix(repo_root)
261 .ok()
262 .map(|value| value.to_string_lossy().replace('\\', "/"))
263 })
264 .collect::<Vec<_>>();
265 let starting_worktree = git::worktree_state_excluding(repo_root, &runtime_exclusions)?;
266 repo.worktree_clean = starting_worktree.clean;
267 repo.staged_change_count = starting_worktree.staged_change_count;
268 repo.modified_tracked_file_count = starting_worktree.modified_tracked_file_count;
269 repo.untracked_file_count = starting_worktree.untracked_file_count;
270 repo.worktree_state_digest = starting_worktree.digest;
271 if repo.is_shallow && !allow_shallow {
272 bail!(
273 "repository history is shallow; rerun with git slop find --allow-shallow to acknowledge incomplete history"
274 );
275 }
276 let all_tracked_paths = git::list_tracked_files(repo_root)?;
277 let scope = normalize_scope(scope)?;
278 if let Some(scope) = scope.as_deref() {
279 if fs::symlink_metadata(repo_root.join(scope)).is_err() {
280 bail!("--scope does not exist in the repository: {scope}");
281 }
282 }
283 let mut tracked_paths = all_tracked_paths
284 .iter()
285 .filter(|path| {
286 scope
287 .as_deref()
288 .is_none_or(|scope| *path == scope || path.starts_with(&format!("{scope}/")))
289 })
290 .cloned()
291 .collect::<Vec<_>>();
292 if tracked_paths.is_empty() && !allow_empty_scope {
293 bail!(
294 "{} selected no tracked paths; pass --allow-empty-scope only when an empty report is intentional",
295 scope.as_deref().map_or_else(
296 || "repository".to_string(),
297 |scope| format!("--scope {scope:?}")
298 )
299 );
300 }
301 let original_selected_path_count = tracked_paths.len();
302 let initial_estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
303 if initial_estimate.estimated_peak_memory_bytes > initial_estimate.memory_budget_bytes
304 && options.allow_degraded
305 {
306 let mut low = 0usize;
307 let mut high = tracked_paths.len();
308 while low < high {
309 let middle = (low + high).div_ceil(2);
310 let candidate = estimate::build(repo_root, &tracked_paths[..middle], &loaded_config);
311 if candidate.estimated_peak_memory_bytes <= candidate.memory_budget_bytes {
312 low = middle;
313 } else {
314 high = middle.saturating_sub(1);
315 }
316 }
317 tracked_paths = balanced_path_sample(&tracked_paths, low);
318 }
319 let scope_identity = ScopeIdentity {
320 mode: if scope.is_some() {
321 "scoped"
322 } else {
323 "repository"
324 }
325 .to_string(),
326 path: scope.clone(),
327 selected_path_count: tracked_paths.len(),
328 selected_path_digest: selected_path_digest(&tracked_paths),
329 };
330 let starting_content_digest = selected_content_digest(repo_root, &tracked_paths)?;
331 let estimate = estimate::build(repo_root, &tracked_paths, &loaded_config);
332 if estimate.estimated_peak_memory_bytes > estimate.memory_budget_bytes {
333 return Err(ClassifiedError::new(
334 ErrorKind::ResourceLimit,
335 "estimated_memory_budget_exceeded",
336 format!(
337 "analysis bounded before inventory: estimated {} MiB exceeds resources.memory_budget_mb={}; narrow --scope, use --allow-degraded, or raise the explicit budget",
338 estimate.estimated_peak_memory_bytes.div_ceil(1024 * 1024),
339 estimate.memory_budget_bytes / 1024 / 1024
340 ),
341 )
342 .at("/resources/memory_budget_mb")
343 .into());
344 }
345 let mut measured_peak_rss_bytes = None;
346 let mut memory_budget_exceeded_checkpoints = Vec::new();
347 measure_rss_checkpoint(
348 "pre_inventory",
349 estimate.memory_budget_bytes,
350 options.allow_degraded,
351 &mut measured_peak_rss_bytes,
352 &mut memory_budget_exceeded_checkpoints,
353 )?;
354 let (inventory_files, skipped) = inventory::build(repo_root, &tracked_paths, &loaded_config)?;
355 phase("inventory");
356 measure_rss_checkpoint(
357 "post_inventory",
358 estimate.memory_budget_bytes,
359 options.allow_degraded,
360 &mut measured_peak_rss_bytes,
361 &mut memory_budget_exceeded_checkpoints,
362 )?;
363 let encoder = configured_context_encoder(&loaded_config).map_err(|error| {
364 ClassifiedError::new(
365 ErrorKind::Contract,
366 "unsupported_tokenizer",
367 format!("{error:#}"),
368 )
369 .at("/tokenization/context_tokenizer_name")
370 })?;
371 let mut token_counts = BTreeMap::new();
372 let mut line_counts = BTreeMap::new();
373 let mut token_data = HashMap::new();
374 let tokenizer = config::pointer_str(&loaded_config, "/tokenization/context_tokenizer_name")
375 .unwrap_or("cl100k_base")
376 .to_string();
377 let large_file_bytes =
378 config::pointer_u64(&loaded_config, "/resources/large_file_bytes", 2_097_152) as usize;
379 let cache_path = state_root.join("cache").join("token-v4.sqlite3");
380 let mut cache_cleanup_warnings = Vec::new();
381 if !options.no_cache {
382 for version in ["token-v1", "token-v2", "token-v3"] {
383 let legacy_cache = state_root.join("cache").join(version);
384 if legacy_cache.exists() {
385 if let Err(error) = fs::remove_dir_all(&legacy_cache) {
386 cache_cleanup_warnings.push(format!(
387 "failed to remove legacy cache {}: {error}",
388 legacy_cache.display()
389 ));
390 }
391 }
392 }
393 }
394 let mut cache = if options.no_cache {
395 None
396 } else {
397 match TokenCache::open(&cache_path) {
398 Ok(cache) => Some(cache),
399 Err(error) => {
400 cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
401 None
402 }
403 }
404 };
405 let mut cache_hits = 0usize;
406 let mut cache_misses = 0usize;
407 let mut structurally_skipped_large_files = 0usize;
408 for file in &inventory_files {
409 if file.analysis_status != "analyzed" || file.bytes > large_file_bytes {
410 structurally_skipped_large_files += 1;
411 }
412 if file.analysis_status != "analyzed" {
413 let conservative_tokens = if file.skipped_reason.as_deref() == Some("large_file_limit")
414 {
415 file.bytes.div_ceil(4)
416 } else {
417 0
418 };
419 token_counts.insert(file.path.clone(), conservative_tokens);
420 line_counts.insert(file.path.clone(), 0);
421 token_data.insert(
422 file.path.clone(),
423 (
424 0,
425 Vec::new(),
426 Vec::new(),
427 format!(
428 "incomplete:{}:{}",
429 file.skipped_reason.as_deref().unwrap_or("unknown"),
430 file.bytes
431 ),
432 ),
433 );
434 continue;
435 }
436 let mode = structural_mode(&file.path);
437 let cache_key = token_cache_key(&file.text, &tokenizer, large_file_bytes, mode);
438 let cached_value = if let Some(active_cache) = cache.as_ref() {
439 match active_cache.get(&cache_key) {
440 Ok(value) => value,
441 Err(error) => {
442 cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
443 cache = None;
444 None
445 }
446 }
447 } else {
448 None
449 };
450 let cached = if let Some(cached) = cached_value {
451 cache_hits += 1;
452 cached
453 } else {
454 cache_misses += 1;
455 let cached = CachedTokenData {
456 token_count: encoder.encode_ordinary(&file.text).len(),
457 structural_tokens: if file.bytes > large_file_bytes {
458 Vec::new()
459 } else {
460 structural_content_tokens(mode, &file.text)
461 },
462 content_fingerprint: content_fingerprint(&file.text),
463 };
464 let put_error = cache
465 .as_ref()
466 .and_then(|active_cache| active_cache.put(&cache_key, &cached).err());
467 if let Some(error) = put_error {
468 cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
469 cache = None;
470 }
471 cached
472 };
473 let count = cached.token_count;
474 let mut structural = cached.structural_tokens;
475 if file.bytes <= large_file_bytes {
476 structural.extend(structural_path_tokens(&file.path));
477 }
478 let top_term_limit =
479 config::pointer_u64(&loaded_config, "/semantic_drift/top_term_limit", 25) as usize;
480 let fingerprint = cached.content_fingerprint;
481 token_counts.insert(file.path.clone(), count);
482 line_counts.insert(file.path.clone(), file.lines);
483 token_data.insert(
484 file.path.clone(),
485 (
486 structural.len(),
487 top_terms(&structural, top_term_limit),
488 structural,
489 fingerprint,
490 ),
491 );
492 }
493 let cache_stats = if let Some(cache) = &cache {
494 match cache.enforce_limits(
495 config::pointer_u64(&loaded_config, "/resources/cache_max_entries", 10_000) as usize,
496 config::pointer_u64(&loaded_config, "/resources/cache_max_bytes", 536_870_912),
497 ) {
498 Ok(stats) => stats,
499 Err(error) => {
500 cache_cleanup_warnings.push(quarantine_cache(&cache_path, &error));
501 CacheStats::default()
502 }
503 }
504 } else {
505 CacheStats::default()
506 };
507 phase("tokenization");
508 measure_rss_checkpoint(
509 "post_tokenization",
510 estimate.memory_budget_bytes,
511 options.allow_degraded,
512 &mut measured_peak_rss_bytes,
513 &mut memory_budget_exceeded_checkpoints,
514 )?;
515 repo.analyzed_content_digest = Some(starting_content_digest.clone());
516 let analyzed_paths: Vec<String> = inventory_files
517 .iter()
518 .map(|file| file.path.clone())
519 .collect();
520 let now = options.as_of.unwrap_or_else(Utc::now);
521 let (history_by_path, commits, history_diagnostics) = history::analyze_history(
522 repo_root,
523 &analyzed_paths,
524 &token_counts,
525 &line_counts,
526 &loaded_config,
527 now,
528 )?;
529 phase("history");
530 measure_rss_checkpoint(
531 "post_history",
532 estimate.memory_budget_bytes,
533 options.allow_degraded,
534 &mut measured_peak_rss_bytes,
535 &mut memory_budget_exceeded_checkpoints,
536 )?;
537 let mut files = Vec::with_capacity(inventory_files.len());
538 for file in inventory_files {
539 let tokens = token_counts.get(&file.path).copied().unwrap_or_default();
540 let inline_tests = has_inline_tests(&file.language, &file.text);
541 let (structural_token_count, top_structural_terms, structural_tokens, content_fingerprint) =
542 token_data
543 .remove(&file.path)
544 .unwrap_or_else(|| (0, Vec::new(), Vec::new(), String::new()));
545 let history = history_by_path.get(&file.path).cloned().unwrap_or_default();
546 let categories = structural_categories(structural_mode(&file.path), &file.text);
547 files.push(FileAnalysis {
548 path: file.path,
549 bytes: file.bytes,
550 lines: file.lines,
551 blank_lines: file.blank_lines,
552 code_lines: file.code_lines,
553 comment_lines: file.comment_lines,
554 language: file.language,
555 profile: file.profile.clone(),
556 classification: file.classification,
557 analysis_status: file.analysis_status,
558 skipped_reason: file.skipped_reason,
559 symlink_metadata: file.symlink_metadata,
560 has_inline_tests: inline_tests,
561 tokens,
562 context_band: scoring::context_band_for_profile(tokens, &file.profile, &loaded_config),
563 context_pressure: scoring::context_pressure_for_profile(
564 tokens,
565 &file.profile,
566 &loaded_config,
567 ),
568 content_fingerprint,
569 structural_tokens,
570 structural_token_count,
571 top_structural_terms,
572 structural_categories: categories,
573 age_days: history.age_days,
574 revisions_window: history.revisions_window,
575 recency_weighted_commits: history.recency_weighted_commits,
576 added_window: history.added_window,
577 deleted_window: history.deleted_window,
578 churn_lines_window: history.line_churn_window,
579 line_churn_window: history.line_churn_window,
580 token_churn_window: history.token_churn_window,
581 relative_churn_window: history.relative_churn_window,
582 late_churn_spike: history.late_churn_spike,
583 author_count_window: history.author_count_window,
584 author_entropy: history.author_entropy,
585 top_author_share: history.top_author_share,
586 days_since_non_bot_edit: history.days_since_non_bot_edit,
587 recent_maintainer_diversity: history.recent_maintainer_diversity,
588 age_pressure: 0.0,
589 revision_norm: 0.0,
590 relative_churn_norm: 0.0,
591 churn_pressure: 0.0,
592 slop_score: 0.0,
593 slop_band: "low".to_string(),
594 reason_codes: Vec::new(),
595 costs: json!({}),
596 overlays: json!({}),
597 });
598 }
599 let history_evidence_reliable = !repo.is_shallow
600 && !history_diagnostics
601 .get("history_cap_reached")
602 .and_then(Value::as_bool)
603 .unwrap_or(false)
604 && ![
605 "full_history_cap_status",
606 "window_status_cap_status",
607 "window_numstat_cap_status",
608 ]
609 .into_iter()
610 .any(|field| history_diagnostics.get(field).and_then(Value::as_str) == Some("truncated"));
611 scoring::apply_scoring_with_evidence(&mut files, &loaded_config, history_evidence_reliable);
612 let organization = overlays::analyze(&mut files, &commits, &loaded_config)?;
613 phase("relationships");
614 measure_rss_checkpoint(
615 "post_relationships",
616 estimate.memory_budget_bytes,
617 options.allow_degraded,
618 &mut measured_peak_rss_bytes,
619 &mut memory_budget_exceeded_checkpoints,
620 )?;
621 let folders = scoring::build_folder_analyses(&files, &loaded_config);
622 let queue = action_queue(&files, history_evidence_reliable, &loaded_config);
623 let generated_at = now.to_rfc3339_opts(SecondsFormat::Secs, true);
624 let analyzed_revision_at = repo.head_commit_timestamp.clone();
625 let ending_worktree = git::worktree_state_excluding(repo_root, &runtime_exclusions)?;
626 if ending_worktree.digest != repo.worktree_state_digest {
627 bail!("repository changed during analysis; no mixed-snapshot report was published");
628 }
629 if selected_content_digest(repo_root, &tracked_paths)? != starting_content_digest {
630 bail!(
631 "selected file content changed during analysis; no mixed-snapshot report was published"
632 );
633 }
634 let estimator_error_ratio = measured_peak_rss_bytes.map(|measured| {
635 let estimated = estimate.estimated_peak_memory_bytes.max(1) as f64;
636 ((measured as f64 - estimated) / estimated * 1_000_000.0).round() / 1_000_000.0
637 });
638 let estimate_range_contains_measurement = measured_peak_rss_bytes.map(|measured| {
639 let measured = u128::from(measured);
640 measured >= estimate.estimated_peak_memory_low_bytes
641 && measured <= estimate.estimated_peak_memory_high_bytes
642 });
643 let analysis = Analysis {
644 output_root,
645 report_profile: if options.report_profile.is_empty() {
646 "standard".to_string()
647 } else {
648 options.report_profile.clone()
649 },
650 compression: if options.compression.is_empty() {
651 "none".to_string()
652 } else {
653 options.compression.clone()
654 },
655 repo,
656 config: loaded_config,
657 generated_at,
658 analyzed_revision_at,
659 skipped,
660 tracked_file_count: all_tracked_paths.len(),
661 scope: scope_identity,
662 files,
663 folders,
664 organization,
665 action_queue: queue,
666 diagnostics: json!({
667 "analysis_elapsed_ms_before_report": started.elapsed().as_millis(),
668 "estimate": estimate,
669 "measured_peak_rss_bytes": measured_peak_rss_bytes,
670 "estimator_error_ratio": estimator_error_ratio,
671 "estimate_range_contains_measurement": estimate_range_contains_measurement,
672 "memory_budget_exceeded_checkpoints": memory_budget_exceeded_checkpoints,
673 "memory_measurement_status": if measured_peak_rss_bytes.is_some() { "measured" } else { "unsupported" },
674 "cache_hits": cache_hits,
675 "cache_misses": cache_misses,
676 "cache_entries": cache_stats.entries,
677 "cache_bytes": cache_stats.bytes,
678 "cache_failed_evictions": cache_stats.failed_evictions,
679 "cache_cleanup_warnings": cache_cleanup_warnings,
680 "cache_status": if options.no_cache { "disabled" } else { "enabled" },
681 "structurally_skipped_large_files": structurally_skipped_large_files,
682 "analysis_status": if tracked_paths.len() < original_selected_path_count || !memory_budget_exceeded_checkpoints.is_empty() { "degraded_resource_budget" } else if structurally_skipped_large_files > 0 { "degraded_large_files" } else { "complete" },
683 "resource_mode": if tracked_paths.len() < original_selected_path_count { "degraded_path_prefix" } else if !memory_budget_exceeded_checkpoints.is_empty() { "degraded_measured_rss" } else { "complete" },
684 "original_selected_path_count": original_selected_path_count,
685 "degraded_omitted_path_count": original_selected_path_count.saturating_sub(tracked_paths.len()),
686 "history": history_diagnostics,
687 "history_evidence_status": if history_evidence_reliable { "supported_with_per_file_shrinkage" } else { "incomplete_suppressed" },
688 "scope": scope
689 }),
690 };
691 let rollup = health::build_health_rollup(&analysis);
692 let result = report::write_report_bundle(&analysis, &rollup)?;
693 phase("report writing");
694 if result.report.get("schema_version").and_then(Value::as_u64) != Some(5) {
695 bail!("internal error: report writer did not produce schema 5");
696 }
697 Ok(result)
698}
699
700include!("analyze/tests.rs");