1use anyhow::{Context, Result, bail};
7use serde::{Deserialize, Serialize};
8use sha2::{Digest, Sha256};
9use std::cmp::Ordering;
10use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet};
11use std::fs;
12use std::path::{Path, PathBuf};
13use std::time::SystemTime;
14
15const MAX_FILE_BYTES: u64 = 1024 * 1024;
16
17#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
18pub struct RepoIndex {
19 pub root: PathBuf,
20 pub indexed_at_ms: u64,
21 pub files: Vec<IndexedFile>,
22 pub symbols: Vec<SymbolRecord>,
23 pub imports: Vec<ImportRecord>,
24 pub tests: Vec<TestTarget>,
25}
26
27#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
28pub struct IndexedFile {
29 pub path: PathBuf,
30 pub language: String,
31 pub hash: String,
32 pub bytes: u64,
33}
34
35#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
36pub struct SymbolRecord {
37 pub name: String,
38 pub kind: String,
39 pub path: PathBuf,
40 pub line: usize,
41 pub signature: String,
42}
43
44#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
45pub struct RankedSymbolRecord {
46 pub symbol: SymbolRecord,
47 pub score: f64,
48 pub reasons: Vec<String>,
49}
50
51#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
52pub struct TextMatchRecord {
53 pub path: PathBuf,
54 pub line: usize,
55 pub kind: String,
56 pub text: String,
57 pub score: f64,
58}
59
60#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
61pub struct ImportRecord {
62 pub path: PathBuf,
63 pub target: String,
64 pub line: usize,
65}
66
67#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
68pub struct ReferenceRecord {
69 pub name: String,
70 pub path: PathBuf,
71 pub line: usize,
72 pub text: String,
73}
74
75#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
76pub struct DependencyEdge {
77 pub from: PathBuf,
78 pub to: String,
79}
80
81#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
82pub struct TestTarget {
83 pub command: String,
84 pub scope: String,
85 pub confidence: u8,
86 pub reason: String,
87}
88
89#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
90pub struct ChurnRecord {
91 pub path: PathBuf,
92 pub commits: u32,
93}
94
95#[derive(Debug, Default)]
96pub struct RepoIntelligenceCache {
97 files: HashMap<PathBuf, CachedFile>,
98}
99
100#[derive(Debug, Clone)]
101struct CachedFile {
102 metadata_key: String,
103 indexed: IndexedFile,
104 symbols: Vec<SymbolRecord>,
105 imports: Vec<ImportRecord>,
106 tests: Vec<TestTarget>,
107}
108
109impl RepoIntelligenceCache {
110 pub fn index_project(&mut self, root: &Path) -> Result<RepoIndex> {
111 build_index_with_cache(root, Some(self))
112 }
113}
114
115pub fn build_index(root: &Path) -> Result<RepoIndex> {
116 build_index_with_cache(root, None)
117}
118
119pub fn search_symbols(index: &RepoIndex, query: &str, kind: Option<&str>) -> Vec<SymbolRecord> {
120 ranked_symbol_matches(index, query, kind)
121 .into_iter()
122 .map(|ranked| ranked.symbol)
123 .collect()
124}
125
126pub fn goto_symbol(index: &RepoIndex, name: &str) -> Option<SymbolRecord> {
127 ranked_symbol_matches(index, name, None)
128 .into_iter()
129 .next()
130 .map(|ranked| ranked.symbol)
131}
132
133pub fn ranked_symbol_matches(
134 index: &RepoIndex,
135 query: &str,
136 kind: Option<&str>,
137) -> Vec<RankedSymbolRecord> {
138 SymbolRanker::new(query).rank(index, kind)
139}
140
141pub fn search_text_matches(
142 index: &RepoIndex,
143 query: &str,
144 max_results: usize,
145) -> Vec<TextMatchRecord> {
146 let query_alternatives = query_alternatives(query);
147 if query_alternatives
148 .iter()
149 .all(|alternative| alternative.tokens.is_empty())
150 {
151 return Vec::new();
152 }
153
154 let documents = text_documents(index);
155 let mut best_matches = documents
156 .iter()
157 .filter_map(|document| {
158 let score = query_alternatives
159 .iter()
160 .map(|alternative| bm25_score(&documents, &alternative.tokens, document))
161 .fold(0.0, f64::max);
162 (score > 0.0).then(|| TextMatchRecord {
163 path: document.path.clone(),
164 line: document.line,
165 kind: document.kind.clone(),
166 text: document.text.clone(),
167 score: round_score(score),
168 })
169 })
170 .collect::<Vec<_>>();
171
172 best_matches.sort_by(|left, right| {
173 score_cmp(right.score, left.score)
174 .then_with(|| left.path.cmp(&right.path))
175 .then_with(|| left.line.cmp(&right.line))
176 .then_with(|| left.kind.cmp(&right.kind))
177 .then_with(|| left.text.cmp(&right.text))
178 });
179 best_matches.truncate(max_results);
180 best_matches
181}
182
183#[derive(Debug, Clone)]
184struct QueryAlternative {
185 normalized: String,
186 tokens: Vec<String>,
187}
188
189#[derive(Debug)]
190struct SymbolRanker {
191 alternatives: Vec<QueryAlternative>,
192}
193
194#[derive(Debug, Clone)]
195struct SymbolScore {
196 score: f64,
197 reasons: Vec<String>,
198}
199
200impl SymbolRanker {
201 fn new(query: &str) -> Self {
202 Self {
203 alternatives: query_alternatives(query),
204 }
205 }
206
207 fn rank(&self, index: &RepoIndex, kind: Option<&str>) -> Vec<RankedSymbolRecord> {
208 let signature_documents = symbol_signature_documents(index);
209 let kind = kind.map(str::to_ascii_lowercase);
210 let mut ranked = index
211 .symbols
212 .iter()
213 .filter(|symbol| {
214 kind.as_ref()
215 .map(|kind| symbol.kind.eq_ignore_ascii_case(kind))
216 .unwrap_or(true)
217 })
218 .filter_map(|symbol| {
219 let score = self.score_symbol(symbol, &signature_documents);
220 (score.score > 0.0).then(|| RankedSymbolRecord {
221 symbol: symbol.clone(),
222 score: round_score(score.score),
223 reasons: score.reasons,
224 })
225 })
226 .collect::<Vec<_>>();
227
228 ranked.sort_by(|left, right| {
229 score_cmp(right.score, left.score)
230 .then_with(|| left.symbol.path.cmp(&right.symbol.path))
231 .then_with(|| left.symbol.line.cmp(&right.symbol.line))
232 .then_with(|| left.symbol.name.cmp(&right.symbol.name))
233 });
234 ranked
235 }
236
237 fn score_symbol(
238 &self,
239 symbol: &SymbolRecord,
240 signature_documents: &[TextDocument],
241 ) -> SymbolScore {
242 if self
243 .alternatives
244 .iter()
245 .all(|alternative| alternative.tokens.is_empty())
246 {
247 return SymbolScore {
248 score: 1.0 + kind_boost(symbol),
249 reasons: vec!["empty_query".to_string()],
250 };
251 }
252
253 self.alternatives
254 .iter()
255 .map(|alternative| {
256 score_symbol_for_alternative(symbol, alternative, signature_documents)
257 })
258 .max_by(|left, right| score_cmp(left.score, right.score))
259 .unwrap_or(SymbolScore {
260 score: 0.0,
261 reasons: Vec::new(),
262 })
263 }
264}
265
266fn score_symbol_for_alternative(
267 symbol: &SymbolRecord,
268 query: &QueryAlternative,
269 signature_documents: &[TextDocument],
270) -> SymbolScore {
271 let name_tokens = tokenize_identifier(&symbol.name);
272 let signature_tokens = tokenize_identifier(&symbol.signature);
273 let path_text = symbol.path.to_string_lossy();
274 let path_tokens = tokenize_identifier(&path_text);
275 let qualified_text = format!("{}::{}", path_text, symbol.name);
276 let normalized_name = normalize_identifier(&symbol.name);
277 let normalized_signature = normalize_identifier(&symbol.signature);
278 let normalized_path = normalize_identifier(&path_text);
279 let normalized_qualified = normalize_identifier(&qualified_text);
280
281 let mut score = 0.0;
282 let mut reasons = Vec::new();
283
284 if normalized_name == query.normalized {
285 score += 1000.0;
286 reasons.push("exact_name".to_string());
287 } else if normalized_qualified == query.normalized {
288 score += 950.0;
289 reasons.push("exact_qualified".to_string());
290 }
291
292 if !query.tokens.is_empty() {
293 let name_coverage = token_coverage(&query.tokens, &name_tokens);
294 if name_coverage > 0.0 {
295 score += 360.0 * name_coverage;
296 reasons.push(format!("name_token_coverage:{name_coverage:.2}"));
297 if name_coverage >= 1.0 {
298 score += 140.0;
299 reasons.push("all_name_tokens".to_string());
300 }
301 }
302
303 if tokens_in_order(&query.tokens, &name_tokens) {
304 score += 90.0;
305 reasons.push("tokens_in_order".to_string());
306 }
307
308 let path_coverage = token_coverage(&query.tokens, &path_tokens);
309 if path_coverage > 0.0 {
310 score += 22.0 * path_coverage;
311 reasons.push(format!("path_tokens:{path_coverage:.2}"));
312 }
313
314 let signature_coverage = token_coverage(&query.tokens, &signature_tokens);
315 if signature_coverage > 0.0 {
316 score += 28.0 * signature_coverage;
317 reasons.push(format!("signature_tokens:{signature_coverage:.2}"));
318 }
319 }
320
321 if !query.normalized.is_empty() {
322 if normalized_name.starts_with(&query.normalized) {
323 score += 160.0;
324 reasons.push("name_prefix".to_string());
325 }
326 if normalized_name.ends_with(&query.normalized) {
327 score += 95.0;
328 reasons.push("name_suffix".to_string());
329 }
330 if normalized_name.contains(&query.normalized) {
331 score += 130.0;
332 reasons.push("name_contains".to_string());
333 } else if is_subsequence(&query.normalized, &normalized_name) {
334 score += 65.0;
335 reasons.push("name_subsequence".to_string());
336 }
337
338 if normalized_signature.contains(&query.normalized) {
339 score += 32.0;
340 reasons.push("signature_contains".to_string());
341 }
342 if normalized_path.contains(&query.normalized) {
343 score += 18.0;
344 reasons.push("path_contains".to_string());
345 }
346
347 if let Some(distance_score) = edit_distance_score(&query.normalized, &normalized_name) {
348 score += distance_score;
349 reasons.push("edit_distance".to_string());
350 }
351 }
352
353 let bm25 = signature_documents
354 .iter()
355 .find(|document| document.path == symbol.path && document.line == symbol.line)
356 .map(|document| bm25_score(signature_documents, &query.tokens, document))
357 .unwrap_or(0.0)
358 .min(35.0);
359 if bm25 > 0.0 {
360 score += bm25;
361 reasons.push("bm25_signature".to_string());
362 }
363
364 if score > 0.0 {
365 let boost = kind_boost(symbol);
366 if boost > 0.0 {
367 score += boost;
368 reasons.push(format!("kind:{}", symbol.kind));
369 }
370 }
371
372 SymbolScore { score, reasons }
373}
374
375fn query_alternatives(query: &str) -> Vec<QueryAlternative> {
376 let alternatives = query
377 .split('|')
378 .map(str::trim)
379 .filter(|alternative| !alternative.is_empty())
380 .map(|alternative| QueryAlternative {
381 normalized: normalize_identifier(alternative),
382 tokens: tokenize_identifier(alternative),
383 })
384 .collect::<Vec<_>>();
385 if alternatives.is_empty() {
386 vec![QueryAlternative {
387 normalized: String::new(),
388 tokens: Vec::new(),
389 }]
390 } else {
391 alternatives
392 }
393}
394
395fn tokenize_identifier(text: &str) -> Vec<String> {
396 let mut parts = Vec::new();
397 let mut current = String::new();
398 let mut previous: Option<char> = None;
399 let mut chars = text.chars().peekable();
400
401 while let Some(ch) = chars.next() {
402 if !ch.is_ascii_alphanumeric() {
403 push_token(&mut parts, &mut current);
404 previous = None;
405 continue;
406 }
407
408 let next = chars.peek().copied();
409 let boundary = previous.is_some_and(|prev| {
410 (prev.is_ascii_lowercase() && ch.is_ascii_uppercase())
411 || (prev.is_ascii_alphabetic() && ch.is_ascii_digit())
412 || (prev.is_ascii_digit() && ch.is_ascii_alphabetic())
413 || (prev.is_ascii_uppercase()
414 && ch.is_ascii_uppercase()
415 && next.is_some_and(|next| next.is_ascii_lowercase()))
416 });
417 if boundary {
418 push_token(&mut parts, &mut current);
419 }
420 current.push(ch);
421 previous = Some(ch);
422 }
423 push_token(&mut parts, &mut current);
424 parts
425}
426
427fn push_token(parts: &mut Vec<String>, current: &mut String) {
428 if current.is_empty() {
429 return;
430 }
431 let token = normalize_token(current);
432 if !token.is_empty() {
433 parts.push(token);
434 }
435 current.clear();
436}
437
438fn normalize_token(token: &str) -> String {
439 let lower = token.to_ascii_lowercase();
440 if lower.len() > 5 && lower.ends_with("ing") {
441 return lower.trim_end_matches("ing").to_string();
442 }
443 if lower.len() > 4 && lower.ends_with("ies") {
444 return format!("{}y", &lower[..lower.len() - 3]);
445 }
446 if lower.len() > 4 && lower.ends_with("es") && !lower.ends_with("ses") {
447 return lower[..lower.len() - 2].to_string();
448 }
449 if lower.len() > 3 && lower.ends_with('s') && !lower.ends_with("ss") {
450 return lower[..lower.len() - 1].to_string();
451 }
452 lower
453}
454
455fn normalize_identifier(text: &str) -> String {
456 tokenize_identifier(text).join("")
457}
458
459fn token_coverage(query_tokens: &[String], candidate_tokens: &[String]) -> f64 {
460 if query_tokens.is_empty() {
461 return 0.0;
462 }
463 let candidate_tokens = candidate_tokens.iter().collect::<HashSet<_>>();
464 let covered = query_tokens
465 .iter()
466 .filter(|token| candidate_tokens.contains(token))
467 .count();
468 covered as f64 / query_tokens.len() as f64
469}
470
471fn tokens_in_order(query_tokens: &[String], candidate_tokens: &[String]) -> bool {
472 if query_tokens.is_empty() {
473 return false;
474 }
475 let mut candidate_iter = candidate_tokens.iter();
476 query_tokens
477 .iter()
478 .all(|query| candidate_iter.any(|candidate| candidate == query))
479}
480
481fn is_subsequence(needle: &str, haystack: &str) -> bool {
482 if needle.is_empty() {
483 return false;
484 }
485 let mut haystack = haystack.chars();
486 needle
487 .chars()
488 .all(|needle_ch| haystack.any(|haystack_ch| haystack_ch == needle_ch))
489}
490
491fn edit_distance_score(query: &str, candidate: &str) -> Option<f64> {
492 if query.len() < 4 || candidate.len() < 4 || query.len().abs_diff(candidate.len()) > 2 {
493 return None;
494 }
495 let distance = edit_distance(query, candidate);
496 (distance <= 2).then_some(80.0 - (distance as f64 * 22.0))
497}
498
499fn edit_distance(left: &str, right: &str) -> usize {
500 let right_chars = right.chars().collect::<Vec<_>>();
501 let mut previous = (0..=right_chars.len()).collect::<Vec<_>>();
502 for (left_idx, left_ch) in left.chars().enumerate() {
503 let mut current = vec![left_idx + 1];
504 for (right_idx, right_ch) in right_chars.iter().enumerate() {
505 let insert = current[right_idx] + 1;
506 let delete = previous[right_idx + 1] + 1;
507 let replace = previous[right_idx] + usize::from(left_ch != *right_ch);
508 current.push(insert.min(delete).min(replace));
509 }
510 previous = current;
511 }
512 previous[right_chars.len()]
513}
514
515fn kind_boost(symbol: &SymbolRecord) -> f64 {
516 match symbol.kind.as_str() {
517 "function" | "struct" | "class" | "trait" | "interface" => 18.0,
518 "enum" | "type" => 12.0,
519 "impl" | "constant" => 8.0,
520 _ => 0.0,
521 }
522}
523
524#[derive(Debug, Clone)]
525struct TextDocument {
526 path: PathBuf,
527 line: usize,
528 kind: String,
529 text: String,
530 tokens: Vec<String>,
531}
532
533fn text_documents(index: &RepoIndex) -> Vec<TextDocument> {
534 let mut documents = symbol_signature_documents(index);
535 for file in &index.files {
536 let path = index.root.join(&file.path);
537 let Ok(content) = fs::read_to_string(path) else {
538 continue;
539 };
540 documents.extend(comment_documents(&file.path, &file.language, &content));
541 documents.extend(snippet_documents(&file.path, &content));
542 }
543 dedupe_documents(documents)
544}
545
546fn symbol_signature_documents(index: &RepoIndex) -> Vec<TextDocument> {
547 index
548 .symbols
549 .iter()
550 .map(|symbol| {
551 let text = format!("{} {}", symbol.name, symbol.signature);
552 TextDocument {
553 path: symbol.path.clone(),
554 line: symbol.line,
555 kind: "signature".to_string(),
556 tokens: tokenize_identifier(&text),
557 text: compact_text(&text),
558 }
559 })
560 .collect()
561}
562
563fn comment_documents(path: &Path, language: &str, content: &str) -> Vec<TextDocument> {
564 let mut documents = Vec::new();
565 for (idx, line) in content.lines().enumerate() {
566 let trimmed = line.trim_start();
567 let (kind, text) = if let Some(text) = trimmed.strip_prefix("///") {
568 ("doc", text)
569 } else if let Some(text) = trimmed.strip_prefix("//!") {
570 ("doc", text)
571 } else if let Some(text) = trimmed.strip_prefix("//") {
572 ("comment", text)
573 } else if language == "python" {
574 if let Some(text) = trimmed.strip_prefix('#') {
575 ("comment", text)
576 } else {
577 continue;
578 }
579 } else if let Some(text) = trimmed
580 .strip_prefix("/*")
581 .map(|value| value.trim_end_matches("*/"))
582 {
583 ("comment", text)
584 } else if let Some(text) = trimmed
585 .strip_prefix('*')
586 .map(|value| value.trim_end_matches("*/"))
587 {
588 ("comment", text)
589 } else {
590 continue;
591 };
592
593 let text = text.trim();
594 if text.is_empty() {
595 continue;
596 }
597 documents.push(TextDocument {
598 path: path.to_path_buf(),
599 line: idx + 1,
600 kind: kind.to_string(),
601 text: compact_text(text),
602 tokens: tokenize_identifier(text),
603 });
604 }
605 documents
606}
607
608fn snippet_documents(path: &Path, content: &str) -> Vec<TextDocument> {
609 content
610 .lines()
611 .enumerate()
612 .filter_map(|(idx, line)| {
613 let text = line.trim();
614 if text.is_empty()
615 || text.starts_with("//")
616 || text.starts_with("///")
617 || text.starts_with("#")
618 || text.len() > 280
619 {
620 return None;
621 }
622 let tokens = tokenize_identifier(text);
623 (tokens.len() >= 3).then(|| TextDocument {
624 path: path.to_path_buf(),
625 line: idx + 1,
626 kind: "snippet".to_string(),
627 text: compact_text(text),
628 tokens,
629 })
630 })
631 .collect()
632}
633
634fn dedupe_documents(documents: Vec<TextDocument>) -> Vec<TextDocument> {
635 let mut seen = BTreeSet::new();
636 documents
637 .into_iter()
638 .filter(|document| {
639 seen.insert((
640 document.path.clone(),
641 document.line,
642 document.kind.clone(),
643 document.text.clone(),
644 ))
645 })
646 .collect()
647}
648
649fn bm25_score(corpus: &[TextDocument], query_tokens: &[String], document: &TextDocument) -> f64 {
650 if corpus.is_empty() || query_tokens.is_empty() || document.tokens.is_empty() {
651 return 0.0;
652 }
653 let average_len = corpus
654 .iter()
655 .map(|doc| doc.tokens.len() as f64)
656 .sum::<f64>()
657 / corpus.len() as f64;
658 let document_len = document.tokens.len() as f64;
659 let mut frequencies = HashMap::<&str, usize>::new();
660 for token in &document.tokens {
661 *frequencies.entry(token.as_str()).or_default() += 1;
662 }
663
664 let unique_query_tokens = query_tokens.iter().collect::<BTreeSet<_>>();
665 let mut score = 0.0;
666 for token in unique_query_tokens {
667 let Some(term_frequency) = frequencies.get(token.as_str()).copied() else {
668 continue;
669 };
670 let document_frequency = corpus
671 .iter()
672 .filter(|doc| doc.tokens.iter().any(|candidate| candidate == token))
673 .count() as f64;
674 let corpus_len = corpus.len() as f64;
675 let idf = (1.0 + (corpus_len - document_frequency + 0.5) / (document_frequency + 0.5)).ln();
676 let tf = term_frequency as f64;
677 let k1 = 1.2;
678 let b = 0.75;
679 score += idf * (tf * (k1 + 1.0)) / (tf + k1 * (1.0 - b + b * document_len / average_len));
680 }
681 score * 12.0
682}
683
684fn compact_text(text: &str) -> String {
685 text.split_whitespace()
686 .collect::<Vec<_>>()
687 .join(" ")
688 .chars()
689 .take(240)
690 .collect()
691}
692
693fn score_cmp(left: f64, right: f64) -> Ordering {
694 left.partial_cmp(&right).unwrap_or(Ordering::Equal)
695}
696
697fn round_score(score: f64) -> f64 {
698 (score * 100.0).round() / 100.0
699}
700
701pub fn references(index: &RepoIndex, name: &str) -> Vec<ReferenceRecord> {
702 let mut refs = Vec::new();
703 for file in &index.files {
704 let path = index.root.join(&file.path);
705 let Ok(content) = fs::read_to_string(path) else {
706 continue;
707 };
708 for (idx, line) in content.lines().enumerate() {
709 if contains_identifier(line, name) {
710 refs.push(ReferenceRecord {
711 name: name.to_string(),
712 path: file.path.clone(),
713 line: idx + 1,
714 text: line.trim().chars().take(240).collect(),
715 });
716 }
717 }
718 }
719 refs
720}
721
722pub fn dependency_edges(index: &RepoIndex) -> Vec<DependencyEdge> {
723 index
724 .imports
725 .iter()
726 .map(|import| DependencyEdge {
727 from: import.path.clone(),
728 to: import.target.clone(),
729 })
730 .collect()
731}
732
733pub fn discover_tests(root: &Path, touched_paths: &[PathBuf]) -> Vec<TestTarget> {
734 let mut targets = BTreeMap::<String, TestTarget>::new();
735 let justfile = root.join("justfile");
736 let has_just = justfile.exists();
737 let has_cargo = root.join("Cargo.toml").exists();
738
739 if has_just && has_cargo {
740 for path in touched_paths {
741 if let Some(crate_name) = crate_name_from_path(path) {
742 let command = format!("just test-crate {crate_name}");
743 targets.insert(
744 command.clone(),
745 TestTarget {
746 command,
747 scope: crate_name,
748 confidence: 95,
749 reason: "Rust crate path under workspace with just test-crate recipe"
750 .to_string(),
751 },
752 );
753 }
754 }
755 targets
756 .entry("just test".to_string())
757 .or_insert(TestTarget {
758 command: "just test".to_string(),
759 scope: "workspace".to_string(),
760 confidence: 70,
761 reason: "Rust workspace with justfile".to_string(),
762 });
763 } else if has_cargo {
764 targets
765 .entry("cargo test".to_string())
766 .or_insert(TestTarget {
767 command: "cargo test".to_string(),
768 scope: "workspace".to_string(),
769 confidence: 60,
770 reason: "Cargo.toml detected".to_string(),
771 });
772 }
773
774 if root.join("package.json").exists() {
775 targets.entry("npm test".to_string()).or_insert(TestTarget {
776 command: "npm test".to_string(),
777 scope: "package".to_string(),
778 confidence: 60,
779 reason: "package.json detected".to_string(),
780 });
781 }
782 if root.join("pyproject.toml").exists() || root.join("pytest.ini").exists() {
783 targets
784 .entry("pytest -q".to_string())
785 .or_insert(TestTarget {
786 command: "pytest -q".to_string(),
787 scope: "python".to_string(),
788 confidence: 60,
789 reason: "Python test config detected".to_string(),
790 });
791 }
792
793 dedupe_tests(targets.into_values().collect())
794}
795
796pub fn churn_from_git_log(root: &Path, max_entries: usize) -> Vec<ChurnRecord> {
797 let output = std::process::Command::new("git")
798 .arg("log")
799 .arg("--name-only")
800 .arg("--pretty=format:")
801 .current_dir(root)
802 .output();
803 let Ok(output) = output else {
804 return Vec::new();
805 };
806 if !output.status.success() {
807 return Vec::new();
808 }
809 let mut counts = BTreeMap::<PathBuf, u32>::new();
810 for line in String::from_utf8_lossy(&output.stdout).lines() {
811 let line = line.trim();
812 if line.is_empty() {
813 continue;
814 }
815 *counts.entry(PathBuf::from(line)).or_default() += 1;
816 }
817 let mut records = counts
818 .into_iter()
819 .map(|(path, commits)| ChurnRecord { path, commits })
820 .collect::<Vec<_>>();
821 records.sort_by(|left, right| {
822 right
823 .commits
824 .cmp(&left.commits)
825 .then_with(|| left.path.cmp(&right.path))
826 });
827 records.truncate(max_entries);
828 records
829}
830
831fn build_index_with_cache(
832 root: &Path,
833 mut cache: Option<&mut RepoIntelligenceCache>,
834) -> Result<RepoIndex> {
835 let root = root
836 .canonicalize()
837 .with_context(|| format!("failed to resolve repo root {}", root.display()))?;
838 let mut files = Vec::new();
839 collect_source_files(&root, &root, &mut files)?;
840 files.sort();
841
842 let mut indexed_files = Vec::new();
843 let mut symbols = Vec::new();
844 let mut imports = Vec::new();
845 let mut tests = discover_tests(&root, &[]);
846
847 for relative in files {
848 let absolute = root.join(&relative);
849 let metadata = fs::metadata(&absolute)?;
850 let modified = metadata
851 .modified()
852 .ok()
853 .and_then(|time| time.duration_since(SystemTime::UNIX_EPOCH).ok())
854 .map(|duration| duration.as_millis())
855 .unwrap_or(0);
856 let metadata_key = format!("{}:{}:{modified}", metadata.len(), relative.display());
857 if let Some(cache) = cache.as_deref_mut()
858 && let Some(cached) = cache.files.get(&relative)
859 && cached.metadata_key == metadata_key
860 {
861 indexed_files.push(cached.indexed.clone());
862 symbols.extend(cached.symbols.clone());
863 imports.extend(cached.imports.clone());
864 tests.extend(cached.tests.clone());
865 continue;
866 }
867
868 let content = fs::read_to_string(&absolute)
869 .with_context(|| format!("failed to read source file {}", absolute.display()))?;
870 let language = language_for_path(&relative).to_string();
871 let hash = hash_content(&content);
872 let indexed = IndexedFile {
873 path: relative.clone(),
874 language,
875 hash,
876 bytes: metadata.len(),
877 };
878 let file_symbols = extract_symbols(&relative, &content);
879 let file_imports = extract_imports(&relative, &content);
880 let file_tests = test_targets_for_file(&relative);
881 if let Some(cache) = cache.as_deref_mut() {
882 cache.files.insert(
883 relative.clone(),
884 CachedFile {
885 metadata_key,
886 indexed: indexed.clone(),
887 symbols: file_symbols.clone(),
888 imports: file_imports.clone(),
889 tests: file_tests.clone(),
890 },
891 );
892 }
893 indexed_files.push(indexed);
894 symbols.extend(file_symbols);
895 imports.extend(file_imports);
896 tests.extend(file_tests);
897 }
898
899 tests = dedupe_tests(tests);
900 Ok(RepoIndex {
901 root,
902 indexed_at_ms: now_ms(),
903 files: indexed_files,
904 symbols,
905 imports,
906 tests,
907 })
908}
909
910fn collect_source_files(root: &Path, dir: &Path, out: &mut Vec<PathBuf>) -> Result<()> {
911 for entry in fs::read_dir(dir).with_context(|| format!("failed to read {}", dir.display()))? {
912 let entry = entry?;
913 let path = entry.path();
914 let name = entry.file_name();
915 let name = name.to_string_lossy();
916 if name.starts_with('.') || matches!(name.as_ref(), "target" | "node_modules" | "dist") {
917 continue;
918 }
919 if path.is_dir() {
920 collect_source_files(root, &path, out)?;
921 continue;
922 }
923 let Ok(metadata) = fs::metadata(&path) else {
924 continue;
925 };
926 if metadata.len() > MAX_FILE_BYTES {
927 continue;
928 }
929 let Ok(relative) = path.strip_prefix(root) else {
930 continue;
931 };
932 if language_for_path(relative) != "unknown" {
933 out.push(relative.to_path_buf());
934 }
935 }
936 Ok(())
937}
938
939fn extract_symbols(path: &Path, content: &str) -> Vec<SymbolRecord> {
940 let mut out = Vec::new();
941 for (idx, line) in content.lines().enumerate() {
942 let trimmed = line.trim_start();
943 let Some((kind, name)) = symbol_from_line(trimmed) else {
944 continue;
945 };
946 out.push(SymbolRecord {
947 name,
948 kind: kind.to_string(),
949 path: path.to_path_buf(),
950 line: idx + 1,
951 signature: compact_signature(trimmed),
952 });
953 }
954 out
955}
956
957fn compact_signature(line: &str) -> String {
958 line.split('{')
959 .next()
960 .unwrap_or(line)
961 .trim_end()
962 .chars()
963 .take(240)
964 .collect()
965}
966
967fn symbol_from_line(line: &str) -> Option<(&'static str, String)> {
968 let line = line.strip_prefix("pub ").unwrap_or(line);
969 for (prefix, kind) in [
970 ("fn ", "function"),
971 ("async fn ", "function"),
972 ("struct ", "struct"),
973 ("enum ", "enum"),
974 ("trait ", "trait"),
975 ("type ", "type"),
976 ("impl ", "impl"),
977 ("const ", "constant"),
978 ("class ", "class"),
979 ("function ", "function"),
980 ("interface ", "interface"),
981 ("export function ", "function"),
982 ("export class ", "class"),
983 ("export interface ", "interface"),
984 ] {
985 if let Some(rest) = line.strip_prefix(prefix) {
986 let name = rest
987 .split(|ch: char| !(ch.is_ascii_alphanumeric() || ch == '_'))
988 .next()
989 .unwrap_or_default();
990 if !name.is_empty() {
991 return Some((kind, name.to_string()));
992 }
993 }
994 }
995 None
996}
997
998fn extract_imports(path: &Path, content: &str) -> Vec<ImportRecord> {
999 let mut out = Vec::new();
1000 for (idx, line) in content.lines().enumerate() {
1001 let trimmed = line.trim();
1002 let target = if let Some(rest) = trimmed.strip_prefix("use ") {
1003 rest.trim_end_matches(';')
1004 .split("::")
1005 .next()
1006 .map(str::to_string)
1007 } else if let Some(rest) = trimmed.strip_prefix("mod ") {
1008 rest.trim_end_matches(';')
1009 .split_whitespace()
1010 .next()
1011 .map(str::to_string)
1012 } else if trimmed.starts_with("import ") || trimmed.starts_with("export ") {
1013 quoted_module(trimmed)
1014 } else {
1015 None
1016 };
1017 if let Some(target) = target.filter(|value| !value.is_empty()) {
1018 out.push(ImportRecord {
1019 path: path.to_path_buf(),
1020 target,
1021 line: idx + 1,
1022 });
1023 }
1024 }
1025 out
1026}
1027
1028fn quoted_module(line: &str) -> Option<String> {
1029 for quote in ['"', '\''] {
1030 let mut parts = line.rsplit(quote);
1031 let _tail = parts.next()?;
1032 let value = parts.next()?;
1033 if !value.is_empty() && !value.contains(' ') {
1034 return Some(value.to_string());
1035 }
1036 }
1037 None
1038}
1039
1040fn test_targets_for_file(path: &Path) -> Vec<TestTarget> {
1041 let path_str = path.to_string_lossy();
1042 let mut out = Vec::new();
1043 if path_str.ends_with("_test.rs") || path_str.contains("/tests/") {
1044 out.push(TestTarget {
1045 command: "just test".to_string(),
1046 scope: path_str.to_string(),
1047 confidence: 80,
1048 reason: "test file detected".to_string(),
1049 });
1050 }
1051 if path_str.ends_with(".test.ts")
1052 || path_str.ends_with(".test.tsx")
1053 || path_str.ends_with(".spec.ts")
1054 || path_str.ends_with(".spec.tsx")
1055 {
1056 out.push(TestTarget {
1057 command: format!("npm test -- {}", path.display()),
1058 scope: path_str.to_string(),
1059 confidence: 75,
1060 reason: "JS/TS test file detected".to_string(),
1061 });
1062 }
1063 out
1064}
1065
1066fn crate_name_from_path(path: &Path) -> Option<String> {
1067 let mut components = path
1068 .components()
1069 .map(|component| component.as_os_str().to_string_lossy());
1070 while let Some(component) = components.next() {
1071 if component == "crates" {
1072 return components.next().map(|value| value.to_string());
1073 }
1074 }
1075 None
1076}
1077
1078fn dedupe_tests(tests: Vec<TestTarget>) -> Vec<TestTarget> {
1079 let mut seen = BTreeSet::new();
1080 let mut out = Vec::new();
1081 for test in tests {
1082 if seen.insert(test.command.clone()) {
1083 out.push(test);
1084 }
1085 }
1086 out.sort_by(|left, right| {
1087 right
1088 .confidence
1089 .cmp(&left.confidence)
1090 .then_with(|| left.command.cmp(&right.command))
1091 });
1092 out
1093}
1094
1095fn contains_identifier(line: &str, name: &str) -> bool {
1096 line.split(|ch: char| !(ch.is_ascii_alphanumeric() || ch == '_'))
1097 .any(|part| part == name)
1098}
1099
1100fn language_for_path(path: &Path) -> &'static str {
1101 match path.extension().and_then(|ext| ext.to_str()) {
1102 Some("rs") => "rust",
1103 Some("ts") | Some("tsx") => "typescript",
1104 Some("js") | Some("jsx") => "javascript",
1105 Some("go") => "go",
1106 Some("py") => "python",
1107 _ => "unknown",
1108 }
1109}
1110
1111fn hash_content(content: &str) -> String {
1112 hex::encode(Sha256::digest(content.as_bytes()))
1113}
1114
1115fn now_ms() -> u64 {
1116 SystemTime::now()
1117 .duration_since(SystemTime::UNIX_EPOCH)
1118 .map(|duration| duration.as_millis() as u64)
1119 .unwrap_or(0)
1120}
1121
1122pub fn ensure_indexed(index: &RepoIndex) -> Result<()> {
1123 if index.files.is_empty() {
1124 bail!("repository index is empty");
1125 }
1126 Ok(())
1127}
1128
1129#[cfg(test)]
1130mod tests {
1131 use super::*;
1132
1133 fn write_source(root: &Path, relative: &str, content: &str) {
1134 let path = root.join(relative);
1135 fs::create_dir_all(path.parent().unwrap()).unwrap();
1136 fs::write(path, content).unwrap();
1137 }
1138
1139 #[test]
1140 fn indexes_rust_symbols_and_references() {
1141 let dir = tempfile::tempdir().unwrap();
1142 fs::create_dir_all(dir.path().join("src")).unwrap();
1143 fs::write(
1144 dir.path().join("src/lib.rs"),
1145 "use std::fmt;\npub struct Engine;\npub fn run_engine() {}\nfn call() { run_engine(); }\n",
1146 )
1147 .unwrap();
1148
1149 let index = build_index(dir.path()).unwrap();
1150
1151 let matches = search_symbols(&index, "run_engine", Some("function"));
1152 assert_eq!(matches.len(), 1, "symbols: {:?}", index.symbols);
1153 assert_eq!(goto_symbol(&index, "Engine").unwrap().kind, "struct");
1154 assert_eq!(references(&index, "run_engine").len(), 2);
1155 assert_eq!(dependency_edges(&index)[0].to, "std");
1156 }
1157
1158 #[test]
1159 fn discovers_smallest_just_test_command_for_crate_path() {
1160 let dir = tempfile::tempdir().unwrap();
1161 fs::write(dir.path().join("Cargo.toml"), "[workspace]\n").unwrap();
1162 fs::write(dir.path().join("justfile"), "test:\n cargo test\n").unwrap();
1163
1164 let tests = discover_tests(dir.path(), &[PathBuf::from("crates/navi-core/src/lib.rs")]);
1165
1166 assert_eq!(
1167 tests[0].command, "just test-crate navi-core",
1168 "tests: {tests:?}"
1169 );
1170 }
1171
1172 #[test]
1173 fn cache_reuses_unchanged_file_records() {
1174 let dir = tempfile::tempdir().unwrap();
1175 fs::create_dir_all(dir.path().join("src")).unwrap();
1176 fs::write(dir.path().join("src/lib.rs"), "pub fn stable() {}\n").unwrap();
1177 let mut cache = RepoIntelligenceCache::default();
1178
1179 let first = cache.index_project(dir.path()).unwrap();
1180 let second = cache.index_project(dir.path()).unwrap();
1181
1182 assert_eq!(first.files[0].hash, second.files[0].hash);
1183 assert_eq!(second.symbols[0].name, "stable");
1184 }
1185
1186 #[test]
1187 fn tokenizes_identifiers_and_tool_names() {
1188 assert_eq!(
1189 tokenize_identifier("FuzzyToolSearch"),
1190 vec!["fuzzy", "tool", "search"]
1191 );
1192 assert_eq!(tokenize_identifier("tool_search"), vec!["tool", "search"]);
1193 assert_eq!(tokenize_identifier("symbol.goto"), vec!["symbol", "goto"]);
1194 assert_eq!(
1195 tokenize_identifier("dependency_graph.query"),
1196 vec!["dependency", "graph", "query"]
1197 );
1198 }
1199
1200 #[test]
1201 fn ranks_symbol_alternatives_by_best_candidate() {
1202 let dir = tempfile::tempdir().unwrap();
1203 write_source(
1204 dir.path(),
1205 "src/lib.rs",
1206 "pub struct FuzzyToolSearch;\npub struct OtherThing;\n",
1207 );
1208 let index = build_index(dir.path()).unwrap();
1209
1210 let matches = search_symbols(&index, "ToolSearch|SearchTool|Search", None);
1211
1212 assert_eq!(matches[0].name, "FuzzyToolSearch");
1213 }
1214
1215 #[test]
1216 fn symbolic_matches_outrank_bm25_text_matches() {
1217 let dir = tempfile::tempdir().unwrap();
1218 write_source(
1219 dir.path(),
1220 "src/lib.rs",
1221 "/// This comment talks about search but is not the target.\npub struct FuzzyToolSearch;\n",
1222 );
1223 let index = build_index(dir.path()).unwrap();
1224
1225 let ranked = ranked_symbol_matches(&index, "Search", None);
1226 let text_matches = search_text_matches(&index, "Search", 10);
1227
1228 assert_eq!(ranked[0].symbol.name, "FuzzyToolSearch");
1229 assert!(
1230 ranked[0].score > text_matches[0].score,
1231 "symbol={:?} text={:?}",
1232 ranked[0],
1233 text_matches[0]
1234 );
1235 }
1236
1237 #[test]
1238 fn goto_symbol_uses_ranker_instead_of_first_contains() {
1239 let dir = tempfile::tempdir().unwrap();
1240 write_source(
1241 dir.path(),
1242 "src/lib.rs",
1243 "pub struct PrefixSearchToolSuffix;\npub struct SearchTool;\n",
1244 );
1245 let index = build_index(dir.path()).unwrap();
1246
1247 let symbol = goto_symbol(&index, "SearchTool").unwrap();
1248
1249 assert_eq!(symbol.name, "SearchTool");
1250 }
1251
1252 #[test]
1253 fn bm25_returns_docs_comments_and_snippets_for_natural_language() {
1254 let dir = tempfile::tempdir().unwrap();
1255 write_source(
1256 dir.path(),
1257 "src/lib.rs",
1258 "/// Tool that searches symbols in docs and comments.\npub fn fuzzy_tool_search() {}\nfn caller() { fuzzy_tool_search(); }\n",
1259 );
1260 let index = build_index(dir.path()).unwrap();
1261
1262 let text_matches = search_text_matches(&index, "tool that searches symbols in docs", 10);
1263
1264 assert!(
1265 text_matches.iter().any(|record| record.kind == "doc"),
1266 "text_matches: {text_matches:?}"
1267 );
1268 assert!(
1269 text_matches
1270 .iter()
1271 .any(|record| record.kind == "signature" || record.kind == "snippet"),
1272 "text_matches: {text_matches:?}"
1273 );
1274 }
1275
1276 #[test]
1277 fn ranking_ties_are_deterministic() {
1278 let dir = tempfile::tempdir().unwrap();
1279 write_source(dir.path(), "src/b.rs", "pub struct SearchThing;\n");
1280 write_source(dir.path(), "src/a.rs", "pub struct SearchThing;\n");
1281 let index = build_index(dir.path()).unwrap();
1282
1283 let matches = search_symbols(&index, "SearchThing", Some("struct"));
1284
1285 assert_eq!(matches[0].path, PathBuf::from("src/a.rs"));
1286 assert_eq!(matches[1].path, PathBuf::from("src/b.rs"));
1287 }
1288}