Skip to main content

vtcode_indexer/
lib.rs

1#![allow(
2    missing_docs,
3    dead_code,
4    unused_imports,
5    reason = "Intentional compatibility, platform, or test-only suppression."
6)]
7#![expect(
8    unused_results,
9    reason = "Indexer cache and walker updates intentionally use mutation APIs for side effects and discard prior entries."
10)]
11
12//! Workspace-friendly file indexer and file utilities for VT Code.
13//!
14//! `vtcode-indexer` provides:
15//! - A lightweight workspace file indexer with markdown-backed persistence
16//! - Fast parallel fuzzy file search (via `file_search` module)
17//! - Markdown-backed storage utilities (via `markdown_store` module)
18
19pub mod file_search;
20pub mod markdown_store;
21
22use anyhow::Result;
23use hashbrown::HashMap;
24use ignore::{DirEntry, Walk};
25use rayon::prelude::*;
26use regex::Regex;
27use serde::{Deserialize, Serialize};
28use std::fmt::Write as FmtWrite;
29use std::fs;
30use std::io::{BufWriter, ErrorKind, Write};
31use std::path::{Path, PathBuf};
32use std::sync::Arc;
33use std::time::SystemTime;
34
35/// Persistence backend for [`SimpleIndexer`].
36pub trait IndexStorage: Send + Sync {
37    /// Prepare any directories or resources required for persistence.
38    fn init(&self, index_dir: &Path) -> Result<()>;
39
40    /// Persist an indexed file entry.
41    fn persist(&self, index_dir: &Path, entry: &FileIndex) -> Result<()>;
42
43    /// Whether this backend expects full-snapshot persistence.
44    ///
45    /// Snapshot-aware backends receive the complete in-memory index on each
46    /// update so on-disk state stays consistent across single-file and
47    /// directory indexing flows.
48    fn prefers_snapshot_persistence(&self) -> bool {
49        false
50    }
51
52    /// Remove a previously persisted file entry.
53    ///
54    /// Defaults to a no-op to keep existing custom storage backends compatible.
55    fn remove(&self, _index_dir: &Path, _file_path: &Path) -> Result<()> {
56        Ok(())
57    }
58
59    /// Persist a batch of indexed file entries.
60    ///
61    /// Defaults to calling [`IndexStorage::persist`] for each entry, keeping
62    /// existing custom storage backends compatible.
63    fn persist_batch(&self, index_dir: &Path, entries: &[FileIndex]) -> Result<()> {
64        for entry in entries {
65            self.persist(index_dir, entry)?;
66        }
67        Ok(())
68    }
69
70    /// Persist a batch of indexed file entries borrowed from the in-memory cache.
71    ///
72    /// Defaults to cloning the borrowed entries and delegating to
73    /// [`IndexStorage::persist_batch`] so existing custom storage backends remain
74    /// compatible.
75    fn persist_batch_refs(&self, index_dir: &Path, entries: &[&FileIndex]) -> Result<()> {
76        let owned = entries.iter().map(|entry| (*entry).clone()).collect::<Vec<_>>();
77        self.persist_batch(index_dir, &owned)
78    }
79}
80
81/// Directory traversal filter hook for [`SimpleIndexer`].
82pub trait TraversalFilter: Send + Sync {
83    /// Determine if the indexer should descend into the provided directory.
84    fn should_descend(&self, path: &Path, config: &SimpleIndexerConfig) -> bool;
85
86    /// Determine if the indexer should process the provided file.
87    fn should_index_file(&self, path: &Path, config: &SimpleIndexerConfig) -> bool;
88}
89
90/// Markdown-backed [`IndexStorage`] implementation.
91#[derive(Debug, Default, Clone)]
92pub struct MarkdownIndexStorage;
93
94impl IndexStorage for MarkdownIndexStorage {
95    fn init(&self, index_dir: &Path) -> Result<()> {
96        fs::create_dir_all(index_dir)?;
97        Ok(())
98    }
99
100    fn persist(&self, index_dir: &Path, entry: &FileIndex) -> Result<()> {
101        fs::create_dir_all(index_dir)?;
102        let file_name = format!("{}.md", calculate_hash(&entry.path));
103        let index_path = index_dir.join(file_name);
104        let file = fs::File::create(index_path)?;
105        let mut writer = BufWriter::new(file);
106        writeln!(writer, "# File Index: {}", entry.path)?;
107        writeln!(writer)?;
108        write_markdown_fields(&mut writer, entry)?;
109        writer.flush()?;
110        Ok(())
111    }
112
113    fn prefers_snapshot_persistence(&self) -> bool {
114        true
115    }
116
117    fn remove(&self, index_dir: &Path, file_path: &Path) -> Result<()> {
118        let file_name = format!("{}.md", calculate_hash(file_path.to_string_lossy().as_ref()));
119        let index_path = index_dir.join(file_name);
120        match fs::remove_file(index_path) {
121            Ok(()) => Ok(()),
122            Err(err) if err.kind() == ErrorKind::NotFound => Ok(()),
123            Err(err) => Err(err.into()),
124        }
125    }
126
127    fn persist_batch(&self, index_dir: &Path, entries: &[FileIndex]) -> Result<()> {
128        persist_markdown_snapshot(index_dir, entries.iter())
129    }
130
131    fn persist_batch_refs(&self, index_dir: &Path, entries: &[&FileIndex]) -> Result<()> {
132        persist_markdown_snapshot(index_dir, entries.iter().copied())
133    }
134}
135
136fn persist_markdown_snapshot<'a>(index_dir: &Path, entries: impl IntoIterator<Item = &'a FileIndex>) -> Result<()> {
137    let entries = entries.into_iter().collect::<Vec<_>>();
138
139    fs::create_dir_all(index_dir)?;
140    let temp_path = index_dir.join(".index.md.tmp");
141    let final_path = index_dir.join("index.md");
142    let file = fs::File::create(&temp_path)?;
143    let mut writer = BufWriter::new(file);
144
145    writeln!(writer, "# Workspace File Index")?;
146    writeln!(writer)?;
147    writeln!(writer, "- **Entries**: {}", entries.len())?;
148    writeln!(writer)?;
149
150    for entry in entries {
151        write_markdown_entry(&mut writer, entry)?;
152    }
153
154    writer.flush()?;
155    fs::rename(temp_path, final_path)?;
156    cleanup_legacy_markdown_entries(index_dir)?;
157    Ok(())
158}
159
160/// Default traversal filter powered by [`SimpleIndexerConfig`].
161#[derive(Debug, Default, Clone)]
162pub struct ConfigTraversalFilter;
163
164impl TraversalFilter for ConfigTraversalFilter {
165    fn should_descend(&self, path: &Path, config: &SimpleIndexerConfig) -> bool {
166        !should_skip_dir(path, config)
167    }
168
169    fn should_index_file(&self, path: &Path, config: &SimpleIndexerConfig) -> bool {
170        if !path.is_file() {
171            return false;
172        }
173
174        // Skip hidden files when configured.
175        if config.ignore_hidden && path.file_name().and_then(|n| n.to_str()).is_some_and(|s| s.starts_with('.')) {
176            return false;
177        }
178
179        // Always skip known sensitive files regardless of config.
180        if let Some(file_name) = path.file_name().and_then(|n| n.to_str())
181            && (vtcode_commons::exclusions::is_sensitive_file(file_name)
182                || file_name == ".gitignore"
183                || file_name == ".git")
184        {
185            return false;
186        }
187
188        true
189    }
190}
191
192/// Configuration for [`SimpleIndexer`].
193#[derive(Clone, Debug)]
194pub struct SimpleIndexerConfig {
195    workspace_root: PathBuf,
196    index_dir: PathBuf,
197    ignore_hidden: bool,
198    excluded_dirs: Vec<PathBuf>,
199    allowed_dirs: Vec<PathBuf>,
200}
201
202impl SimpleIndexerConfig {
203    /// Builds a configuration using VT Code's legacy layout as defaults.
204    fn new(workspace_root: PathBuf) -> Self {
205        let index_dir = workspace_root.join(".vtcode").join("index");
206        let vtcode_dir = workspace_root.join(".vtcode");
207        let external_dir = vtcode_dir.join("external");
208
209        let mut excluded_dirs: Vec<PathBuf> = vtcode_commons::exclusions::DEFAULT_EXCLUDED_DIRS
210            .iter()
211            .map(|name| workspace_root.join(name))
212            .collect();
213        excluded_dirs.push(index_dir.clone());
214        excluded_dirs.push(vtcode_dir);
215
216        excluded_dirs.dedup();
217
218        Self {
219            workspace_root,
220            index_dir,
221            ignore_hidden: true,
222            excluded_dirs,
223            allowed_dirs: vec![external_dir],
224        }
225    }
226
227    /// Updates the index directory used for persisted metadata.
228    pub fn with_index_dir(mut self, index_dir: impl Into<PathBuf>) -> Self {
229        let index_dir = index_dir.into();
230        self.index_dir = index_dir.clone();
231        self.push_unique_excluded(index_dir);
232        self
233    }
234
235    /// Adds an allowed directory that should be indexed even if hidden or inside an excluded parent.
236    pub fn add_allowed_dir(mut self, path: impl Into<PathBuf>) -> Self {
237        let path = path.into();
238        if !self.allowed_dirs.iter().any(|existing| existing == &path) {
239            self.allowed_dirs.push(path);
240        }
241        self
242    }
243
244    /// Adds an additional excluded directory to skip during traversal.
245    pub fn add_excluded_dir(mut self, path: impl Into<PathBuf>) -> Self {
246        let path = path.into();
247        self.push_unique_excluded(path);
248        self
249    }
250
251    /// Toggles whether hidden directories (prefix `.`) are ignored.
252    fn ignore_hidden(mut self, ignore_hidden: bool) -> Self {
253        self.ignore_hidden = ignore_hidden;
254        self
255    }
256
257    /// Workspace root accessor.
258    pub fn workspace_root(&self) -> &Path {
259        &self.workspace_root
260    }
261
262    /// Index directory accessor.
263    fn index_dir(&self) -> &Path {
264        &self.index_dir
265    }
266
267    fn push_unique_excluded(&mut self, path: PathBuf) {
268        if !self.excluded_dirs.iter().any(|existing| existing == &path) {
269            self.excluded_dirs.push(path);
270        }
271    }
272}
273
274/// Simple file index entry.
275#[derive(Debug, Clone, Serialize, Deserialize)]
276pub struct FileIndex {
277    /// File path.
278    path: String,
279    /// File content hash for change detection.
280    hash: String,
281    /// Last modified timestamp.
282    modified: u64,
283    /// File size.
284    size: u64,
285    /// Language/extension.
286    language: String,
287    /// Simple tags.
288    tags: Vec<String>,
289}
290
291/// Simple search result.
292#[derive(Debug, Clone, Serialize, Deserialize)]
293pub struct SearchResult {
294    file_path: String,
295    line_number: usize,
296    line_content: String,
297    matches: Vec<String>,
298}
299
300/// Simple file indexer.
301pub struct SimpleIndexer {
302    config: SimpleIndexerConfig,
303    index_cache: HashMap<String, FileIndex>,
304    content_cache: HashMap<String, (String, Arc<str>)>,
305    storage: Arc<dyn IndexStorage>,
306    filter: Arc<dyn TraversalFilter>,
307}
308
309impl SimpleIndexer {
310    /// Create a new simple indexer with default VT Code paths.
311    pub fn new(workspace_root: PathBuf) -> Self {
312        Self::with_components(
313            SimpleIndexerConfig::new(workspace_root),
314            Arc::new(MarkdownIndexStorage),
315            Arc::new(ConfigTraversalFilter),
316        )
317    }
318
319    /// Create a simple indexer with the provided configuration.
320    fn with_config(config: SimpleIndexerConfig) -> Self {
321        Self::with_components(config, Arc::new(MarkdownIndexStorage), Arc::new(ConfigTraversalFilter))
322    }
323
324    /// Create a new simple indexer using a custom index directory.
325    pub fn with_index_dir(workspace_root: PathBuf, index_dir: PathBuf) -> Self {
326        let config = SimpleIndexerConfig::new(workspace_root).with_index_dir(index_dir);
327        Self::with_config(config)
328    }
329
330    /// Create an indexer with explicit storage and traversal filter implementations.
331    fn with_components(
332        config: SimpleIndexerConfig,
333        storage: Arc<dyn IndexStorage>,
334        filter: Arc<dyn TraversalFilter>,
335    ) -> Self {
336        Self {
337            config,
338            index_cache: HashMap::new(),
339            content_cache: HashMap::new(),
340            storage,
341            filter,
342        }
343    }
344
345    /// Replace the storage backend used to persist index entries.
346    fn with_storage(self, storage: Arc<dyn IndexStorage>) -> Self {
347        Self { storage, ..self }
348    }
349
350    /// Replace the traversal filter used to decide which files and directories are indexed.
351    fn with_filter(self, filter: Arc<dyn TraversalFilter>) -> Self {
352        Self { filter, ..self }
353    }
354
355    /// Initialize the index directory.
356    pub fn init(&self) -> Result<()> {
357        self.storage.init(self.config.index_dir())
358    }
359
360    /// Get the workspace root path.
361    pub fn workspace_root(&self) -> &Path {
362        self.config.workspace_root()
363    }
364
365    /// Get the index directory used for persisted metadata.
366    pub fn index_dir(&self) -> &Path {
367        self.config.index_dir()
368    }
369
370    /// Index a single file.
371    fn index_file(&mut self, file_path: &Path) -> Result<()> {
372        let cache_key = file_path.to_string_lossy().into_owned();
373
374        if self.storage.prefers_snapshot_persistence() {
375            let next_entry = if file_path.exists() && self.should_process_file_path(file_path) {
376                self.build_file_index(file_path)?
377            } else {
378                None
379            };
380
381            self.apply_snapshot_file_update(cache_key, next_entry)?;
382            return Ok(());
383        }
384
385        if !file_path.exists() || !self.should_process_file_path(file_path) {
386            self.index_cache.remove(cache_key.as_str());
387            self.storage.remove(self.config.index_dir(), file_path)?;
388            return Ok(());
389        }
390
391        if let Some(index) = self.build_file_index(file_path)? {
392            self.storage.persist(self.config.index_dir(), &index)?;
393            self.index_cache.insert(index.path.clone(), index);
394        } else {
395            self.index_cache.remove(cache_key.as_str());
396            self.storage.remove(self.config.index_dir(), file_path)?;
397        }
398
399        Ok(())
400    }
401
402    /// Index all files in directory recursively.
403    /// Respects .gitignore, .ignore, and other ignore files.
404    /// SECURITY: Always skips hidden files and sensitive data (.env, .git, etc.)
405    pub fn index_directory(&mut self, dir_path: &Path) -> Result<()> {
406        let walker = self.build_walker(dir_path);
407
408        let mut entries = Vec::new();
409
410        for entry in walker.filter_map(|e| e.ok()) {
411            let path = entry.path();
412
413            // Only index files, not directories
414            if entry.file_type().is_some_and(|ft| ft.is_file())
415                && let Some(index) = self.build_file_index(path)?
416            {
417                entries.push(index);
418            }
419        }
420
421        if self.storage.prefers_snapshot_persistence() {
422            self.apply_snapshot_directory_update(dir_path, &entries)?;
423        } else {
424            entries.sort_unstable_by(|left, right| left.path.cmp(&right.path));
425            self.storage.persist_batch(self.config.index_dir(), &entries)?;
426        }
427
428        self.replace_cached_entries(dir_path, &entries);
429
430        Ok(())
431    }
432
433    /// Discover all files in directory recursively without indexing them.
434    /// This is much faster than `index_directory` as it avoids hashing and persistence.
435    pub fn discover_files(&self, dir_path: &Path) -> Vec<String> {
436        let walker = self.build_walker(dir_path);
437
438        let mut files = walker
439            .filter_map(|e| e.ok())
440            .filter(|e| {
441                if !e.file_type().is_some_and(|ft| ft.is_file()) {
442                    return false;
443                }
444
445                self.should_process_file_path(e.path())
446            })
447            .map(|e| e.path().to_string_lossy().into_owned())
448            .collect::<Vec<_>>();
449        files.sort_unstable();
450        files
451    }
452
453    /// List the immediate children of `dir_path` (one level deep) without
454    /// recursing into subdirectories.
455    ///
456    /// Uses the same ignore/hidden/excluded-directory rules as [`Self::discover_files`]
457    /// (via the shared traversal filter), so expensive subtrees such as
458    /// `node_modules`, `.git`, and build directories are never listed. This lets
459    /// callers build a directory navigator that only touches the directories the
460    /// user actually opens, instead of walking the entire workspace up front.
461    ///
462    /// Returns `(path, is_dir)` pairs with directories sorted before files.
463    pub fn discover_dir_entries(&self, dir_path: &Path) -> Vec<(PathBuf, bool)> {
464        let walker = self.build_shallow_walker(dir_path);
465
466        let mut entries: Vec<(PathBuf, bool)> = walker
467            .filter_map(|e| e.ok())
468            .filter(|e| e.path() != dir_path)
469            .map(|e| {
470                let path = e.path().to_path_buf();
471                let is_dir = e.file_type().is_some_and(|ft| ft.is_dir());
472                (path, is_dir)
473            })
474            .filter(|(path, is_dir)| {
475                if *is_dir {
476                    !should_skip_dir(path, &self.config)
477                } else {
478                    self.should_process_file_path(path)
479                }
480            })
481            .collect();
482
483        entries.sort_by_cached_key(|(path, is_dir)| {
484            // Cache the lowercased path once per entry: a plain `sort_by` would
485            // allocate two temporary strings on every one of the ~N·log N
486            // comparisons, and this runs on the interactive file-palette path.
487            (std::cmp::Reverse(*is_dir), path.to_string_lossy().to_lowercase())
488        });
489        entries
490    }
491
492    /// Internal helper for regex-based file content search.
493    /// Used by both `search()` and `grep()` to avoid code duplication.
494    fn search_files_internal(
495        &self,
496        regex: &Regex,
497        path_filter: Option<&str>,
498        extract_matches: bool,
499    ) -> Vec<SearchResult> {
500        const PARALLEL_THRESHOLD: usize = 64;
501
502        let candidate_paths: Vec<&String> = self
503            .index_cache
504            .keys()
505            .filter(|file_path| path_filter.is_none_or(|filter| file_path.contains(filter)))
506            .collect();
507
508        let mut content_lookup: HashMap<&String, Arc<str>> = HashMap::with_capacity(candidate_paths.len());
509        for path in &candidate_paths {
510            if let Some(index) = self.index_cache.get(*path)
511                && let Some((hash, content)) = self.content_cache.get(*path)
512                && hash == &index.hash
513            {
514                content_lookup.insert(*path, content.clone());
515            }
516        }
517
518        let map_file = move |file_path: &&String| -> Vec<SearchResult> {
519            if let Some(cached) = content_lookup.get(file_path) {
520                return Self::search_content(cached.as_ref(), regex, file_path, extract_matches);
521            }
522            let text = match fs::read_to_string(file_path) {
523                Ok(text) => text,
524                Err(_) => return Vec::new(),
525            };
526            Self::search_content(&text, regex, file_path, extract_matches)
527        };
528
529        let mut results: Vec<SearchResult> = if candidate_paths.len() <= PARALLEL_THRESHOLD {
530            candidate_paths.iter().flat_map(map_file).collect()
531        } else {
532            candidate_paths.par_iter().flat_map(map_file).collect()
533        };
534
535        results.sort_unstable_by(|left, right| {
536            left.file_path
537                .cmp(&right.file_path)
538                .then_with(|| left.line_number.cmp(&right.line_number))
539        });
540        results
541    }
542
543    #[inline]
544    fn search_content(content: &str, regex: &Regex, file_path: &&String, extract_matches: bool) -> Vec<SearchResult> {
545        let mut local = Vec::new();
546        for (line_num, line) in content.lines().enumerate() {
547            if regex.is_match(line) {
548                let line_content = line.to_string();
549                let matches = if extract_matches {
550                    regex.find_iter(line).map(|m| m.as_str().to_string()).collect()
551                } else {
552                    vec![line_content.clone()]
553                };
554
555                local.push(SearchResult {
556                    file_path: (*file_path).clone(),
557                    line_number: line_num + 1,
558                    line_content,
559                    matches,
560                });
561            }
562        }
563        local
564    }
565
566    /// Search files using regex pattern.
567    pub fn search(&self, pattern: &str, path_filter: Option<&str>) -> Result<Vec<SearchResult>> {
568        let regex = Regex::new(pattern)?;
569        Ok(self.search_files_internal(&regex, path_filter, true))
570    }
571
572    /// Find files by name pattern.
573    fn find_files(&self, pattern: &str) -> Result<Vec<String>> {
574        let regex = Regex::new(pattern)?;
575        let mut results = Vec::with_capacity(self.index_cache.len());
576
577        for file_path in self.index_cache.keys() {
578            if regex.is_match(file_path) {
579                results.push(file_path.clone());
580            }
581        }
582
583        results.sort_unstable();
584        Ok(results)
585    }
586
587    /// Get all indexed files without pattern matching.
588    /// This is more efficient than using find_files(".*").
589    fn all_files(&self) -> Vec<String> {
590        let mut files = self.index_cache.keys().cloned().collect::<Vec<_>>();
591        files.sort_unstable();
592        files
593    }
594
595    /// Get file content with line numbers.
596    fn get_file_content(&self, file_path: &str, start_line: Option<usize>, end_line: Option<usize>) -> Result<String> {
597        let content = fs::read_to_string(file_path)?;
598        let start = start_line.unwrap_or(1).max(1);
599        let end = end_line.unwrap_or(usize::MAX);
600
601        if start > end {
602            return Ok(String::new());
603        }
604
605        let mut result = String::with_capacity(content.len());
606        for (line_number, line) in content.lines().enumerate() {
607            let line_number = line_number + 1;
608            if line_number < start {
609                continue;
610            }
611            if line_number > end {
612                break;
613            }
614            writeln!(&mut result, "{line_number}: {line}")?;
615        }
616
617        Ok(result)
618    }
619
620    /// List files in directory (like ls).
621    pub fn list_files(&self, dir_path: &str, show_hidden: bool) -> Result<Vec<String>> {
622        let path = Path::new(dir_path);
623        if !path.exists() {
624            return Ok(vec![]);
625        }
626
627        let mut files = Vec::new();
628
629        for entry in fs::read_dir(path)? {
630            let entry = entry?;
631            let file_name = entry.file_name().to_string_lossy().into_owned();
632
633            if !show_hidden && file_name.starts_with('.') {
634                continue;
635            }
636
637            files.push(file_name);
638        }
639
640        files.sort_unstable();
641        Ok(files)
642    }
643
644    /// Grep-like search (like grep command).
645    pub fn grep(&self, pattern: &str, file_pattern: Option<&str>) -> Result<Vec<SearchResult>> {
646        let regex = Regex::new(pattern)?;
647        Ok(self.search_files_internal(&regex, file_pattern, false))
648    }
649
650    fn is_allowed_path(&self, path: &Path) -> bool {
651        self.config.allowed_dirs.iter().any(|allowed| path.starts_with(allowed))
652    }
653
654    #[inline]
655    fn get_modified_time(&self, file_path: &Path) -> Result<u64> {
656        let metadata = fs::metadata(file_path)?;
657        let modified = metadata.modified()?;
658        Ok(modified.duration_since(SystemTime::UNIX_EPOCH)?.as_secs())
659    }
660
661    #[inline]
662    fn detect_language(&self, file_path: &Path) -> String {
663        file_path
664            .extension()
665            .and_then(|ext| ext.to_str())
666            .unwrap_or("unknown")
667            .to_string()
668    }
669
670    fn build_file_index(&mut self, file_path: &Path) -> Result<Option<FileIndex>> {
671        if !self.should_process_file_path(file_path) {
672            return Ok(None);
673        }
674
675        let content = match fs::read_to_string(file_path) {
676            Ok(text) => text,
677            Err(err) => {
678                if err.kind() == ErrorKind::InvalidData {
679                    return Ok(None);
680                }
681                return Err(err.into());
682            }
683        };
684
685        let index = FileIndex {
686            path: file_path.to_string_lossy().into_owned(),
687            hash: calculate_hash(&content),
688            modified: self.get_modified_time(file_path)?,
689            size: content.len() as u64,
690            language: self.detect_language(file_path),
691            tags: vec![],
692        };
693
694        self.content_cache
695            .insert(index.path.clone(), (index.hash.clone(), Arc::from(content)));
696
697        Ok(Some(index))
698    }
699
700    #[inline]
701    fn is_excluded_path(&self, path: &Path) -> bool {
702        self.config.excluded_dirs.iter().any(|excluded| path.starts_with(excluded))
703    }
704
705    #[inline]
706    fn should_index_file_path(&self, path: &Path) -> bool {
707        self.filter.should_index_file(path, &self.config)
708    }
709
710    #[inline]
711    fn should_process_file_path(&self, path: &Path) -> bool {
712        if self.is_allowed_path(path) {
713            return self.should_index_file_path(path);
714        }
715
716        !self.is_excluded_path(path) && self.should_index_file_path(path)
717    }
718
719    fn build_walker(&self, dir_path: &Path) -> Walk {
720        let walk_root = dir_path.to_path_buf();
721        let config = self.config.clone();
722        let filter = Arc::clone(&self.filter);
723
724        let mut builder = vtcode_commons::walk::build_default_walker(dir_path);
725        builder.filter_entry(move |entry| should_visit_entry(entry, walk_root.as_path(), &config, filter.as_ref()));
726        builder.build()
727    }
728
729    fn build_shallow_walker(&self, dir_path: &Path) -> Walk {
730        let walk_root = dir_path.to_path_buf();
731        let config = self.config.clone();
732        let filter = Arc::clone(&self.filter);
733
734        let mut builder = vtcode_commons::walk::build_default_walker(dir_path);
735        // Only immediate children — directory navigation lists one level at a time.
736        builder.max_depth(Some(1));
737        builder.filter_entry(move |entry| should_visit_entry(entry, walk_root.as_path(), &config, filter.as_ref()));
738        builder.build()
739    }
740
741    fn replace_cached_entries(&mut self, dir_path: &Path, entries: &[FileIndex]) {
742        self.index_cache.retain(|path, _| !Path::new(path).starts_with(dir_path));
743        self.content_cache.retain(|path, _| !Path::new(path).starts_with(dir_path));
744
745        self.index_cache
746            .extend(entries.iter().cloned().map(|entry| (entry.path.clone(), entry)));
747    }
748
749    fn apply_snapshot_file_update(&mut self, cache_key: String, next_entry: Option<FileIndex>) -> Result<()> {
750        let previous_entry = match next_entry {
751            Some(entry) => self.index_cache.insert(cache_key.clone(), entry),
752            None => self.index_cache.remove(cache_key.as_str()),
753        };
754        let previous_content = self.content_cache.remove(cache_key.as_str());
755
756        if let Err(err) = self.persist_current_snapshot() {
757            match previous_entry {
758                Some(entry) => {
759                    self.index_cache.insert(cache_key.clone(), entry);
760                }
761                None => {
762                    self.index_cache.remove(cache_key.as_str());
763                }
764            }
765            if let Some(content) = previous_content {
766                self.content_cache.insert(cache_key, content);
767            }
768            return Err(err);
769        }
770
771        Ok(())
772    }
773
774    fn apply_snapshot_directory_update(&mut self, dir_path: &Path, entries: &[FileIndex]) -> Result<()> {
775        let previous_entries = self.take_cached_entries(dir_path);
776        self.index_cache
777            .extend(entries.iter().cloned().map(|entry| (entry.path.clone(), entry)));
778
779        if let Err(err) = self.persist_current_snapshot() {
780            self.index_cache.retain(|path, _| !Path::new(path).starts_with(dir_path));
781            self.index_cache
782                .extend(previous_entries.into_iter().map(|entry| (entry.path.clone(), entry)));
783            return Err(err);
784        }
785
786        Ok(())
787    }
788
789    fn take_cached_entries(&mut self, dir_path: &Path) -> Vec<FileIndex> {
790        let keys = self
791            .index_cache
792            .keys()
793            .filter(|path| Path::new(path).starts_with(dir_path))
794            .cloned()
795            .collect::<Vec<_>>();
796
797        keys.into_iter()
798            .filter_map(|path| self.index_cache.remove(path.as_str()))
799            .collect()
800    }
801
802    fn persist_current_snapshot(&self) -> Result<()> {
803        let mut snapshot = self.index_cache.values().collect::<Vec<_>>();
804        snapshot.sort_unstable_by(|left, right| left.path.cmp(&right.path));
805        self.storage.persist_batch_refs(self.config.index_dir(), &snapshot)
806    }
807}
808
809impl Clone for SimpleIndexer {
810    fn clone(&self) -> Self {
811        Self {
812            config: self.config.clone(),
813            index_cache: self.index_cache.clone(),
814            content_cache: self.content_cache.clone(),
815            storage: self.storage.clone(),
816            filter: self.filter.clone(),
817        }
818    }
819}
820
821fn should_skip_dir(path: &Path, config: &SimpleIndexerConfig) -> bool {
822    if is_allowed_path_or_ancestor(path, config) {
823        return false;
824    }
825
826    if config.excluded_dirs.iter().any(|excluded| path.starts_with(excluded)) {
827        return true;
828    }
829
830    if config.ignore_hidden
831        && path
832            .file_name()
833            .and_then(|name| name.to_str())
834            .is_some_and(|name_str| name_str.starts_with('.'))
835    {
836        return true;
837    }
838
839    false
840}
841
842fn is_allowed_path_or_ancestor(path: &Path, config: &SimpleIndexerConfig) -> bool {
843    config
844        .allowed_dirs
845        .iter()
846        .any(|allowed| path.starts_with(allowed) || allowed.starts_with(path))
847}
848
849fn should_visit_entry(
850    entry: &DirEntry,
851    walk_root: &Path,
852    config: &SimpleIndexerConfig,
853    filter: &dyn TraversalFilter,
854) -> bool {
855    if entry.path() == walk_root {
856        return true;
857    }
858
859    if !entry.file_type().is_some_and(|file_type| file_type.is_dir()) {
860        return true;
861    }
862
863    filter.should_descend(entry.path(), config)
864}
865
866#[inline]
867fn calculate_hash(content: &str) -> String {
868    vtcode_commons::utils::calculate_sha256(content.as_bytes())
869}
870
871fn write_markdown_entry(writer: &mut impl Write, entry: &FileIndex) -> std::io::Result<()> {
872    writeln!(writer, "## {}", entry.path)?;
873    writeln!(writer)?;
874    write_markdown_fields(writer, entry)?;
875    writeln!(writer)?;
876    Ok(())
877}
878
879fn write_markdown_fields(writer: &mut impl Write, entry: &FileIndex) -> std::io::Result<()> {
880    writeln!(writer, "- **Path**: {}", entry.path)?;
881    writeln!(writer, "- **Hash**: {}", entry.hash)?;
882    writeln!(writer, "- **Modified**: {}", entry.modified)?;
883    writeln!(writer, "- **Size**: {} bytes", entry.size)?;
884    writeln!(writer, "- **Language**: {}", entry.language)?;
885    writeln!(writer, "- **Tags**: {}", entry.tags.join(", "))?;
886    Ok(())
887}
888
889fn cleanup_legacy_markdown_entries(index_dir: &Path) -> Result<()> {
890    for entry in fs::read_dir(index_dir)? {
891        let entry = entry?;
892        let file_name = entry.file_name();
893        let file_name = file_name.to_string_lossy();
894        if is_legacy_markdown_entry_name(file_name.as_ref()) {
895            fs::remove_file(entry.path())?;
896        }
897    }
898    Ok(())
899}
900
901#[inline]
902fn is_legacy_markdown_entry_name(file_name: &str) -> bool {
903    let Some(hash_part) = file_name.strip_suffix(".md") else {
904        return false;
905    };
906    hash_part.len() == 64 && hash_part.bytes().all(|byte| byte.is_ascii_hexdigit())
907}
908
909#[cfg(test)]
910mod tests {
911    use super::*;
912    use std::fs;
913    use std::sync::{Arc, Mutex};
914    use tempfile::tempdir;
915
916    #[test]
917    fn skips_hidden_directories_by_default() -> Result<()> {
918        let temp = tempdir()?;
919        let workspace = temp.path();
920        let hidden_dir = workspace.join(".private");
921        fs::create_dir_all(&hidden_dir)?;
922        fs::write(hidden_dir.join("secret.txt"), "classified")?;
923
924        let visible_dir = workspace.join("src");
925        fs::create_dir_all(&visible_dir)?;
926        fs::write(visible_dir.join("lib.rs"), "fn main() {}")?;
927
928        let mut indexer = SimpleIndexer::new(workspace.to_path_buf());
929        indexer.init()?;
930        indexer.index_directory(workspace)?;
931
932        assert!(indexer.find_files("secret\\.txt$")?.is_empty());
933        assert!(!indexer.find_files("lib\\.rs$")?.is_empty());
934
935        Ok(())
936    }
937
938    #[test]
939    fn can_include_hidden_directories_when_configured() -> Result<()> {
940        let temp = tempdir()?;
941        let workspace = temp.path();
942        let hidden_dir = workspace.join(".cache");
943        fs::create_dir_all(&hidden_dir)?;
944        fs::write(hidden_dir.join("data.log"), "details")?;
945
946        let config = SimpleIndexerConfig::new(workspace.to_path_buf()).ignore_hidden(false);
947        let mut indexer = SimpleIndexer::with_config(config);
948        indexer.init()?;
949        indexer.index_directory(workspace)?;
950
951        let results = indexer.find_files("data\\.log$")?;
952        assert_eq!(results.len(), 1);
953
954        Ok(())
955    }
956
957    #[test]
958    fn indexes_allowed_directories_inside_hidden_excluded_parents() -> Result<()> {
959        let temp = tempdir()?;
960        let workspace = temp.path();
961        let allowed_dir = workspace.join(".vtcode").join("external");
962        fs::create_dir_all(&allowed_dir)?;
963        fs::write(allowed_dir.join("plugin.toml"), "name = 'demo'")?;
964
965        let mut indexer = SimpleIndexer::new(workspace.to_path_buf());
966        indexer.init()?;
967        indexer.index_directory(workspace)?;
968
969        let results = indexer.find_files("plugin\\.toml$")?;
970        assert_eq!(results.len(), 1);
971
972        Ok(())
973    }
974
975    #[test]
976    fn reindexing_prunes_deleted_files_from_cache() -> Result<()> {
977        let temp = tempdir()?;
978        let workspace = temp.path();
979        let file_path = workspace.join("notes.txt");
980        fs::write(&file_path, "remember this")?;
981
982        let mut indexer = SimpleIndexer::new(workspace.to_path_buf());
983        indexer.init()?;
984        indexer.index_directory(workspace)?;
985        assert_eq!(indexer.find_files("notes\\.txt$")?.len(), 1);
986
987        fs::remove_file(&file_path)?;
988        indexer.index_directory(workspace)?;
989
990        assert!(indexer.find_files("notes\\.txt$")?.is_empty());
991        assert!(indexer.all_files().is_empty());
992
993        Ok(())
994    }
995
996    #[test]
997    fn index_file_skips_excluded_paths() -> Result<()> {
998        let temp = tempdir()?;
999        let workspace = temp.path();
1000        let index_dir = workspace.join(".vtcode").join("index");
1001        fs::create_dir_all(&index_dir)?;
1002        let generated_index = index_dir.join("index.md");
1003        fs::write(&generated_index, "# generated")?;
1004
1005        let mut indexer = SimpleIndexer::new(workspace.to_path_buf());
1006        indexer.init()?;
1007        indexer.index_file(&generated_index)?;
1008
1009        assert!(indexer.all_files().is_empty());
1010
1011        Ok(())
1012    }
1013
1014    #[test]
1015    fn index_file_removes_stale_entry_when_file_becomes_unreadable() -> Result<()> {
1016        let temp = tempdir()?;
1017        let workspace = temp.path();
1018        let file_path = workspace.join("notes.txt");
1019        fs::write(&file_path, "remember this")?;
1020
1021        let mut indexer = SimpleIndexer::new(workspace.to_path_buf());
1022        indexer.init()?;
1023        indexer.index_file(&file_path)?;
1024        assert!(
1025            indexer
1026                .find_files("notes\\.txt$")?
1027                .iter()
1028                .any(|file| file.ends_with("notes.txt"))
1029        );
1030
1031        fs::write(&file_path, [0xFF, 0xFE, 0xFD])?;
1032        indexer.index_file(&file_path)?;
1033
1034        assert!(indexer.find_files("notes\\.txt$")?.is_empty());
1035
1036        let index_content = fs::read_to_string(workspace.join(".vtcode").join("index").join("index.md"))?;
1037        assert!(!index_content.contains(file_path.to_string_lossy().as_ref()));
1038
1039        Ok(())
1040    }
1041
1042    #[test]
1043    fn index_file_maintains_markdown_snapshot_across_updates() -> Result<()> {
1044        let temp = tempdir()?;
1045        let workspace = temp.path();
1046        let first = workspace.join("first.txt");
1047        let second = workspace.join("second.txt");
1048        fs::write(&first, "one")?;
1049        fs::write(&second, "two")?;
1050
1051        let mut indexer = SimpleIndexer::new(workspace.to_path_buf());
1052        indexer.init()?;
1053        indexer.index_file(&first)?;
1054        indexer.index_file(&second)?;
1055
1056        let index_dir = workspace.join(".vtcode").join("index");
1057        let files = fs::read_dir(&index_dir)?
1058            .filter_map(|entry| entry.ok())
1059            .map(|entry| entry.file_name().to_string_lossy().into_owned())
1060            .collect::<Vec<_>>();
1061        assert_eq!(files, vec!["index.md".to_string()]);
1062
1063        let index_content = fs::read_to_string(index_dir.join("index.md"))?;
1064        assert!(index_content.contains(first.to_string_lossy().as_ref()));
1065        assert!(index_content.contains(second.to_string_lossy().as_ref()));
1066
1067        Ok(())
1068    }
1069
1070    #[test]
1071    fn index_directory_writes_markdown_snapshot_without_manual_init() -> Result<()> {
1072        let temp = tempdir()?;
1073        let workspace = temp.path();
1074        fs::write(workspace.join("notes.txt"), "remember this")?;
1075
1076        let mut indexer = SimpleIndexer::new(workspace.to_path_buf());
1077        indexer.index_directory(workspace)?;
1078
1079        let index_content = fs::read_to_string(workspace.join(".vtcode").join("index").join("index.md"))?;
1080        assert!(index_content.contains(workspace.join("notes.txt").to_string_lossy().as_ref()));
1081
1082        Ok(())
1083    }
1084
1085    #[test]
1086    fn get_file_content_clamps_ranges_without_panicking() -> Result<()> {
1087        let temp = tempdir()?;
1088        let workspace = temp.path();
1089        let file_path = workspace.join("notes.txt");
1090        fs::write(&file_path, "first\nsecond")?;
1091
1092        let indexer = SimpleIndexer::new(workspace.to_path_buf());
1093        let file_path = file_path.to_string_lossy().into_owned();
1094
1095        assert_eq!(indexer.get_file_content(&file_path, Some(5), None)?, "");
1096        assert_eq!(indexer.get_file_content(&file_path, Some(0), Some(1))?, "1: first\n");
1097        assert_eq!(indexer.get_file_content(&file_path, Some(2), Some(1))?, "");
1098
1099        Ok(())
1100    }
1101
1102    #[test]
1103    fn supports_custom_storage_backends() -> Result<()> {
1104        #[derive(Clone, Default)]
1105        struct MemoryStorage {
1106            records: Arc<Mutex<Vec<FileIndex>>>,
1107        }
1108
1109        impl MemoryStorage {
1110            fn new(records: Arc<Mutex<Vec<FileIndex>>>) -> Self {
1111                Self { records }
1112            }
1113        }
1114
1115        impl IndexStorage for MemoryStorage {
1116            fn init(&self, _index_dir: &Path) -> Result<()> {
1117                Ok(())
1118            }
1119
1120            fn persist(&self, _index_dir: &Path, entry: &FileIndex) -> Result<()> {
1121                let mut guard = self.records.lock().expect("lock poisoned");
1122                guard.push(entry.clone());
1123                Ok(())
1124            }
1125        }
1126
1127        let temp = tempdir()?;
1128        let workspace = temp.path();
1129        fs::write(workspace.join("notes.txt"), "remember this")?;
1130
1131        let records: Arc<Mutex<Vec<FileIndex>>> = Arc::new(Mutex::new(Vec::new()));
1132        let storage = MemoryStorage::new(records.clone());
1133
1134        let config = SimpleIndexerConfig::new(workspace.to_path_buf());
1135        let mut indexer = SimpleIndexer::with_config(config).with_storage(Arc::new(storage));
1136        indexer.init()?;
1137        indexer.index_directory(workspace)?;
1138
1139        let entries = records.lock().expect("lock poisoned");
1140        assert_eq!(entries.len(), 1);
1141        assert_eq!(entries[0].path, workspace.join("notes.txt").to_string_lossy().into_owned());
1142
1143        Ok(())
1144    }
1145
1146    #[test]
1147    fn custom_filters_can_skip_files() -> Result<()> {
1148        #[derive(Default)]
1149        struct SkipRustFilter {
1150            inner: ConfigTraversalFilter,
1151        }
1152
1153        impl TraversalFilter for SkipRustFilter {
1154            fn should_descend(&self, path: &Path, config: &SimpleIndexerConfig) -> bool {
1155                self.inner.should_descend(path, config)
1156            }
1157
1158            fn should_index_file(&self, path: &Path, config: &SimpleIndexerConfig) -> bool {
1159                if path
1160                    .extension()
1161                    .and_then(|ext| ext.to_str())
1162                    .is_some_and(|ext| ext.eq_ignore_ascii_case("rs"))
1163                {
1164                    return false;
1165                }
1166
1167                self.inner.should_index_file(path, config)
1168            }
1169        }
1170
1171        let temp = tempdir()?;
1172        let workspace = temp.path();
1173        fs::write(workspace.join("lib.rs"), "fn main() {}")?;
1174        fs::write(workspace.join("README.md"), "# Notes")?;
1175
1176        let config = SimpleIndexerConfig::new(workspace.to_path_buf());
1177        let mut indexer = SimpleIndexer::with_config(config).with_filter(Arc::new(SkipRustFilter::default()));
1178        indexer.init()?;
1179        indexer.index_directory(workspace)?;
1180
1181        assert!(indexer.find_files("lib\\.rs$")?.is_empty());
1182        assert!(!indexer.find_files("README\\.md$")?.is_empty());
1183
1184        Ok(())
1185    }
1186
1187    #[test]
1188    fn custom_filters_can_skip_directories() -> Result<()> {
1189        #[derive(Default)]
1190        struct SkipGeneratedFilter {
1191            inner: ConfigTraversalFilter,
1192        }
1193
1194        impl TraversalFilter for SkipGeneratedFilter {
1195            fn should_descend(&self, path: &Path, config: &SimpleIndexerConfig) -> bool {
1196                if path.ends_with("generated") {
1197                    return false;
1198                }
1199
1200                self.inner.should_descend(path, config)
1201            }
1202
1203            fn should_index_file(&self, path: &Path, config: &SimpleIndexerConfig) -> bool {
1204                self.inner.should_index_file(path, config)
1205            }
1206        }
1207
1208        let temp = tempdir()?;
1209        let workspace = temp.path();
1210        let generated_dir = workspace.join("generated");
1211        fs::create_dir_all(&generated_dir)?;
1212        fs::write(generated_dir.join("skip.txt"), "ignore me")?;
1213        fs::write(workspace.join("README.md"), "# Notes")?;
1214
1215        let config = SimpleIndexerConfig::new(workspace.to_path_buf());
1216        let indexer = SimpleIndexer::with_config(config).with_filter(Arc::new(SkipGeneratedFilter::default()));
1217        let files = indexer.discover_files(workspace);
1218
1219        assert!(!files.iter().any(|file| file.ends_with("skip.txt")));
1220        assert!(files.iter().any(|file| file.ends_with("README.md")));
1221
1222        Ok(())
1223    }
1224
1225    #[test]
1226    fn discover_dir_entries_is_shallow_and_ignore_aware() -> Result<()> {
1227        let temp = tempdir()?;
1228        let workspace = temp.path();
1229        fs::create_dir_all(workspace.join("src"))?;
1230        fs::create_dir_all(workspace.join("node_modules"))?;
1231        fs::create_dir_all(workspace.join(".git"))?;
1232        fs::write(workspace.join("README.md"), "# Notes")?;
1233        fs::write(workspace.join("src").join("lib.rs"), "fn main() {}")?;
1234        fs::write(workspace.join("node_modules").join("dep.js"), "x")?;
1235        fs::write(workspace.join(".git").join("config"), "x")?;
1236
1237        let config = SimpleIndexerConfig::new(workspace.to_path_buf());
1238        let indexer = SimpleIndexer::with_config(config);
1239        let entries = indexer.discover_dir_entries(workspace);
1240
1241        let names: Vec<String> = entries
1242            .iter()
1243            .map(|(p, _)| p.file_name().unwrap().to_string_lossy().into_owned())
1244            .collect();
1245
1246        assert_eq!(entries.len(), 2, "expected exactly README.md and src, got {names:?}");
1247        assert!(names.contains(&"README.md".to_string()));
1248        assert!(names.contains(&"src".to_string()));
1249
1250        let (_, src_is_dir) = entries.iter().find(|(p, _)| p.ends_with("src")).unwrap();
1251        assert!(*src_is_dir);
1252
1253        Ok(())
1254    }
1255
1256    #[test]
1257    fn indexing_multiple_directories_preserves_existing_cache_entries() -> Result<()> {
1258        let temp = tempdir()?;
1259        let workspace = temp.path();
1260        let src_dir = workspace.join("src");
1261        let docs_dir = workspace.join("docs");
1262        fs::create_dir_all(&src_dir)?;
1263        fs::create_dir_all(&docs_dir)?;
1264        fs::write(src_dir.join("lib.rs"), "fn main() {}")?;
1265        fs::write(docs_dir.join("guide.md"), "# Guide")?;
1266
1267        let mut indexer = SimpleIndexer::new(workspace.to_path_buf());
1268        indexer.init()?;
1269        indexer.index_directory(&src_dir)?;
1270        indexer.index_directory(&docs_dir)?;
1271
1272        assert!(indexer.find_files("lib\\.rs$")?.iter().any(|file| file.ends_with("lib.rs")));
1273        assert!(indexer.find_files("guide\\.md$")?.iter().any(|file| file.ends_with("guide.md")));
1274
1275        let index_content = fs::read_to_string(workspace.join(".vtcode").join("index").join("index.md"))?;
1276        assert!(index_content.contains(src_dir.join("lib.rs").to_string_lossy().as_ref()));
1277        assert!(index_content.contains(docs_dir.join("guide.md").to_string_lossy().as_ref()));
1278
1279        Ok(())
1280    }
1281
1282    #[test]
1283    fn batch_indexing_writes_single_markdown_file() -> Result<()> {
1284        let temp = tempdir()?;
1285        let workspace = temp.path();
1286        fs::write(workspace.join("lib.rs"), "fn main() {}")?;
1287        fs::write(workspace.join("README.md"), "# Notes")?;
1288
1289        let mut indexer = SimpleIndexer::new(workspace.to_path_buf());
1290        indexer.init()?;
1291        indexer.index_directory(workspace)?;
1292
1293        let index_dir = workspace.join(".vtcode").join("index");
1294        let files = fs::read_dir(&index_dir)?
1295            .filter_map(|entry| entry.ok())
1296            .map(|entry| entry.file_name().to_string_lossy().into_owned())
1297            .collect::<Vec<_>>();
1298        assert_eq!(files, vec!["index.md".to_string()]);
1299
1300        let index_content = fs::read_to_string(index_dir.join("index.md"))?;
1301        assert!(index_content.contains(workspace.join("lib.rs").to_string_lossy().as_ref()));
1302        assert!(index_content.contains(workspace.join("README.md").to_string_lossy().as_ref()));
1303
1304        Ok(())
1305    }
1306
1307    #[test]
1308    fn batch_indexing_removes_legacy_hashed_entries() -> Result<()> {
1309        let temp = tempdir()?;
1310        let workspace = temp.path();
1311        fs::write(workspace.join("lib.rs"), "fn main() {}")?;
1312
1313        let mut indexer = SimpleIndexer::new(workspace.to_path_buf());
1314        indexer.init()?;
1315
1316        let legacy_file_name = format!("{}.md", calculate_hash("legacy-path"));
1317        let legacy_file_path = workspace.join(".vtcode").join("index").join(&legacy_file_name);
1318        fs::write(&legacy_file_path, "# legacy")?;
1319        assert!(legacy_file_path.exists());
1320
1321        indexer.index_directory(workspace)?;
1322
1323        assert!(!legacy_file_path.exists());
1324        let files = fs::read_dir(workspace.join(".vtcode").join("index"))?
1325            .filter_map(|entry| entry.ok())
1326            .map(|entry| entry.file_name().to_string_lossy().into_owned())
1327            .collect::<Vec<_>>();
1328        assert_eq!(files, vec!["index.md".to_string()]);
1329
1330        Ok(())
1331    }
1332
1333    #[test]
1334    fn snapshot_storage_uses_default_ref_batch_persistence() -> Result<()> {
1335        #[derive(Clone, Default)]
1336        struct SnapshotMemoryStorage {
1337            snapshots: Arc<Mutex<Vec<Vec<FileIndex>>>>,
1338        }
1339
1340        impl SnapshotMemoryStorage {
1341            fn new(snapshots: Arc<Mutex<Vec<Vec<FileIndex>>>>) -> Self {
1342                Self { snapshots }
1343            }
1344        }
1345
1346        impl IndexStorage for SnapshotMemoryStorage {
1347            fn init(&self, _index_dir: &Path) -> Result<()> {
1348                Ok(())
1349            }
1350
1351            fn persist(&self, _index_dir: &Path, _entry: &FileIndex) -> Result<()> {
1352                Ok(())
1353            }
1354
1355            fn prefers_snapshot_persistence(&self) -> bool {
1356                true
1357            }
1358
1359            fn persist_batch(&self, _index_dir: &Path, entries: &[FileIndex]) -> Result<()> {
1360                self.snapshots.lock().expect("lock poisoned").push(entries.to_vec());
1361                Ok(())
1362            }
1363        }
1364
1365        let temp = tempdir()?;
1366        let workspace = temp.path();
1367        let file_path = workspace.join("notes.txt");
1368        fs::write(&file_path, "remember this")?;
1369
1370        let snapshots = Arc::new(Mutex::new(Vec::new()));
1371        let storage = SnapshotMemoryStorage::new(snapshots.clone());
1372
1373        let config = SimpleIndexerConfig::new(workspace.to_path_buf());
1374        let mut indexer = SimpleIndexer::with_config(config).with_storage(Arc::new(storage));
1375        indexer.index_file(&file_path)?;
1376
1377        let snapshots = snapshots.lock().expect("lock poisoned");
1378        assert_eq!(snapshots.len(), 1);
1379        assert_eq!(snapshots[0].len(), 1);
1380        assert_eq!(snapshots[0][0].path, workspace.join("notes.txt").to_string_lossy().into_owned());
1381
1382        Ok(())
1383    }
1384
1385    #[test]
1386    fn snapshot_index_file_rolls_back_cache_when_persist_fails() -> Result<()> {
1387        #[derive(Clone, Default)]
1388        struct FlakySnapshotStorage {
1389            persist_count: Arc<Mutex<usize>>,
1390        }
1391
1392        impl IndexStorage for FlakySnapshotStorage {
1393            fn init(&self, _index_dir: &Path) -> Result<()> {
1394                Ok(())
1395            }
1396
1397            fn persist(&self, _index_dir: &Path, _entry: &FileIndex) -> Result<()> {
1398                Ok(())
1399            }
1400
1401            fn prefers_snapshot_persistence(&self) -> bool {
1402                true
1403            }
1404
1405            fn persist_batch(&self, _index_dir: &Path, _entries: &[FileIndex]) -> Result<()> {
1406                let mut count = self.persist_count.lock().expect("lock poisoned");
1407                *count += 1;
1408                if *count == 2 {
1409                    anyhow::bail!("simulated snapshot persistence failure");
1410                }
1411                Ok(())
1412            }
1413        }
1414
1415        let temp = tempdir()?;
1416        let workspace = temp.path();
1417        let first = workspace.join("first.txt");
1418        let second = workspace.join("second.txt");
1419        fs::write(&first, "one")?;
1420        fs::write(&second, "two")?;
1421
1422        let config = SimpleIndexerConfig::new(workspace.to_path_buf());
1423        let storage = Arc::new(FlakySnapshotStorage::default());
1424        let mut indexer = SimpleIndexer::with_config(config).with_storage(storage);
1425
1426        indexer.index_file(&first)?;
1427        assert!(
1428            indexer
1429                .find_files("first\\.txt$")?
1430                .iter()
1431                .any(|path| path.ends_with("first.txt"))
1432        );
1433
1434        let err = indexer.index_file(&second).expect_err("second persist should fail");
1435        assert!(err.to_string().contains("simulated snapshot persistence failure"));
1436        assert!(
1437            indexer
1438                .find_files("first\\.txt$")?
1439                .iter()
1440                .any(|path| path.ends_with("first.txt"))
1441        );
1442        assert!(indexer.find_files("second\\.txt$")?.is_empty());
1443
1444        Ok(())
1445    }
1446}