1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
//! greplm-core: an extreme-performance, trigram-based code indexer for LLM agents.
//!
//! The index lives in a `.greplm/` directory at the project root and consists of
//! immutable, mmap-backed segments (trigram FST + roaring posting lists + doc and
//! symbol tables). Search filters candidate documents by trigram intersection,
//! then verifies matches with the real literal/regex matcher.
mod error;
pub mod cache;
pub mod client;
pub mod config;
pub mod context;
pub mod daemon;
pub(crate) mod fsutil;
pub mod git;
pub mod indexer;
pub mod io_backend;
pub mod lang;
pub mod meta;
pub mod paths;
pub mod proto;
pub mod resolve;
pub mod savings;
pub mod search;
pub mod segment;
#[cfg(feature = "semantic")]
pub mod semantic;
pub mod structural;
pub mod symbol;
pub(crate) mod table;
pub mod trigram;
pub mod walk;
pub mod watch;
pub use error::{Error, Result};
/// Test-only crash/fault-injection seam for the atomic-write path (see
/// [`fsutil::faults`]). Hidden from public docs; used by greplm's own
/// durability tests to simulate crashes mid-index/compaction.
#[doc(hidden)]
pub use fsutil::faults;
use std::path::{Path, PathBuf};
use config::Config;
use indexer::{IndexStats, Indexer};
use io_backend::IoBackend;
use meta::Meta;
use paths::Paths;
use search::Searcher;
/// Status snapshot for `greplm status`.
#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)]
pub struct Status {
pub root: PathBuf,
pub indexed: bool,
pub segments: usize,
pub doc_count: u64,
pub symbol_count: u64,
pub last_indexed: u64,
pub backend: String,
}
/// A handle to a greplm project (a directory and its `.greplm` index).
pub struct Greplm {
paths: Paths,
config: Config,
backend: Box<dyn IoBackend>,
/// Serializes index mutation within this process. The daemon can drive two
/// indexers at once — the background watcher and a `strict`-freshness query
/// that reindexes before answering — and two concurrent incremental passes
/// over the same `.greplm` would race on segment writes and the manifest.
/// Held for the duration of [`index`](Self::index).
index_lock: std::sync::Mutex<()>,
}
impl Greplm {
/// Open (and lazily initialize) the index for `root`.
pub fn open(root: impl AsRef<Path>) -> Result<Greplm> {
let paths = Paths::new(root);
let config = Config::load(&paths.config_file())?;
Ok(Greplm {
backend: io_backend::select(config.backend),
paths,
config,
index_lock: std::sync::Mutex::new(()),
})
}
/// Find the nearest ancestor of `start` containing a `.greplm` directory,
/// falling back to `start` itself if none is found.
pub fn discover(start: impl AsRef<Path>) -> Result<Greplm> {
let start = start.as_ref();
let mut cur = Some(start);
while let Some(dir) = cur {
if dir.join(paths::DIR_NAME).is_dir() {
return Greplm::open(dir);
}
cur = dir.parent();
}
Greplm::open(start)
}
pub fn root(&self) -> &Path {
&self.paths.root
}
pub fn config(&self) -> &Config {
&self.config
}
/// Ensure the `.greplm` directory exists with a default config and gitignore.
pub fn ensure_initialized(&self) -> Result<()> {
std::fs::create_dir_all(self.paths.segments_dir())
.map_err(|e| Error::io(self.paths.segments_dir(), e))?;
let cfg = self.paths.config_file();
if !cfg.exists() {
self.config.save(&cfg)?;
}
let gi = self.paths.gitignore_file();
if !gi.exists() {
std::fs::write(&gi, "*\n").map_err(|e| Error::io(&gi, e))?;
}
Ok(())
}
/// Acquire the index-mutation lock. Held by [`index`](Self::index) for the
/// duration of a build, and exposed so the daemon can serialize a searcher
/// hot-swap (manifest read + publish) against indexing *and against other
/// swaps*. Without that, two concurrent swaps (the watcher and a
/// `strict`-freshness query reacting to the same edit) can store
/// out of order, letting an older searcher clobber a newer one and leaving
/// the index stale until the next file event.
///
/// A poisoned lock just means a prior holder panicked — on-disk state is
/// still consistent (atomic writes), so recover the guard and proceed.
pub fn index_guard(&self) -> std::sync::MutexGuard<'_, ()> {
self.index_lock.lock().unwrap_or_else(|e| e.into_inner())
}
/// Build or refresh the index.
pub fn index(&self, force: bool) -> Result<IndexStats> {
self.ensure_initialized()?;
// Serialize concurrent in-process indexers (watcher vs. strict query).
let _guard = self.index_guard();
let indexer = Indexer::new(&self.paths, &self.config, self.backend.as_ref());
if force {
indexer.index_full()
} else {
indexer.index_incremental()
}
}
/// Ensure a usable, current-schema index exists, building it if absent,
/// empty, or left unreadable by an on-disk format change. Returns `true` if
/// a (re)build happened. This is the self-healing entry point for query
/// paths: a fresh checkout or a post-upgrade stale index transparently
/// builds instead of erroring with "run `greplm index` first".
///
/// Cheap when a good index already exists (one manifest read). The actual
/// rebuild-on-corrupt logic lives in [`Indexer::index_incremental`], which
/// falls back to a full rebuild on an unreadable/outdated manifest.
pub fn ensure_indexed(&self) -> Result<bool> {
match self.status() {
// A populated, current-schema index — nothing to do.
Ok(s) if s.indexed => return Ok(false),
// Initialized but empty, or readable-but-empty manifest: build.
Ok(_) => {}
// Unreadable/outdated manifest (e.g. schema bump): index() rebuilds.
Err(_) => {}
}
self.index(false)?;
Ok(true)
}
/// Stat-only freshness probe: does any file on disk differ from what the
/// index recorded (new, modified by size/mtime, or deleted)? No content
/// hashing and no reads — just the same cheap pre-check the incremental
/// indexer uses. The daemon calls this to guarantee read-after-write
/// consistency: if dirty, it reindexes before answering.
pub fn is_dirty(&self) -> Result<bool> {
let cache = cache::Cache::open(&self.paths.cache_file())?;
let existing = match cache.load_all() {
Ok(existing) => existing,
// An undecodable cache is a lost optimization, not a probe failure:
// report dirty so the caller reindexes, which rebuilds the cache
// (see `Indexer::index_incremental`). Never surface a hard error here.
Err(Error::Postcard(_)) => return Ok(true),
Err(e) => return Err(e),
};
let walked = walk::walk(&self.paths, &self.config)?;
let mut seen = std::collections::HashSet::with_capacity(walked.entries.len());
for e in &walked.entries {
seen.insert(e.rel.clone());
let (_, mtime_ns, size) = cache::stat_key(&e.metadata);
match existing.get(&e.rel) {
// Already indexed; a changed stat key means it may have changed.
Some(rec) if rec.size == size && rec.mtime_ns == mtime_ns => {}
Some(_) => return Ok(true),
// Not in the index. It's only "dirty" if the indexer would
// actually index it — files it intentionally skips (binary) are
// never cached, so counting them would make any project with a
// binary file look perpetually stale.
None => {
if self.would_index(&e.path) {
return Ok(true);
}
}
}
}
// A previously indexed file that disappeared (or became un-indexable,
// e.g. now too large/binary and dropped from the walk) is also dirty.
for path in existing.keys() {
if !seen.contains(path) {
return Ok(true);
}
}
Ok(false)
}
/// Would the indexer index this not-yet-cached file, or skip it the way its
/// read stage does (binary content, unreadable)? Mirrors `indexer::process`
/// so [`is_dirty`](Self::is_dirty) doesn't flag intentionally-skipped files.
fn would_index(&self, path: &Path) -> bool {
match std::fs::read(path) {
Ok(data) => self.config.index_binary || memchr::memchr(0, &data).is_none(),
Err(_) => false,
}
}
/// Merge all segments into a single compact segment, dropping tombstoned
/// documents. Falls back to a full rebuild if the merge cannot proceed.
pub fn compact(&self) -> Result<IndexStats> {
self.ensure_initialized()?;
Indexer::new(&self.paths, &self.config, self.backend.as_ref()).compact()
}
/// Open a searcher over the current index.
pub fn searcher(&self) -> Result<Searcher> {
Searcher::open(&self.paths)
}
/// Open a searcher over the current index, sharing unchanged segments and
/// the warm content cache with `prev` (see [`Searcher::open_reusing`]).
/// The daemon uses this so a hot-swap after an incremental index re-reads
/// only what actually changed instead of re-parsing every segment.
pub fn searcher_reusing(&self, prev: &Searcher) -> Result<Searcher> {
Searcher::open_reusing(&self.paths, prev)
}
/// Content search that always returns results: it queries the index, and if
/// the index is missing or errors, transparently falls back to an
/// index-free walk+scan (grep parity). The fallback is logged at WARN.
pub fn search_or_grep(&self, query: &search::SearchQuery) -> Result<Vec<search::SearchHit>> {
match self.searcher().and_then(|s| s.search(query)) {
Ok(hits) => Ok(hits),
Err(e) => {
tracing::warn!("index unavailable ({e}); falling back to grep walk");
search::grep_walk(&self.paths, &self.config, query)
}
}
}
/// Report index status.
pub fn status(&self) -> Result<Status> {
let meta = if self.paths.meta_file().exists() {
Meta::load(&self.paths.meta_file())?
} else {
Meta::default()
};
Ok(Status {
root: self.paths.root.clone(),
indexed: !meta.segments.is_empty(),
segments: meta.segments.len(),
doc_count: meta.doc_count,
symbol_count: meta.symbol_count,
last_indexed: meta.last_indexed,
backend: self.backend.name().to_string(),
})
}
/// Remove the entire `.greplm` directory.
pub fn clean(&self) -> Result<()> {
if self.paths.base.is_dir() {
std::fs::remove_dir_all(&self.paths.base)
.map_err(|e| Error::io(&self.paths.base, e))?;
}
Ok(())
}
/// Watch the project tree and re-index incrementally on changes.
///
/// `on_change` is called after each successful incremental update. This call
/// blocks until an error occurs or the watcher is dropped.
pub fn watch<F: FnMut(&IndexStats)>(
&self,
debounce: std::time::Duration,
on_change: F,
) -> Result<()> {
watch::run(self, debounce, on_change)
}
/// Path to the daemon's Unix socket for this project.
pub fn socket_path(&self) -> PathBuf {
self.paths.base.join(proto::SOCKET_NAME)
}
/// Record a query's token savings (grep+read baseline vs. returned payload).
/// Best-effort; never fails a query.
pub fn record_savings(
&self,
kind: &str,
files: &std::collections::BTreeSet<String>,
returned_chars: u64,
results: u64,
) {
savings::record(&self.paths, kind, files, returned_chars, results);
}
/// Aggregate the recorded token-savings log.
pub fn savings_report(&self) -> savings::SavingsReport {
savings::report(&self.paths)
}
pub(crate) fn paths(&self) -> &Paths {
&self.paths
}
}