Skip to main content

cosh_tools/find/
mod.rs

1//! File-search tools built on the `cosh-sdk` find engine.
2//!
3//! - [`glob::glob`]: find files and directories by glob pattern.
4//! - [`grep::grep`]: search file content with a regex.
5//!
6//! [`Find`] wraps both tools in a single builder — configure shared options
7//! once and call either operation.
8//!
9//! # Example
10//!
11//! ```ignore
12//! use cosh_tools::find::Find;
13//!
14//! let f = Find::new()
15//!     .recursive(true)
16//!     .max_results(20)
17//!     .gitignore(true);
18//!
19//! f.glob("*.rs", "/project/src");
20//! f.grep("fn main", "/project/src");
21//! ```
22
23pub mod glob;
24pub mod grep;
25pub mod types;
26
27#[cfg(test)]
28mod test;
29
30pub use glob::{GlobMatchCallback, GlobTargetSpec, glob, glob_targets_with, glob_with};
31pub use grep::{GrepMatchCallback, grep, grep_targets, grep_targets_with};
32pub use types::{
33    ContextEntry, Glob, GlobCallOptions, GlobEntry, GlobInput, GlobOutput, Grep, GrepFileEntry,
34    GrepInput, GrepMatchEntry, GrepOutput,
35};
36
37use std::path::{Path, PathBuf};
38use std::sync::Arc;
39
40/// Deepest common ancestor directory of the given (existing) paths. A file
41/// target contributes its parent directory. Returns `None` only for an empty
42/// input.
43///
44/// Note: when the targets share NO common ancestor (e.g. different mount
45/// points), the rebase falls back to each target's raw relative paths, which
46/// can collide across targets. On a single-`/` filesystem every absolute path
47/// shares `/`, so this only matters for exotic layouts.
48pub(crate) fn common_ancestor(paths: &[PathBuf]) -> Option<PathBuf> {
49    let mut iter = paths.iter();
50    let first = iter.next()?;
51    let mut ancestor = if first.is_file() {
52        first.parent()?.to_path_buf()
53    } else {
54        first.clone()
55    };
56    for path in iter {
57        while !path.starts_with(&ancestor) {
58            if !ancestor.pop() {
59                return None;
60            }
61        }
62    }
63    Some(ancestor)
64}
65
66use crate::ToolDescription;
67use crate::util::path_guard::PathGuard;
68use glob::parse_find_pattern;
69use types::Grep as GrepConfig;
70
71/// Shared-state wrapper for file-search tool operations.
72///
73/// Holds configuration for both [`glob`] and [`grep`] searches. Fields
74/// that are not set via builder methods remain `None`, deferring to each
75/// tool's internal default.
76pub struct Find {
77    file_type: Option<String>,
78    recursive: Option<bool>,
79    max_results: Option<u32>,
80    sort_by_mtime: Option<bool>,
81    format: Option<String>,
82    name_glob: Option<String>,
83    language: Option<String>,
84    ignore_case: Option<bool>,
85    max_count: Option<u32>,
86    context_before: Option<u32>,
87    context_after: Option<u32>,
88    hidden: Option<bool>,
89    gitignore: Option<bool>,
90    timeout_ms: Option<u32>,
91
92    /// Centralized path guard for path-validation.
93    guard: PathGuard,
94
95    /// MCP Tool description for `glob`.
96    pub description_glob: ToolDescription,
97    /// MCP Tool description for `grep`.
98    pub description_grep: ToolDescription,
99}
100
101impl Default for Find {
102    fn default() -> Self {
103        Self::new()
104    }
105}
106
107impl Find {
108    /// Create a new `Find` with all options unset (tool defaults apply).
109    #[must_use]
110    pub fn new() -> Self {
111        Self {
112            file_type: None,
113            recursive: None,
114            max_results: None,
115            sort_by_mtime: None,
116            name_glob: None,
117            language: None,
118            ignore_case: None,
119            max_count: None,
120            context_before: None,
121            context_after: None,
122            hidden: None,
123            gitignore: None,
124            guard: PathGuard::new(Path::new(""), None, None),
125            timeout_ms: None,
126            format: None,
127            description_glob: serde_json::json!({
128                "name": "find_glob",
129                "description": concat!(
130                    "Find files and directories matching a glob pattern. ",
131                    "Search one or more roots in a single call: `paths` (array) ",
132                    "overrides `path`; each entry is a root, glob, or literal ",
133                    "file/directory (e.g. \"src/**/*.rs\", \"*.rs\", or \"src\"). ",
134                    "Literal directories are searched recursively; a bare glob (no ",
135                    "directory prefix) is matched recursively across its root; a ",
136                    "glob WITH a directory prefix (\"src/*.rs\") stays scoped to ",
137                    "that directory (shallow). Missing targets are skipped with a ",
138                    "warning. Zero matches are marked `useless` with a `note` ",
139                    "(adjust the pattern or scope, do not retry blindly). A ",
140                    "timeout returns the partial results with `timed_out: true` — ",
141                    "an empty timed-out result is an INCOMPLETE scan, NOT proof of ",
142                    "absence; scope to a deeper directory instead of retrying. ",
143                    "Results are capped at 200 by default; the optional ",
144                    "`max_results` only LOWERS that cap (a value above 200 ",
145                    "is clamped to 200). ",
146                    "When the cap cuts the list, `limit_reached: true` means more ",
147                    "entries may exist. `format` renders the `formatted` ",
148                    "field as \"flat\", \"grouped\" (per-directory headers), or ",
149                    "\"tree\" (indented). Each entry carries path, file type, ",
150                    "size, and modification time."
151                ),
152                "inputSchema": {
153                    "type": "object",
154                    "properties": {
155                        "pattern": {
156                            "type": "string",
157                            "description": "The glob pattern to match (e.g. \"**/*.rs\", \"src/**\", \"*.toml\")"
158                        },
159                        "path": {
160                            "type": "string",
161                            "description": "A search root, glob, or literal file/directory. May be relative; CWD-relative results are returned. Omitted: defaults to the working directory."
162                        },
163                        "paths": {
164                            "type": "array",
165                            "items": { "type": "string" },
166                            "description": "Multiple search roots/globs in one call; overrides `path`. Each target is validated individually; missing targets are skipped with a warning."
167                        },
168                        "file_type": {
169                            "type": "string",
170                            "enum": ["file", "dir", "symlink"],
171                            "description": "Restrict results to one filesystem kind. Omitted: return files, dirs, and symlinks mixed."
172                        },
173                        "hidden": {
174                            "type": "boolean",
175                            "description": "Include hidden files/directories (names starting with `.`). Default true; `.git` is ALWAYS excluded regardless."
176                        },
177                        "gitignore": {
178                            "type": "boolean",
179                            "description": "Respect .gitignore rules. Default true; pass false to include ignored files explicitly."
180                        },
181                        "max_results": {
182                            "type": "integer",
183                            "minimum": 1,
184                            "maximum": 200,
185                            "description": "Maximum results to return. Defaults to 200; the ceiling is 200, so a larger value only clamps down to it. Use it to shrink, never to request more."
186                        },
187                        "format": {
188                            "type": "string",
189                            "enum": ["flat", "grouped", "tree"],
190                            "description": "Output layout for the `formatted` field (default \"flat\")"
191                        },
192                        "sort_by_mtime": {
193                            "type": "boolean",
194                            "description": "Sort results by modification time, most recent first. Default true."
195                        },
196                        "timeout_ms": {
197                            "type": "integer",
198                            "minimum": 1,
199                            "description": "Abort the search after this many milliseconds; partial results are returned with `timed_out: true`."
200                        }
201                    },
202                    "required": ["pattern"]
203                }
204            }),
205            description_grep: serde_json::json!({
206                "name": "find_grep",
207                "description": concat!(
208                    "Search file content for lines matching a regex pattern. ",
209                    "Multi-file searches surface at most 20 distinct files per call ",
210                    "(per-file match cap 20); a single-file scope surfaces up to 200 ",
211                    "matches. When more files matched, `file_limit_reached` is true and ",
212                    "`note` suggests paginating with `skip` (call again with the same ",
213                    "pattern/path plus skip=<N> for the next page). Lines longer than ",
214                    "200 characters are truncated with a `...` suffix and flagged via ",
215                    "`truncated` — read the file for the full line. Patterns containing ",
216                    "a newline (or the `\\n` escape) automatically enable multiline ",
217                    "matching. Zero selected matches are marked `useless` with a `note` ",
218                    "— adjust the pattern or scope instead of blindly retrying. A ",
219                    "timeout returns the partial matches with `timed_out: true` — an ",
220                    "empty timed-out result is an INCOMPLETE scan, NOT proof of ",
221                    "absence; narrow the scope instead of retrying blindly. Search ",
222                    "several targets in one call with `paths` (each validated ",
223                    "individually; when absent, `path` is used). Restrict to a 1-based ",
224                    "inclusive line range with `line_range` (\"start-end\", requires ",
225                    "single-file targets). Each shown file carries a hashline anchor in ",
226                    "the `files` array (path, file_hash, header like \u{00b6}path#TAG). ",
227                    "To edit a matched file directly, pass the header's absolute path ",
228                    "and the tag (file_hash) to fs_edit — no re-read is needed to ",
229                    "obtain the current hash. Files beyond the anchor window have no ",
230                    "entry."
231                ),
232                "inputSchema": {
233                    "type": "object",
234                    "properties": {
235                        "pattern": {
236                            "type": "string",
237                            "description": "The regex pattern to search for in file contents"
238                        },
239                        "path": {
240                            "type": "string",
241                            "description": "The root directory to search within"
242                        },
243                        "paths": {
244                            "type": "array",
245                            "items": { "type": "string" },
246                            "description": "Multiple search targets (files or directories) in one call; overrides `path`. Each target is validated individually."
247                        },
248                        "line_range": {
249                            "type": "string",
250                            "description": "1-based inclusive line range \"start-end\" (requires single-file targets)"
251                        },
252                        "skip": {
253                            "type": "number",
254                            "description": "Files to skip before collecting results — paginate when the prior call hit the file window limit"
255                        }
256                    },
257                    "required": ["pattern"]
258                }
259            }),
260        }
261    }
262
263    // Glob-specific
264
265    /// Restrict glob results to a filesystem kind (`"file"`, `"dir"`, `"symlink"`).
266    #[must_use]
267    pub fn file_type(mut self, ft: impl Into<String>) -> Self {
268        self.file_type = Some(ft.into());
269        self
270    }
271
272    /// Search subdirectories recursively (default: `true`).
273    #[must_use]
274    pub const fn recursive(mut self, v: bool) -> Self {
275        self.recursive = Some(v);
276        self
277    }
278
279    /// Maximum number of glob entries to return.
280    #[must_use]
281    pub const fn max_results(mut self, n: u32) -> Self {
282        self.max_results = Some(n);
283        self
284    }
285
286    /// Sort glob results by modification time, most recent first.
287    #[must_use]
288    pub const fn sort_by_mtime(mut self, v: bool) -> Self {
289        self.sort_by_mtime = Some(v);
290        self
291    }
292
293    /// Output layout for glob results (`"flat"`, `"grouped"`, `"tree"`).
294    #[must_use]
295    pub fn glob_format(mut self, f: impl Into<String>) -> Self {
296        self.format = Some(f.into());
297        self
298    }
299
300    // Grep-specific
301
302    /// Restrict grep to files whose names match this glob (e.g. `"*.rs"`).
303    #[must_use]
304    pub fn name_glob(mut self, g: impl Into<String>) -> Self {
305        self.name_glob = Some(g.into());
306        self
307    }
308
309    /// Restrict grep to files of a given language (e.g. `"rust"`, `"py"`).
310    #[must_use]
311    pub fn language(mut self, lang: impl Into<String>) -> Self {
312        self.language = Some(lang.into());
313        self
314    }
315
316    /// Case-insensitive matching.
317    #[must_use]
318    pub const fn ignore_case(mut self, v: bool) -> Self {
319        self.ignore_case = Some(v);
320        self
321    }
322
323    /// Maximum total number of grep matches across all files.
324    #[must_use]
325    pub const fn max_count(mut self, n: u32) -> Self {
326        self.max_count = Some(n);
327        self
328    }
329
330    /// Lines of context to include before each match.
331    #[must_use]
332    pub const fn context_before(mut self, n: u32) -> Self {
333        self.context_before = Some(n);
334        self
335    }
336
337    /// Lines of context to include after each match.
338    #[must_use]
339    pub const fn context_after(mut self, n: u32) -> Self {
340        self.context_after = Some(n);
341        self
342    }
343
344    // Shared
345
346    /// Include hidden files / directories (names starting with `.`).
347    #[must_use]
348    pub const fn hidden(mut self, v: bool) -> Self {
349        self.hidden = Some(v);
350        self
351    }
352
353    /// Respect `.gitignore` rules.
354    #[must_use]
355    pub const fn gitignore(mut self, v: bool) -> Self {
356        self.gitignore = Some(v);
357        self
358    }
359
360    /// Abort the search after this many milliseconds.
361    #[must_use]
362    pub const fn timeout_ms(mut self, ms: u32) -> Self {
363        self.timeout_ms = Some(ms);
364        self
365    }
366
367    // Security guards
368
369    /// Set the project root directory (used for path-validation guards).
370    #[must_use]
371    pub fn cwd(mut self, path: impl Into<PathBuf>) -> Self {
372        self.guard = PathGuard::new(&path.into(), self.guard.allowlist(), self.guard.blocklist());
373        self
374    }
375
376    /// Set the explicit path allowlist.
377    #[must_use]
378    pub fn allowlist(mut self, paths: impl IntoIterator<Item = impl Into<PathBuf>>) -> Self {
379        let list: Vec<PathBuf> = paths.into_iter().map(Into::into).collect();
380        self.guard = PathGuard::new(self.guard.root(), Some(&list), self.guard.blocklist());
381        self
382    }
383
384    /// Set the explicit path blocklist.
385    #[must_use]
386    pub fn blocklist(mut self, paths: impl IntoIterator<Item = impl Into<PathBuf>>) -> Self {
387        let list: Vec<PathBuf> = paths.into_iter().map(Into::into).collect();
388        self.guard = PathGuard::new(self.guard.root(), self.guard.allowlist(), Some(&list));
389        self
390    }
391
392    /// Add a path to the allowlist (for session-level persistence).
393    pub fn add_allowlist_path(&mut self, path: PathBuf) {
394        self.guard.add_allowlist_path(path);
395    }
396
397    /// Remove a path from the allowlist (for AllowOnce cleanup).
398    ///
399    /// If the path is not in the allowlist, this is a no-op.
400    pub fn remove_allowlist_path(&mut self, path: &Path) {
401        self.guard.remove_allowlist_path(path);
402    }
403
404    /// Get the project root path.
405    #[must_use]
406    pub const fn find_root(&self) -> &PathBuf {
407        self.guard.root()
408    }
409
410    /// Get the allowlist (read-only reference).
411    #[must_use]
412    pub fn allowlist_ref(&self) -> Option<&[PathBuf]> {
413        self.guard.allowlist()
414    }
415
416    /// Get the blocklist (read-only reference).
417    #[must_use]
418    pub fn blocklist_ref(&self) -> Option<&[PathBuf]> {
419        self.guard.blocklist()
420    }
421
422    // Operations
423
424    /// Find filesystem entries matching a glob pattern.
425    ///
426    /// See [`glob`] for details.
427    ///
428    /// # Errors
429    ///
430    /// Returns an error when the search path does not exist, the pattern
431    /// is invalid, or the operation times out.
432    pub fn glob(&self, pattern: &str, path: &str) -> Result<GlobOutput, String> {
433        self.glob_with(pattern, Some(path.to_string()), None, None)
434    }
435
436    /// Like [`glob`](Self::glob), but searches one or more targets in a single
437    /// call (`paths` overrides `path`; each target is guard-resolved
438    /// individually), optionally streaming matches live via `on_match`.
439    ///
440    /// Each entry in `path`/`paths` may be a directory, a literal file, or a
441    /// glob (with or without a directory prefix); the effective pattern and
442    /// recursion are derived from its shape. Output paths are rebased to the
443    /// CWD when they live under it (feature: relative-to-CWD resolution).
444    ///
445    /// # Errors
446    ///
447    /// Returns an error when the search paths are missing, a pattern is
448    /// invalid, or the operation times out.
449    pub fn glob_with(
450        &self,
451        pattern: &str,
452        path: Option<String>,
453        paths: Option<Vec<String>>,
454        on_match: Option<Arc<GlobMatchCallback>>,
455    ) -> Result<GlobOutput, String> {
456        self.glob_full(pattern, path, paths, GlobCallOptions::default(), on_match)
457    }
458
459    /// Like [`glob_with`](Self::glob_with), with the schema-driven call options
460    /// bundled in [`GlobCallOptions`]: optional `file_type` (`"file"`,
461    /// `"dir"`, `"symlink"`), `hidden` and `gitignore` toggles, `format`
462    /// (`"flat"`, `"grouped"`, `"tree"`), a `sort_by_mtime` toggle, an optional
463    /// `timeout_ms`, and an optional `max_results`.
464    ///
465    /// The effective result cap defaults to 200 (mirroring the reference
466    /// tool) when neither this call's `max_results` nor the builder's is
467    /// set, and the ceiling is fixed: a `max_results` above 200 is clamped
468    /// down, so the caller can only lower the result set. A `max_results`
469    /// of `0` is rejected.
470    ///
471    /// `hidden` and `gitignore` default to `true` unless explicitly disabled
472    /// (reference-tool behavior). `sort_by_mtime` defaults to `true` —
473    /// the most recently modified files surface first. `file_type` filters
474    /// results to one filesystem kind and is an extension over the
475    /// reference tool.
476    ///
477    /// # Errors
478    ///
479    /// Returns an error when a search path cannot be resolved, a pattern is
480    /// invalid, `file_type` is unknown, `max_results` is zero, or the
481    /// operation times out.
482    pub fn glob_full(
483        &self,
484        pattern: &str,
485        path: Option<String>,
486        paths: Option<Vec<String>>,
487        opts: GlobCallOptions,
488        on_match: Option<Arc<GlobMatchCallback>>,
489    ) -> Result<GlobOutput, String> {
490        let GlobCallOptions {
491            file_type,
492            hidden,
493            gitignore,
494            max_results,
495            format,
496            sort_by_mtime,
497            timeout_ms,
498        } = opts;
499        if let Some(0) = max_results {
500            return Err("max_results must be a positive number".to_string());
501        }
502        let max_results = max_results
503            .or(self.max_results)
504            .unwrap_or(glob::DEFAULT_GLOB_LIMIT)
505            .min(glob::MAX_GLOB_LIMIT);
506        // Model-facing defaults mirror the reference tool (oh-my-pi): hidden
507        // and gitignore are ON unless explicitly disabled. `file_type` is an
508        // optional extension. Each flag falls back to the builder-configured
509        // value first.
510        let file_type = file_type.or_else(|| self.file_type.clone());
511        let hidden = hidden.or(self.hidden).unwrap_or(true);
512        let gitignore = gitignore.or(self.gitignore).unwrap_or(true);
513        let raw_targets: Vec<String> = match paths {
514            Some(list) if !list.is_empty() => list,
515            // No explicit root: default to the workspace root (CWD) so a bare
516            // pattern never needs a `path`.
517            _ => vec![path.unwrap_or_else(|| ".".to_string())],
518        };
519        // Each raw entry becomes one resolved spec. Guard resolution keeps the
520        // existing security contract; the parse step decides directory/glob/
521        // file semantics and effective recursion.
522        let resolved: Vec<GlobTargetSpec> = {
523            let mut resolved: Vec<GlobTargetSpec> = Vec::with_capacity(raw_targets.len());
524            for target in &raw_targets {
525                let parsed = parse_find_pattern(target);
526                let base = self.guard.resolve(&parsed.base_path.to_string_lossy())?;
527                resolved.push(GlobTargetSpec {
528                    base_path: base,
529                    pattern: if parsed.has_glob {
530                        parsed.glob_pattern
531                    } else {
532                        pattern.to_string()
533                    },
534                    has_glob: parsed.has_glob,
535                });
536            }
537            resolved
538        };
539        let cwd = self.guard.root().clone();
540        glob_targets_with(
541            &Glob {
542                file_type,
543                recursive: self.recursive,
544                max_results: Some(max_results),
545                sort_by_mtime: sort_by_mtime.or(self.sort_by_mtime),
546                hidden: Some(hidden),
547                gitignore: Some(gitignore),
548                timeout_ms: timeout_ms.or(self.timeout_ms),
549                format: format.or_else(|| self.format.clone()),
550            },
551            &resolved,
552            on_match,
553            Some(&cwd),
554        )
555    }
556
557    /// Search file content for lines matching a regex pattern.
558    ///
559    /// See [`grep`] for details.
560    ///
561    /// # Errors
562    ///
563    /// Returns an error when the path cannot be resolved, the pattern is
564    /// an invalid regex, or the operation times out.
565    pub fn grep(&self, pattern: &str, path: &str) -> Result<GrepOutput, String> {
566        self.grep_with(pattern, Some(path.to_string()), None, None, None)
567    }
568
569    /// Like [`grep`](Self::grep), but pages past the first file window with
570    /// `skip` (files to skip before collecting results). Ignored for
571    /// single-file scopes.
572    pub fn grep_skipping(
573        &self,
574        pattern: &str,
575        path: &str,
576        skip: Option<u32>,
577    ) -> Result<GrepOutput, String> {
578        self.grep_with(pattern, Some(path.to_string()), None, skip, None)
579    }
580
581    /// Search one or more targets in a single call, with optional pagination
582    /// (`skip`) and a 1-based inclusive `line_range` (`"start-end"`).
583    ///
584    /// `paths` overrides `path` when present; each target is resolved and
585    /// validated individually by the path guard, so the approval shown to the
586    /// user is exactly the set of targets the search opens. `line_range`
587    /// requires every target to be a single file.
588    ///
589    /// # Errors
590    ///
591    /// Returns an error when a target cannot be resolved, no target is
592    /// provided, the pattern is an invalid regex, or the operation times out.
593    pub fn grep_with(
594        &self,
595        pattern: &str,
596        path: Option<String>,
597        paths: Option<Vec<String>>,
598        skip: Option<u32>,
599        line_range: Option<String>,
600    ) -> Result<GrepOutput, String> {
601        self.grep_with_streaming(pattern, path, paths, skip, line_range, None)
602    }
603
604    /// Like [`grep_with`](Self::grep_with), optionally streaming each match
605    /// live via `on_match` (formatted `path:line`, rebased for multi-target
606    /// calls).
607    pub fn grep_with_streaming(
608        &self,
609        pattern: &str,
610        path: Option<String>,
611        paths: Option<Vec<String>>,
612        skip: Option<u32>,
613        line_range: Option<String>,
614        on_match: Option<Arc<GrepMatchCallback>>,
615    ) -> Result<GrepOutput, String> {
616        let raw_targets: Vec<String> = match paths {
617            Some(list) if !list.is_empty() => list,
618            _ => vec![path.ok_or_else(|| "missing 'path' or 'paths'".to_string())?],
619        };
620        let mut resolved: Vec<String> = Vec::with_capacity(raw_targets.len());
621        for target in &raw_targets {
622            let validated = self.guard.resolve(target)?;
623            resolved.push(validated.to_string_lossy().to_string());
624        }
625        grep_targets_with(
626            &GrepConfig {
627                glob: self.name_glob.clone(),
628                file_type: self.language.clone(),
629                ignore_case: self.ignore_case,
630                max_count: self.max_count,
631                skip,
632                line_range,
633                context_before: self.context_before,
634                context_after: self.context_after,
635                hidden: self.hidden,
636                gitignore: self.gitignore,
637                timeout_ms: self.timeout_ms,
638            },
639            pattern,
640            &resolved,
641            on_match,
642        )
643    }
644}