Skip to main content

vtcode_core/tools/summarizers/
search.rs

1//! Search result summarization
2//!
3//! Summarizes grep_file and list_files outputs from full match listings
4//! into concise summaries suitable for LLM context.
5//!
6//! ## Strategy
7//!
8//! Instead of sending all 127 matches across 2,500 tokens, send:
9//! "Found 127 matches in 15 files. Key files: src/tools/grep.rs (3 matches),
10//! src/tools/list.rs (1 match). Pattern in: execute_grep(), grep_impl() functions"
11//!
12//! Target: ~50 tokens vs 2,500 tokens = 98% savings
13
14use super::{Summarizer, truncate_to_tokens};
15use anyhow::Result;
16use hashbrown::HashMap;
17
18/// Summarizer for grep_file results
19pub struct GrepSummarizer {
20    /// Maximum number of files to list in summary
21    pub max_files: usize,
22    /// Maximum number of functions/symbols to mention
23    pub max_symbols: usize,
24    /// Maximum tokens for entire summary
25    pub max_tokens: usize,
26}
27
28impl Default for GrepSummarizer {
29    fn default() -> Self {
30        Self { max_files: 5, max_symbols: 5, max_tokens: 100 }
31    }
32}
33
34impl Summarizer for GrepSummarizer {
35    fn summarize(&self, full_output: &str, _metadata: Option<&serde_json::Value>) -> Result<String> {
36        // Parse grep output to extract key information
37        let stats = parse_grep_output(full_output);
38
39        // Build concise summary
40        let mut summary = format!("Found {} matches in {} files", stats.total_matches, stats.unique_files);
41
42        // Add top files if available
43        if !stats.top_files.is_empty() {
44            let file_list: Vec<String> = stats
45                .top_files
46                .iter()
47                .take(self.max_files)
48                .map(|(file, count)| format!("{file} ({count})"))
49                .collect();
50            summary.push_str(&format!(". Key files: {}", file_list.join(", ")));
51        }
52
53        // Add pattern context if available
54        if !stats.symbols.is_empty() {
55            let symbol_list: Vec<&str> = stats.symbols.iter().take(self.max_symbols).map(|s| s.as_str()).collect();
56            summary.push_str(&format!(". Pattern in: {}", symbol_list.join(", ")));
57        }
58
59        // Truncate to token limit
60        Ok(truncate_to_tokens(&summary, self.max_tokens))
61    }
62}
63
64/// Summarizer for list_files results
65pub struct ListSummarizer {
66    pub max_dirs: usize,
67    pub max_files: usize,
68    pub max_tokens: usize,
69}
70
71impl Default for ListSummarizer {
72    fn default() -> Self {
73        Self { max_dirs: 3, max_files: 10, max_tokens: 80 }
74    }
75}
76
77impl Summarizer for ListSummarizer {
78    fn summarize(&self, full_output: &str, _metadata: Option<&serde_json::Value>) -> Result<String> {
79        let stats = parse_list_output(full_output);
80
81        let mut summary =
82            format!("Listed {} items ({} files, {} directories)", stats.total_items, stats.file_count, stats.dir_count);
83
84        // Add sample files if available
85        if !stats.sample_files.is_empty() {
86            let files: Vec<&str> = stats.sample_files.iter().take(self.max_files).map(|s| s.as_str()).collect();
87            summary.push_str(&format!(". Files: {}", files.join(", ")));
88        }
89
90        Ok(truncate_to_tokens(&summary, self.max_tokens))
91    }
92}
93
94/// Statistics extracted from grep output
95#[derive(Debug, Default)]
96struct GrepStats {
97    total_matches: usize,
98    unique_files: usize,
99    top_files: Vec<(String, usize)>, // (filename, match_count)
100    symbols: Vec<String>,            // function names, identifiers
101}
102
103/// Statistics extracted from list output
104#[derive(Debug, Default)]
105struct ListStats {
106    total_items: usize,
107    file_count: usize,
108    dir_count: usize,
109    sample_files: Vec<String>,
110}
111
112/// Parse grep output to extract statistics
113fn parse_grep_output(output: &str) -> GrepStats {
114    let mut stats = GrepStats::default();
115    let mut file_matches: HashMap<String, usize> = HashMap::new();
116    let mut symbols_set: hashbrown::HashSet<String> = hashbrown::HashSet::new();
117
118    for line in output.lines() {
119        stats.total_matches += 1;
120
121        // Extract filename (format: "path/file.rs:42:content")
122        if let Some(colon_pos) = line.find(':') {
123            let file = &line[..colon_pos];
124            if !file.is_empty() {
125                *file_matches.entry(file.to_string()).or_insert(0) += 1;
126
127                // Extract simple filename for display
128                if let Some(slash_pos) = file.rfind('/') {
129                    let filename = &file[slash_pos + 1..];
130                    if filename.len() < 30 {
131                        // reasonable filename length
132                        *file_matches.entry(filename.to_string()).or_insert(0) += 1;
133                    }
134                }
135            }
136
137            // Extract potential symbols (functions, methods)
138            // Look for patterns like "fn name(", "impl Name", "pub struct"
139            let content = &line[colon_pos..];
140            extract_symbols(content, &mut symbols_set);
141        }
142    }
143
144    stats.unique_files = file_matches.len();
145
146    // Sort files by match count (descending)
147    let mut sorted_files: Vec<(String, usize)> = file_matches.into_iter().collect();
148    sorted_files.sort_by_key(|a| std::cmp::Reverse(a.1));
149    stats.top_files = sorted_files.into_iter().take(10).collect();
150
151    stats.symbols = symbols_set.into_iter().take(10).collect();
152
153    stats
154}
155
156/// Parse list output to extract statistics
157fn parse_list_output(output: &str) -> ListStats {
158    let mut stats = ListStats::default();
159
160    for line in output.lines() {
161        stats.total_items += 1;
162
163        // Detect directories (usually end with / or marked with [dir])
164        if line.ends_with('/') || line.contains("[dir]") || line.contains("DIR") {
165            stats.dir_count += 1;
166        } else {
167            stats.file_count += 1;
168            // Extract simple filename
169            if let Some(name) = line.split('/').next_back()
170                && !name.is_empty()
171                && name.len() < 50
172            {
173                stats.sample_files.push(name.to_string());
174            }
175        }
176    }
177
178    stats
179}
180
181/// Extract potential symbols (function names, types) from code line
182fn extract_symbols(line: &str, symbols: &mut hashbrown::HashSet<String>) {
183    // Look for function definitions: "fn name(" or "async fn name("
184    if let Some(fn_pos) = line.find("fn ") {
185        let after_fn = &line[fn_pos + 3..];
186        if let Some(paren_pos) = after_fn.find('(') {
187            let name = after_fn[..paren_pos].trim();
188            if !name.is_empty() && name.len() < 30 {
189                symbols.insert(format!("{name}()"));
190            }
191        }
192    }
193
194    // Look for struct/impl/trait definitions
195    for keyword in &["struct ", "impl ", "trait ", "enum "] {
196        if let Some(pos) = line.find(keyword) {
197            let after_kw = &line[pos + keyword.len()..];
198            if let Some(first_word) = after_kw.split_whitespace().next()
199                && first_word.len() < 30
200                && !first_word.contains('{')
201            {
202                symbols.insert(first_word.to_string());
203            }
204        }
205    }
206}
207
208#[cfg(test)]
209mod tests {
210    use super::super::estimate_tokens;
211    use super::*;
212
213    #[test]
214    fn test_grep_summarizer() {
215        let full_output = "\
216src/tools/grep.rs:45:    pub fn execute_grep(pattern: &str) -> Result<String> {
217src/tools/grep.rs:67:        let matches = grep_impl(pattern)?;
218src/tools/grep.rs:89:    fn grep_impl(pattern: &str) -> Result<Vec<Match>> {
219src/tools/list.rs:23:    // Uses grep internally for filtering
220src/main.rs:100:    grep.execute(\"test\")?;
221";
222
223        let summarizer = GrepSummarizer::default();
224        let summary = summarizer.summarize(full_output, None).unwrap();
225
226        assert!(summary.contains("Found 5 matches"));
227        assert!(summary.contains("files"));
228        assert!(estimate_tokens(&summary) < 100);
229
230        // Verify savings
231        let savings = summarizer.estimate_savings(full_output, &summary);
232        assert!(
233            savings.savings_percent > 20.0,
234            "Should save >20% (got {:.1}%, {} → {} tokens)",
235            savings.savings_percent,
236            savings.ui_tokens,
237            savings.llm_tokens
238        );
239        assert!(savings.llm_tokens < savings.ui_tokens);
240    }
241
242    #[test]
243    fn test_list_summarizer() {
244        let full_output = "\
245src/main.rs
246src/lib.rs
247src/tools/
248src/tools/grep.rs
249src/tools/list.rs
250tests/
251tests/integration.rs
252README.md
253";
254
255        let summarizer = ListSummarizer::default();
256        let summary = summarizer.summarize(full_output, None).unwrap();
257
258        assert!(summary.contains("Listed 8 items"));
259        assert!(summary.contains("files"));
260        assert!(summary.contains("directories"));
261        assert!(estimate_tokens(&summary) < 100);
262    }
263
264    #[test]
265    fn test_grep_stats_parsing() {
266        let output = "\
267src/tools/grep.rs:45:    pub fn execute_grep(pattern: &str) -> Result<String> {
268src/tools/grep.rs:67:        let matches = grep_impl(pattern)?;
269src/tools/list.rs:23:    // comment
270";
271
272        let stats = parse_grep_output(output);
273
274        assert_eq!(stats.total_matches, 3);
275        assert!(stats.unique_files > 0);
276        assert!(!stats.top_files.is_empty());
277    }
278
279    #[test]
280    fn test_symbol_extraction() {
281        let mut symbols = hashbrown::HashSet::new();
282
283        extract_symbols("    pub fn execute_grep(pattern: &str)", &mut symbols);
284        assert!(symbols.contains("execute_grep()"));
285
286        extract_symbols("impl GrepTool {", &mut symbols);
287        assert!(symbols.contains("GrepTool"));
288
289        extract_symbols("pub struct MyStruct {", &mut symbols);
290        assert!(symbols.contains("MyStruct"));
291    }
292
293    #[test]
294    fn test_list_stats_parsing() {
295        let output = "file1.rs\nfile2.rs\nsrc/\ntests/\nREADME.md";
296        let stats = parse_list_output(output);
297
298        assert_eq!(stats.total_items, 5);
299        assert_eq!(stats.dir_count, 2); // src/ and tests/
300        assert_eq!(stats.file_count, 3);
301    }
302
303    #[test]
304    fn test_large_grep_output() {
305        // Simulate large output with many matches
306        let mut output = String::new();
307        for i in 0..200 {
308            output.push_str(&format!("src/file{}.rs:{}:    match line\n", i % 20, i));
309        }
310
311        let summarizer = GrepSummarizer::default();
312        let summary = summarizer.summarize(&output, None).unwrap();
313
314        // Should be very concise despite 200 matches
315        assert!(estimate_tokens(&summary) < 150);
316        assert!(summary.contains("Found 200 matches"));
317
318        // Verify massive savings
319        let savings = summarizer.estimate_savings(&output, &summary);
320        assert!(savings.savings_percent > 95.0, "Should save >95% on large output");
321    }
322}