Skip to main content

vtcode_core/tools/
autonomous_executor.rs

1//! Autonomous tool execution with safety checks
2//!
3//! Implements safe autonomous execution following AGENTS.md principles:
4//! - Act, don't ask (for safe operations)
5//! - Verify before destructive operations
6//! - Loop detection and prevention
7//! - Context-aware decision making
8
9use crate::command_safety::shell_string_might_be_dangerous;
10use crate::config::constants::tools;
11use crate::core::loop_detector::LoopDetector;
12use crate::tools::apply_patch::decode_apply_patch_input;
13use crate::tools::command_args::{command_text, interactive_input_text};
14use crate::tools::tool_intent::{
15    self, classify_tool_intent, command_session_action, command_session_action_in, file_operation_action_in,
16    file_operation_action_is,
17};
18use anyhow::{Context, Result};
19use hashbrown::{HashMap, HashSet};
20use serde_json::Value;
21use std::collections::VecDeque;
22use std::path::{Path, PathBuf};
23use std::sync::{Arc, RwLock};
24use std::time::{Duration, Instant};
25use tracing::warn;
26
27/// Tools that require verification before execution
28const VERIFICATION_REQUIRED_TOOLS: &[&str] = &[
29    tools::WRITE_FILE,
30    tools::EDIT_FILE,
31    tools::UNIFIED_EXEC,
32    tools::CREATE_PTY_SESSION,
33];
34
35/// Autonomous execution policy for a tool
36#[derive(Debug, Clone, Copy, PartialEq, Eq)]
37pub enum AutonomousPolicy {
38    /// Execute immediately without asking
39    AutoExecute,
40    /// Show dry-run/preview, then execute
41    VerifyThenExecute,
42    /// Always require explicit user confirmation
43    RequireConfirmation,
44}
45
46/// Execution statistics for a tool
47#[derive(Debug, Clone, Default)]
48struct ToolStats {
49    total_attempts: usize,
50    successful_executions: usize,
51    failed_executions: usize,
52}
53
54impl ToolStats {
55    fn success_rate(&self) -> f64 {
56        if self.total_attempts == 0 {
57            0.0
58        } else {
59            self.successful_executions as f64 / self.total_attempts as f64
60        }
61    }
62}
63
64use crate::tools::circuit_breaker::CircuitBreaker;
65use crate::tools::validation::paths::validate_path_safety;
66use crate::utils::path::{normalize_path, resolve_workspace_path};
67
68/// Autonomous tool executor with safety checks.
69///
70/// In the unified interactive VT Code runloop, higher-level turn code owns
71/// user-visible loop recovery. The loop detector here remains a generic
72/// safeguard for legacy and non-unified autonomous execution paths.
73///
74/// # Rate Limiting
75///
76/// This executor maintains its own sliding-window rate limiter (`rate_history`)
77/// for **policy decisions** (auto-execute vs require confirmation). It does NOT
78/// consume tokens -- it only checks whether the tool has been called too
79/// frequently in the recent window.
80///
81/// The separate `PER_TOOL_RATE_LIMITER` (token-bucket) in `rate_limiter.rs`
82/// handles **execution blocking** at the executor level. The `SafetyGateway`
83/// rate limiter is disabled by the runloop in favor of these two systems.
84pub struct AutonomousExecutor {
85    verification_tools: HashSet<String>,
86    loop_detector: Arc<RwLock<LoopDetector>>,
87    execution_stats: Arc<RwLock<HashMap<String, ToolStats>>>,
88    workspace_dir: Option<PathBuf>,
89    rate_limit_window: Duration,
90    rate_limit_max_calls: usize,
91    rate_history: Arc<RwLock<HashMap<String, VecDeque<Instant>>>>,
92    circuit_breaker: CircuitBreaker,
93}
94
95impl AutonomousExecutor {
96    #[inline]
97    fn canonical_tool_key(tool_name: &str) -> &str {
98        tool_intent::canonical_command_session_tool_name(tool_name).unwrap_or(tool_name)
99    }
100
101    #[inline]
102    fn is_command_session_run(tool_name: &str, args: &Value) -> bool {
103        tool_intent::is_command_run_tool_call(tool_name, args)
104            || (tool_name == tools::UNIFIED_EXEC && command_session_action(args).is_none())
105    }
106
107    pub fn new() -> Self {
108        Self::with_loop_detector(Arc::new(RwLock::new(LoopDetector::new())))
109    }
110
111    pub fn with_loop_detector(loop_detector: Arc<RwLock<LoopDetector>>) -> Self {
112        Self {
113            verification_tools: VERIFICATION_REQUIRED_TOOLS.iter().map(|s| s.to_string()).collect(),
114            loop_detector,
115            execution_stats: Arc::new(RwLock::new(HashMap::new())),
116            workspace_dir: std::env::var("WORKSPACE_DIR")
117                .ok()
118                .map(PathBuf::from)
119                .or_else(|| std::env::current_dir().ok()),
120            rate_limit_window: Duration::from_secs(10),
121            rate_limit_max_calls: 5,
122            rate_history: Arc::new(RwLock::new(HashMap::new())),
123            circuit_breaker: CircuitBreaker::default(),
124        }
125    }
126
127    /// Set workspace directory for boundary validation
128    pub fn set_workspace_dir(&mut self, dir: PathBuf) {
129        self.workspace_dir = Some(dir);
130    }
131
132    /// Configure loop detection thresholds
133    pub fn configure_loop_limits(&self, limits: &HashMap<String, usize>) {
134        if let Ok(mut detector) = self.loop_detector.write() {
135            for (tool, limit) in limits {
136                detector.set_tool_limit(Self::canonical_tool_key(tool), *limit);
137            }
138        } else {
139            tracing::warn!("Failed to acquire loop detector lock for configuration");
140        }
141    }
142
143    pub fn set_loop_limit(&self, tool_name: &str, limit: usize) {
144        let tool_key = Self::canonical_tool_key(tool_name);
145        if let Ok(mut detector) = self.loop_detector.write() {
146            detector.set_tool_limit(tool_key, limit);
147        } else {
148            tracing::warn!("Failed to acquire loop detector lock for configuration");
149        }
150    }
151
152    pub fn is_hard_limit_exceeded(&self, tool_name: &str) -> bool {
153        let tool_key = Self::canonical_tool_key(tool_name);
154        self.loop_detector
155            .read()
156            .map(|detector| detector.is_hard_limit_exceeded(tool_key))
157            .unwrap_or(false)
158    }
159
160    /// Reset loop-detection streaks at the start of a new turn.
161    pub fn reset_turn_loop_detection(&self) {
162        if let Ok(mut detector) = self.loop_detector.write() {
163            detector.reset();
164        } else {
165            tracing::warn!("Failed to acquire loop detector lock for turn reset");
166        }
167    }
168
169    /// Determine execution policy for a tool
170    pub fn get_policy(&self, tool_name: &str, args: &Value) -> AutonomousPolicy {
171        if self.is_destructive_operation(tool_name, args) {
172            return AutonomousPolicy::RequireConfirmation;
173        }
174
175        if !classify_tool_intent(tool_name, args).mutating {
176            return AutonomousPolicy::AutoExecute;
177        }
178
179        if self.requires_preview(tool_name, args) {
180            return AutonomousPolicy::VerifyThenExecute;
181        }
182
183        AutonomousPolicy::RequireConfirmation
184    }
185
186    /// Check if tool should be blocked due to loop detection or circuit breaker
187    /// Returns Some(message) if blocked, None if allowed
188    pub fn should_block(&self, tool_name: &str, _args: &Value) -> Option<String> {
189        let tool_key = Self::canonical_tool_key(tool_name);
190
191        // Check circuit breaker first (fail fast)
192        if !self.circuit_breaker.allow_request_for_tool(tool_key) {
193            return Some(format!(
194                "Tool '{tool_key}' blocked by circuit breaker due to repeated failures. \
195                 Cooling down before retrying."
196            ));
197        }
198
199        if self.is_rate_limited(tool_key) {
200            return Some(format!(
201                "Tool '{}' temporarily blocked: rate limit exceeded ({} calls in {:?}).",
202                tool_key, self.rate_limit_max_calls, self.rate_limit_window
203            ));
204        }
205
206        // Use try_read to avoid blocking on contested locks
207        match self.loop_detector.try_read() {
208            Ok(detector) => {
209                // Check if hard limit already exceeded
210                if detector.is_hard_limit_exceeded(tool_key) {
211                    return Some(format!("Tool '{tool_key}' blocked: hard limit exceeded. Agent is stuck in a loop."));
212                }
213
214                // Check call count and provide early warning
215                let count = detector.get_call_count(tool_key);
216                if count >= 3
217                    && let Some(suggestion) = detector.suggest_alternative(tool_key)
218                {
219                    return Some(format!(
220                        "Tool '{tool_key}' called {count} times. Consider alternative approach:\n{suggestion}"
221                    ));
222                }
223            }
224            Err(_) => {
225                // If we can't get the lock, don't block execution
226                tracing::debug!("Could not acquire loop detector read lock for {}", tool_key);
227            }
228        }
229        None
230    }
231
232    /// Record tool call in loop detector
233    /// Returns warning message if loop detected
234    pub fn record_tool_call(&self, tool_name: &str, args: &Value) -> Option<String> {
235        let tool_key = Self::canonical_tool_key(tool_name);
236        self.record_rate_history(tool_key);
237        if let Ok(mut detector) = self.loop_detector.write() {
238            detector.record_call(tool_key, args)
239        } else {
240            None
241        }
242    }
243
244    /// Check if operation is destructive based on tool and arguments
245    fn is_destructive_operation(&self, tool_name: &str, args: &Value) -> bool {
246        match tool_name {
247            tools::APPLY_PATCH | tools::DELETE_FILE => true,
248            tools::UNIFIED_FILE => file_operation_action_in(args, &["delete"]),
249            _ if Self::is_command_session_run(tool_name, args) => command_text(args)
250                .ok()
251                .flatten()
252                .is_some_and(|cmd| self.is_destructive_command(&cmd)),
253            _ if tool_intent::is_command_session_tool(tool_name)
254                && command_session_action_in(args, &["write", "continue"]) =>
255            {
256                interactive_input_text(args).is_some_and(|input| self.is_destructive_command(input))
257            }
258            _ => false,
259        }
260    }
261
262    /// Check if shell command is destructive
263    fn is_destructive_command(&self, cmd: &str) -> bool {
264        if shell_string_might_be_dangerous(cmd) {
265            return true;
266        }
267
268        let cmd_lower = cmd.to_lowercase();
269
270        // Additional destructive patterns that are not captured by the centralized
271        // command safety evaluator.
272        let supplemental_patterns = [
273            "truncate",
274            "> /dev/",
275            "dd if=",
276            "mkfs",
277            "fdisk",
278            "format",
279            // Overwrite operations
280            ">/",
281            "2>/",
282            // Package managers (potentially destructive)
283            "npm uninstall -g",
284            "cargo uninstall",
285            "pip uninstall",
286            // Permissions
287            "chmod -r",
288            "chown -r",
289        ];
290
291        supplemental_patterns.iter().any(|pattern| cmd_lower.contains(pattern))
292    }
293
294    /// Validate tool arguments for safety
295    pub fn validate_args(&self, tool_name: &str, args: &Value) -> Result<()> {
296        match tool_name {
297            tools::WRITE_FILE | tools::EDIT_FILE => self.validate_file_path(args.get("path"))?,
298            _ if Self::is_command_session_run(tool_name, args) => {
299                self.validate_command_text(
300                    &command_text(args)
301                        .map_err(anyhow::Error::msg)?
302                        .context("Missing or invalid 'command' argument")?,
303                )?;
304            }
305            _ if tool_intent::is_command_session_tool(tool_name)
306                && command_session_action_in(args, &["write", "continue"]) =>
307            {
308                if let Some(input) = interactive_input_text(args) {
309                    self.validate_command_text(input)?;
310                }
311            }
312            tools::UNIFIED_FILE if file_operation_action_in(args, &["write", "edit", "delete"]) => {
313                self.validate_file_path(args.get("path"))?;
314            }
315            tools::UNIFIED_FILE if file_operation_action_in(args, &["move", "copy"]) => {
316                self.validate_file_path(args.get("path"))?;
317                self.validate_file_path(args.get("destination"))?;
318            }
319            _ => {}
320        }
321        Ok(())
322    }
323
324    /// Validate file path is within workspace boundaries.
325    ///
326    /// First checks for sensitive system paths via `validate_path_safety`,
327    /// then enforces workspace boundary constraints.
328    fn validate_file_path(&self, path: Option<&Value>) -> Result<()> {
329        let path_str = path.and_then(|v| v.as_str()).context("Missing or invalid 'path' argument")?;
330
331        // Check for sensitive system paths (e.g., /var/db/shadow, /etc/shadow)
332        validate_path_safety(path_str)?;
333
334        let path_obj = Path::new(path_str);
335
336        // Check for absolute paths
337        if path_obj.is_absolute() {
338            // Allow /tmp/vtcode paths
339            if path_str.starts_with("/tmp/vtcode") {
340                return Ok(());
341            }
342
343            // Check if within workspace
344            if let Some(workspace) = &self.workspace_dir
345                && (resolve_workspace_path(workspace, path_obj).is_ok()
346                    || is_within_workspace_lexically(workspace, path_obj))
347            {
348                return Ok(());
349            }
350
351            anyhow::bail!(
352                "Absolute path outside workspace boundary: {path_str}. \
353                 Only paths within WORKSPACE_DIR or /tmp/vtcode are allowed."
354            );
355        }
356
357        // Prevent parent directory traversal that could escape workspace
358        if path_str.contains("..") {
359            warn!("Path contains parent directory traversal: {}", path_str);
360
361            // Resolve the path and check if it stays within workspace
362            if let Some(workspace) = &self.workspace_dir {
363                let path_obj = Path::new(path_str);
364                let canonical_ok = resolve_workspace_path(workspace, &workspace.join(path_obj)).is_ok();
365                let lexical_ok = is_within_workspace_lexically(workspace, path_obj);
366                if !canonical_ok && !lexical_ok {
367                    anyhow::bail!("Path traversal escapes workspace boundary: {path_str}");
368                }
369            } else {
370                anyhow::bail!("Path traversal blocked: workspace boundary is unknown for '{path_str}'");
371            }
372        }
373
374        // If workspace directory is unknown, conservatively block writes to avoid escaping boundaries.
375        if self.workspace_dir.is_none() {
376            anyhow::bail!(
377                "Workspace directory is not set; refusing to write to relative path '{path_str}'. \
378                 Set WORKSPACE_DIR or call set_workspace_dir()."
379            );
380        }
381
382        Ok(())
383    }
384
385    /// Validate shell command for safety
386    fn validate_command_text(&self, cmd: &str) -> Result<()> {
387        if self.is_destructive_command(cmd) {
388            anyhow::bail!("Destructive command requires explicit confirmation: {cmd}");
389        }
390
391        Ok(())
392    }
393
394    /// Generate dry-run preview for verification
395    pub fn generate_preview(&self, tool_name: &str, args: &Value) -> String {
396        if tool_name == tools::WRITE_FILE
397            || (tool_name == tools::UNIFIED_FILE && file_operation_action_is(args, "write"))
398        {
399            let path = args.get("path").and_then(|v| v.as_str()).unwrap_or("unknown");
400            let content = args.get("content").and_then(|v| v.as_str()).unwrap_or("");
401            let lines = content.lines().count();
402            let size_kb = content.len() / 1024;
403
404            let preview = if lines > 10 {
405                let first_lines: Vec<_> = content.lines().take(5).collect();
406                format!("\n  {}\n  ... ({} more lines)", first_lines.join("\n  "), lines - 5)
407            } else {
408                format!("\n  {}", content.lines().collect::<Vec<_>>().join("\n  "))
409            };
410
411            format!("Will write {lines} lines ({size_kb} KB) to: {path}\nPreview:{preview}")
412        } else if tool_name == tools::EDIT_FILE
413            || (tool_name == tools::UNIFIED_FILE && file_operation_action_is(args, "edit"))
414        {
415            let path = args.get("path").and_then(|v| v.as_str()).unwrap_or("unknown");
416            let old_str = args.get("old_str").and_then(|v| v.as_str()).unwrap_or("");
417            let new_str = args.get("new_str").and_then(|v| v.as_str()).unwrap_or("");
418
419            format!(
420                "Will edit file: {}\nReplacing:\n  {}\nWith:\n  {}",
421                path,
422                old_str.lines().take(3).collect::<Vec<_>>().join("\n  "),
423                new_str.lines().take(3).collect::<Vec<_>>().join("\n  ")
424            )
425        } else if Self::is_command_session_run(tool_name, args) {
426            let cmd = command_text(args).ok().flatten().unwrap_or_else(|| "unknown".to_string());
427            let is_destructive = self.is_destructive_command(&cmd);
428
429            let warning = if is_destructive {
430                "\n[WARN] WARNING: This command is potentially destructive!"
431            } else {
432                ""
433            };
434
435            format!("Will execute: {cmd}{warning}")
436        } else if tool_name == tools::APPLY_PATCH {
437            let patch = decode_apply_patch_input(args)
438                .ok()
439                .flatten()
440                .map(|patch| patch.text)
441                .unwrap_or_default();
442            let lines = patch.lines().count();
443            format!("Will apply patch with {lines} lines of changes")
444        } else {
445            format!("Will execute: {tool_name} with args: {args:?}")
446        }
447    }
448
449    fn requires_preview(&self, tool_name: &str, args: &Value) -> bool {
450        let canonical_tool_name = Self::canonical_tool_key(tool_name);
451        if self.verification_tools.contains(canonical_tool_name) {
452            return true;
453        }
454
455        match canonical_tool_name {
456            tools::UNIFIED_FILE => file_operation_action_in(args, &["write", "edit", "move", "copy"]),
457            tools::UNIFIED_EXEC => command_session_action_in(args, &["run", "code", "close"]),
458            _ => false,
459        }
460    }
461
462    /// Record execution result for statistics tracking and circuit breaker
463    pub fn record_execution(&self, tool_name: &str, success: bool) {
464        let tool_key = Self::canonical_tool_key(tool_name);
465
466        // Update circuit breaker
467        if success {
468            self.circuit_breaker.record_success_for_tool(tool_key);
469        } else {
470            // Note: We blindly treat all failures as circuit-breaking for now.
471            // Ideally, the caller should specify if it's an arg error or system error.
472            self.circuit_breaker.record_failure_for_tool(tool_key, false);
473        }
474
475        if let Ok(mut stats) = self.execution_stats.write() {
476            let entry = stats.entry(tool_key.to_string()).or_default();
477            entry.total_attempts += 1;
478            if success {
479                entry.successful_executions += 1;
480            } else {
481                entry.failed_executions += 1;
482            }
483        }
484    }
485
486    /// Get success rate for a tool
487    pub fn get_success_rate(&self, tool_name: &str) -> f64 {
488        if let Ok(stats) = self.execution_stats.read() {
489            stats.get(tool_name).map(|s| s.success_rate()).unwrap_or(0.0)
490        } else {
491            0.0
492        }
493    }
494
495    /// Get execution statistics for a tool
496    pub fn get_tool_stats(&self, tool_name: &str) -> Option<(usize, usize, usize)> {
497        if let Ok(stats) = self.execution_stats.read() {
498            stats
499                .get(tool_name)
500                .map(|s| (s.total_attempts, s.successful_executions, s.failed_executions))
501        } else {
502            None
503        }
504    }
505}
506
507impl Default for AutonomousExecutor {
508    fn default() -> Self {
509        Self::new()
510    }
511}
512
513impl AutonomousExecutor {
514    fn record_rate_history(&self, tool_name: &str) {
515        let now = Instant::now();
516        if let Ok(mut history) = self.rate_history.write() {
517            let entries = history.entry(Self::canonical_tool_key(tool_name).to_string()).or_default();
518            entries.push_back(now);
519            prune_expired_timestamps(entries, now, self.rate_limit_window);
520        }
521    }
522
523    fn is_rate_limited(&self, tool_name: &str) -> bool {
524        let tool_key = Self::canonical_tool_key(tool_name);
525        let now = Instant::now();
526
527        // First, try with a read lock to check without modifying
528        // This is the common fast path when there are no expired entries
529        if let Ok(history) = self.rate_history.read() {
530            if let Some(entries) = history.get(tool_key) {
531                // Quick check: if all entries are within window and at limit, we're rate limited
532                let oldest_within_window = entries
533                    .front()
534                    .is_some_and(|front| now.duration_since(*front) <= self.rate_limit_window);
535                if oldest_within_window {
536                    return entries.len() >= self.rate_limit_max_calls;
537                }
538            } else {
539                // No entries for this tool, definitely not rate limited
540                return false;
541            }
542        }
543
544        // Fall back to write lock only when we need to clean up expired entries
545        if let Ok(mut history) = self.rate_history.write() {
546            let entries = history.entry(tool_key.to_string()).or_default();
547            prune_expired_timestamps(entries, now, self.rate_limit_window);
548            return entries.len() >= self.rate_limit_max_calls;
549        }
550        false
551    }
552}
553
554fn prune_expired_timestamps(entries: &mut VecDeque<Instant>, now: Instant, window: Duration) {
555    while let Some(front) = entries.front() {
556        if now.duration_since(*front) > window {
557            entries.pop_front();
558        } else {
559            break;
560        }
561    }
562}
563
564fn is_within_workspace_lexically(workspace: &Path, candidate: &Path) -> bool {
565    let normalized_workspace = normalize_path(workspace);
566    let normalized_candidate = if candidate.is_absolute() {
567        normalize_path(candidate)
568    } else {
569        normalize_path(&normalized_workspace.join(candidate))
570    };
571    normalized_candidate.starts_with(&normalized_workspace)
572}
573
574#[cfg(test)]
575mod tests {
576    use super::*;
577    use serde_json::json;
578
579    #[test]
580    fn test_readonly_tools_auto_execute() {
581        let executor = AutonomousExecutor::new();
582
583        assert_eq!(
584            executor.get_policy(tools::CODE_SEARCH, &json!({"query": "Widget", "path": "src"})),
585            AutonomousPolicy::AutoExecute
586        );
587        assert_eq!(
588            executor.get_policy(tools::UNIFIED_FILE, &json!({"action": "read", "path": "README.md"})),
589            AutonomousPolicy::AutoExecute
590        );
591        assert_eq!(
592            executor.get_policy(tools::UNIFIED_EXEC, &json!({"action": "poll", "session_id": "run-1"})),
593            AutonomousPolicy::AutoExecute
594        );
595        assert_eq!(
596            executor.get_policy(tools::UNIFIED_EXEC, &json!({"action": "continue", "session_id": "run-1"})),
597            AutonomousPolicy::AutoExecute
598        );
599    }
600
601    #[test]
602    fn test_destructive_commands_require_confirmation() {
603        let executor = AutonomousExecutor::new();
604
605        let destructive_cmds = vec![
606            "rm -rf /tmp/test",
607            "git reset --hard HEAD~1",
608            "git push --force origin main",
609            "git clean -fdx",
610            "chmod -R 777 /",
611        ];
612
613        for cmd in destructive_cmds {
614            let args = json!({"command": cmd});
615            let policy = executor.get_policy("shell", &args);
616            assert_eq!(policy, AutonomousPolicy::RequireConfirmation, "unexpected policy for command: {cmd}");
617        }
618    }
619
620    #[test]
621    fn test_verification_tools_need_preview() {
622        let executor = AutonomousExecutor::new();
623
624        for tool in VERIFICATION_REQUIRED_TOOLS {
625            let policy = executor.get_policy(tool, &json!({}));
626            assert_eq!(policy, AutonomousPolicy::VerifyThenExecute);
627        }
628    }
629
630    #[test]
631    fn test_unified_tools_use_action_specific_policies() {
632        let executor = AutonomousExecutor::new();
633
634        assert_eq!(
635            executor
636                .get_policy(tools::UNIFIED_FILE, &json!({"action": "write", "path": "foo.txt", "content": "hello"})),
637            AutonomousPolicy::VerifyThenExecute
638        );
639        assert_eq!(
640            executor.get_policy(
641                tools::UNIFIED_FILE,
642                &json!({"action": "patch", "input": "*** Begin Patch\n*** End Patch"})
643            ),
644            AutonomousPolicy::RequireConfirmation
645        );
646        assert_eq!(
647            executor.get_policy(tools::UNIFIED_EXEC, &json!({"cmd": "cargo build"})),
648            AutonomousPolicy::VerifyThenExecute
649        );
650        assert_eq!(executor.get_policy(tools::UNIFIED_EXEC, &json!({"cmd": "echo hi"})), AutonomousPolicy::AutoExecute);
651        assert_eq!(
652            executor.get_policy(
653                tools::UNIFIED_EXEC,
654                &json!({"action": "write", "session_id": "run-1", "input": "rm -rf /tmp/test"})
655            ),
656            AutonomousPolicy::RequireConfirmation
657        );
658    }
659
660    #[test]
661    fn test_exec_aliases_use_command_session_preview_policy() {
662        let executor = AutonomousExecutor::new();
663
664        assert_eq!(
665            executor.get_policy(tools::EXEC_COMMAND, &json!({"cmd": "cargo build"})),
666            AutonomousPolicy::VerifyThenExecute
667        );
668        assert_eq!(
669            executor.get_policy(tools::RUN_PTY_CMD, &json!({"command": "cargo build"})),
670            AutonomousPolicy::VerifyThenExecute
671        );
672    }
673
674    #[test]
675    fn test_loop_detection_integration() {
676        let executor = AutonomousExecutor::new();
677        let args = json!({"query": "Widget", "path": "src/"});
678
679        // First two calls should not block
680        assert!(executor.should_block(tools::CODE_SEARCH, &args).is_none());
681        executor.record_tool_call(tools::CODE_SEARCH, &args);
682
683        assert!(executor.should_block(tools::CODE_SEARCH, &args).is_none());
684        executor.record_tool_call(tools::CODE_SEARCH, &args);
685
686        // Third call should trigger warning
687        executor.record_tool_call(tools::CODE_SEARCH, &args);
688        let block_msg = executor.should_block(tools::CODE_SEARCH, &args);
689        assert!(block_msg.is_some());
690        let message = block_msg.unwrap();
691        assert!(
692            message.contains("alternative") || message.contains("blocked"),
693            "unexpected loop warning message: {message}"
694        );
695    }
696
697    #[test]
698    fn test_turn_reset_clears_loop_detection_state() {
699        let executor = AutonomousExecutor::new();
700        let args = json!({"query": "Widget", "path": "src/"});
701
702        executor.record_tool_call(tools::CODE_SEARCH, &args);
703        executor.record_tool_call(tools::CODE_SEARCH, &args);
704        executor.record_tool_call(tools::CODE_SEARCH, &args);
705        assert!(executor.should_block(tools::CODE_SEARCH, &args).is_some());
706
707        executor.reset_turn_loop_detection();
708        assert!(executor.should_block(tools::CODE_SEARCH, &args).is_none());
709    }
710
711    #[test]
712    fn test_execution_stats_tracking() {
713        let executor = AutonomousExecutor::new();
714
715        // Record some executions
716        executor.record_execution(tools::CODE_SEARCH, true);
717        executor.record_execution(tools::CODE_SEARCH, true);
718        executor.record_execution(tools::CODE_SEARCH, false);
719
720        // Check stats
721        let (total, success, failed) = executor.get_tool_stats(tools::CODE_SEARCH).unwrap();
722        assert_eq!(total, 3);
723        assert_eq!(success, 2);
724        assert_eq!(failed, 1);
725
726        // Check success rate
727        let rate = executor.get_success_rate(tools::CODE_SEARCH);
728        assert!((rate - 0.666).abs() < 0.01);
729    }
730
731    #[test]
732    fn test_workspace_boundary_validation() {
733        let mut executor = AutonomousExecutor::new();
734        let temp_dir = std::env::temp_dir();
735        executor.set_workspace_dir(temp_dir.clone());
736
737        // Absolute path outside workspace should fail
738        let args = json!({"path": "/etc/passwd"});
739        let result = executor.validate_args(tools::WRITE_FILE, &args);
740        assert!(result.is_err());
741        let err = result.unwrap_err().to_string();
742        assert!(
743            err.contains("workspace boundary") || err.contains("Path traversal") || err.contains("system directory"),
744            "{err}"
745        );
746
747        // /tmp/vtcode should be allowed
748        let args = json!({"path": "/tmp/vtcode/test.txt"});
749        let result = executor.validate_args(tools::WRITE_FILE, &args);
750        result.unwrap();
751    }
752
753    #[test]
754    fn test_command_session_validation_uses_command_aliases() {
755        let executor = AutonomousExecutor::new();
756
757        let err = executor
758            .validate_args(tools::UNIFIED_EXEC, &json!({"cmd": "rm -rf /tmp/test"}))
759            .expect_err("destructive command should fail");
760
761        assert!(err.to_string().contains("requires explicit confirmation"));
762    }
763
764    #[test]
765    fn test_file_operation_validation_checks_destinations() {
766        let mut executor = AutonomousExecutor::new();
767        executor.set_workspace_dir(PathBuf::from("/workspace"));
768
769        let err = executor
770            .validate_args(
771                tools::UNIFIED_FILE,
772                &json!({
773                    "action": "move",
774                    "path": "src/main.rs",
775                    "destination": "/etc/passwd"
776                }),
777            )
778            .expect_err("destination outside workspace should fail");
779
780        let err = err.to_string();
781        assert!(
782            err.contains("workspace boundary") || err.contains("Path traversal") || err.contains("system directory"),
783            "{err}"
784        );
785    }
786
787    #[test]
788    fn test_enhanced_destructive_patterns() {
789        let executor = AutonomousExecutor::new();
790
791        let destructive_cmds = vec![
792            "rm -r somedir",
793            "git branch -D feature",
794            "npm uninstall -g package",
795            "cargo uninstall tool",
796        ];
797
798        for cmd in destructive_cmds {
799            assert!(executor.is_destructive_command(cmd));
800        }
801    }
802
803    #[test]
804    fn test_enhanced_preview_generation() {
805        let executor = AutonomousExecutor::new();
806
807        // Test write_file preview
808        let args = json!({
809            "path": "test.rs",
810            "content": "line1\nline2\nline3"
811        });
812        let preview = executor.generate_preview(tools::WRITE_FILE, &args);
813        assert!(preview.contains("3 lines"));
814        assert!(preview.contains("test.rs"));
815
816        // Test edit_file preview
817        let args = json!({
818            "path": "main.rs",
819            "old_str": "old code",
820            "new_str": "new code"
821        });
822        let preview = executor.generate_preview(tools::EDIT_FILE, &args);
823        assert!(preview.contains("main.rs"));
824        assert!(preview.contains("old code"));
825        assert!(preview.contains("new code"));
826
827        // Test destructive command preview
828        let args = json!({"command": "rm -rf /tmp/test"});
829        let preview = executor.generate_preview("shell", &args);
830        assert!(preview.contains("WARNING"));
831        assert!(preview.contains("destructive"));
832
833        let preview = executor.generate_preview(tools::EXEC_COMMAND, &json!({"cmd": "git status"}));
834        assert!(preview.contains("git status"));
835
836        let preview = executor.generate_preview(
837            tools::APPLY_PATCH,
838            &json!({
839                "input": "*** Begin Patch\n*** Add File: note.txt\n+hello\n*** End Patch"
840            }),
841        );
842        assert!(preview.contains("apply patch"));
843    }
844
845    #[test]
846    fn test_parent_traversal_detection() {
847        let mut executor = AutonomousExecutor::new();
848        let workspace = PathBuf::from("/workspace");
849        executor.set_workspace_dir(workspace);
850
851        // Parent traversal is rejected before canonical workspace checks.
852        let args = json!({"path": "src/../lib/file.rs"});
853        let result = executor.validate_args(tools::WRITE_FILE, &args);
854        let err = result.expect_err("parent traversal should fail").to_string();
855        assert!(err.contains("Path traversal"), "{err}");
856    }
857}