Skip to main content

vtcode_core/exec/
integration_tests.rs

1//! Integration tests for MCP code execution architecture
2//!
3//! Tests all 5 steps from Anthropic's code execution recommendations:
4//! 1. Progressive tool discovery
5//! 2. Code executor with SDK generation
6//! 3. Skill persistence
7//! 4. Data filtering in code
8//! 5. PII tokenization
9
10#[cfg(test)]
11mod tests {
12    use crate::exec::{AgentBehaviorAnalyzer, ExecutionConfig, PiiTokenizer, Skill, SkillManager, SkillMetadata};
13    use anyhow::Result;
14    use chrono;
15    use tempfile;
16
17    // ============================================================================
18    // Test 1: Discovery → Execution → Filtering
19    // ============================================================================
20
21    #[test]
22    fn test_discovery_to_execution_flow() {
23        // This test validates that tool discovery results can feed into code execution
24        // In real usage: agents discover tools, then use them in written code
25
26        // Note: This test demonstrates the concept but requires proper setup with
27        // actual MCP client. See integration tests documentation
28        // for full example with mocked dependencies.
29
30        // Step 1: Create execution config
31        let config = ExecutionConfig { timeout_secs: 5, ..Default::default() };
32
33        // Verify config is created properly
34        assert_eq!(config.timeout_secs, 5);
35
36        // Step 2: Agent writes code that filters data locally
37        let _code = r#"
38# Simulate filtering without returning all results to model
39data = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10]
40filtered = [x for x in data if x > 5]
41result = {"count": len(filtered), "items": filtered}
42"#;
43
44        // Step 3: In real usage, agent writes code that filters data locally
45        // (actual code runs locally, only aggregated result returns to model)
46
47        // Step 4: Pattern demonstration
48        // The pattern is: write code that processes data locally,
49        // returning only filtered/aggregated results to the model
50        let expected_pattern = r#"
51# Agent writes code that processes locally
52data = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10]
53filtered = [x for x in data if x > 5]
54result = {"count": len(filtered), "items": filtered}
55        "#;
56        assert!(expected_pattern.contains("result = {"));
57
58        // This demonstrates the key benefit: filtering happens in code
59        // instead of in prompt context, saving ~98% of tokens
60    }
61
62    // ============================================================================
63    // Test 2: Execution → Skill Persistence → Reuse
64    // ============================================================================
65
66    #[tokio::test]
67    async fn test_execution_to_skill_reuse() {
68        // This test demonstrates the skill save/load/reuse pattern
69        // from Anthropic's code execution architecture
70
71        // Create temporary directory for skills
72        let temp_dir = tempfile::TempDir::new().unwrap();
73
74        // Step 1: Execution config for testing
75        let config = ExecutionConfig { timeout_secs: 5, ..Default::default() };
76
77        // Verify config is valid
78        assert_eq!(config.timeout_secs, 5);
79
80        let code = r#"
81def double_value(x):
82    return x * 2
83
84result = {"test": double_value(21)}
85"#;
86
87        // Step 2: Create skill manager
88        let skill_manager = SkillManager::new(temp_dir.path());
89
90        // Step 3: Save the code as a reusable skill for later use
91        let skill = Skill {
92            metadata: SkillMetadata {
93                name: "double_value".to_owned(),
94                description: "Double a number".to_owned(),
95                language: "python3".to_owned(),
96                inputs: vec![],
97                output: "integer".to_owned(),
98                examples: vec![],
99                tags: vec!["math".to_owned()],
100                created_at: chrono::Utc::now().to_rfc3339(),
101                modified_at: chrono::Utc::now().to_rfc3339(),
102                tool_dependencies: vec![],
103            },
104            code: code.to_owned(),
105        };
106
107        skill_manager.save_skill(skill).await.unwrap();
108
109        // Step 5: Load and reuse skill
110        let loaded_skill = skill_manager.load_skill("double_value").await.unwrap();
111        assert_eq!(loaded_skill.metadata.name, "double_value");
112        assert_eq!(loaded_skill.metadata.language, "python3");
113
114        // This pattern allows agents to reuse code across conversations,
115        // saving 80%+ on token usage for repeated patterns
116        // temp_dir will be automatically cleaned up when dropped
117    }
118
119    // ============================================================================
120    // Test 3: PII Protection in Pipeline
121    // ============================================================================
122
123    #[test]
124    fn test_pii_protection_in_execution() -> Result<()> {
125        // Create a PII tokenizer
126        let tokenizer = PiiTokenizer::new()?;
127
128        // Step 1: Detect PII patterns
129        let text_with_pii = "Email: john@example.com, SSN: 123-45-6789";
130
131        let detected = tokenizer.detect_pii(text_with_pii)?;
132        assert!(!detected.is_empty());
133
134        // Step 2: Verify we can tokenize
135        let (tokenized, _tokens) = tokenizer.tokenize_string(text_with_pii)?;
136
137        // Step 3: Verify tokenized version doesn't contain plaintext PII
138        assert!(!tokenized.contains("john@example.com"));
139        assert!(!tokenized.contains("123-45-6789"));
140        assert!(tokenized.contains("__PII_"));
141
142        // Step 4: Verify we can detokenize
143        let detokenized = tokenizer.detokenize_string(&tokenized)?;
144        assert!(detokenized.contains("john@example.com"));
145        assert!(detokenized.contains("123-45-6789"));
146        Ok(())
147    }
148
149    // ============================================================================
150    // Test 4: Large Dataset Filtering
151    // ============================================================================
152
153    #[test]
154    fn test_large_dataset_filtering_efficiency() {
155        // Demonstrates data filtering efficiency pattern
156        // Instead of returning all 1000 items to the model,
157        // the code processes locally and returns only aggregated results
158
159        let config = ExecutionConfig { timeout_secs: 5, ..Default::default() };
160
161        // In real usage with actual executor setup:
162        // let executor = CodeExecutor::new(language, client, workspace);
163
164        // Example code pattern for large dataset filtering
165        let code_pattern = r#"
166# Simulate processing large dataset
167items = list(range(1000))
168
169# Filter in code (not returned to model) - saves 98% of tokens!
170filtered_items = [x for x in items if x % 10 == 0]
171stats = {
172    "total": len(items),
173    "filtered": len(filtered_items),
174    "sample": filtered_items[:5]  # Return only sample, not all items
175}
176
177result = stats
178"#;
179
180        // Verify config is valid
181        assert_eq!(config.timeout_secs, 5);
182        assert_eq!(config.max_output_bytes, 10 * 1024 * 1024);
183        assert!(code_pattern.contains("# Filter in code"));
184
185        // Token efficiency: with traditional approach:
186        // - 1000 items × ~100 tokens each = ~100k tokens
187        // With code execution approach:
188        // - Code ~500 tokens + result ~100 tokens = ~600 tokens
189        // Savings: 98% fewer tokens!
190    }
191
192    // ============================================================================
193    // Test 5: Tool Error Handling in Code
194    // ============================================================================
195
196    #[test]
197    fn test_tool_error_handling_in_code() {
198        // Demonstrates error handling pattern in code execution
199        // Agents can write code with try/except blocks to handle errors
200        // without repeated model calls
201
202        let config = ExecutionConfig { timeout_secs: 5, ..Default::default() };
203
204        // In real usage:
205        // let executor = CodeExecutor::new(language, client, workspace);
206
207        // Example code pattern with error handling
208        let code_pattern = r#"
209try:
210        # Try to process data
211    x = 1 / 0  # This will raise ZeroDivisionError
212     result = {"error": False}
213 except ZeroDivisionError as e:
214    result = {"error": True, "type": "ZeroDivisionError", "message": str(e)}
215 except Exception as e:
216    result = {"error": True, "type": type(e).__name__, "message": str(e)}
217"#;
218
219        // Verify config is valid
220        assert_eq!(config.timeout_secs, 5);
221        assert_eq!(config.max_output_bytes, 10 * 1024 * 1024);
222        assert!(code_pattern.contains("try:"));
223        assert!(code_pattern.contains("except"));
224
225        // This pattern allows agents to handle errors in code
226        // without returning every exception to the model
227    }
228
229    // ============================================================================
230    // Test 6: Agent Behavior Analysis
231    // ============================================================================
232
233    #[test]
234    fn test_agent_behavior_tracking() {
235        let mut analyzer = AgentBehaviorAnalyzer::new();
236
237        // Record tool usage
238        analyzer.record_tool_usage(vtcode_config::constants::tools::LIST_FILES);
239        analyzer.record_tool_usage(vtcode_config::constants::tools::LIST_FILES);
240        analyzer.record_tool_usage("read_file");
241
242        // Record skill reuse
243        analyzer.record_skill_reuse("filter_skill");
244        analyzer.record_skill_reuse("filter_skill");
245
246        // Record failures
247        analyzer.record_tool_failure("grep_tool", "timeout");
248        analyzer.record_tool_failure("grep_tool", "pattern_error");
249
250        // Verify statistics
251        assert_eq!(
252            analyzer
253                .tool_stats()
254                .usage_frequency
255                .get(vtcode_config::constants::tools::LIST_FILES),
256            Some(&2)
257        );
258        assert_eq!(analyzer.skill_stats().reused_skills, 2);
259        assert!(!analyzer.failure_patterns().high_failure_tools.is_empty());
260
261        // Get recommendations
262        let tool_recs = analyzer.recommend_tools("list", 1);
263        assert!(tool_recs.contains(&vtcode_config::constants::tools::LIST_FILES.to_owned()));
264
265        // Identify risky tools
266        let risky = analyzer.identify_risky_tools(0.3);
267        assert!(!risky.is_empty());
268    }
269
270    // ============================================================================
271    // Scenario Tests
272    // ============================================================================
273
274    #[test]
275    fn test_scenario_simple_transformation() {
276        // Demonstrates simple data transformation pattern
277        // Transform data locally and return only the needed results
278
279        let config = ExecutionConfig { timeout_secs: 5, ..Default::default() };
280
281        // In real usage:
282        // let executor = CodeExecutor::new(language, client, workspace);
283
284        let code_pattern = r#"
285# Transform data locally before returning
286data = ["hello", "world", "test"]
287transformed = [s.upper() for s in data]
288result = {"original_count": len(data), "transformed": transformed}
289"#;
290
291        assert_eq!(config.max_output_bytes, 10 * 1024 * 1024);
292        assert!(code_pattern.contains("result = {"));
293
294        // This pattern keeps transformations local, reducing context overhead
295    }
296
297    #[test]
298    fn test_javascript_execution() {
299        // Demonstrates JavaScript code execution support
300
301        let config = ExecutionConfig { timeout_secs: 5, ..Default::default() };
302
303        // In real usage:
304        // let executor = CodeExecutor::new(Language::JavaScript, client, workspace);
305
306        let code_pattern = r#"
307const items = [1, 2, 3, 4, 5];
308const filtered = items.filter(x => x > 2);
309result = { count: filtered.length, items: filtered };
310"#;
311
312        assert_eq!(config.timeout_secs, 5);
313        assert!(code_pattern.contains("const items"));
314        assert!(code_pattern.contains("result ="));
315
316        // Agents can write JavaScript code just like Python
317    }
318}