Skip to main content

vtcode_skills/
document_processor.rs

1//! Document Processing for Skills using Vision Models
2//!
3//! Implements OpenAI-style document processing by converting PDFs, DOCX, and spreadsheets
4//! to rendered images for vision model analysis. This preserves layout, formatting, and
5//! visual information that would be lost in text extraction.
6//!
7//! ## Supported Formats
8//!
9//! - **PDF**: Multi-page documents converted to page-by-page PNGs
10//! - **DOCX/DOC**: Word documents rendered per-page
11//! - **Spreadsheets**: Excel/CSV files rendered as visual tables
12//! - **Images**: Direct vision model processing
13//!
14//! ## Architecture
15//!
16//! ```text
17//! Document → Renderer → PNG Images → Vision Model → Structured Data
18//! ```
19//!
20//! Inspired by OpenAI's implementation in ChatGPT's Code Interpreter.
21
22use anyhow::{Result, anyhow};
23use serde::{Deserialize, Serialize};
24use std::path::{Path, PathBuf};
25use tracing::{debug, info, warn};
26
27use vtcode_commons::fs::ensure_dir_exists_sync;
28
29/// Document processing configuration
30#[derive(Debug, Clone, Serialize, Deserialize)]
31pub struct DocumentProcessorConfig {
32    /// Enable vision-based document processing
33    enabled: bool,
34
35    /// Output format for rendered pages
36    image_format: String, // "png" recommended
37
38    /// DPI for rendering (higher = better quality but larger files)
39    dpi: u32,
40
41    /// Maximum number of pages to process (prevent runaway)
42    max_pages: usize,
43
44    /// Enable OCR fallback for text extraction
45    enable_ocr_fallback: bool,
46}
47
48impl Default for DocumentProcessorConfig {
49    fn default() -> Self {
50        Self {
51            enabled: true,
52            image_format: "png".to_string(),
53            dpi: 150,      // Good balance of quality vs file size
54            max_pages: 50, // Reasonable limit for most documents
55            enable_ocr_fallback: true,
56        }
57    }
58}
59
60/// Processed document result
61#[derive(Debug, Clone, Serialize, Deserialize)]
62pub struct ProcessedDocument {
63    /// Original document path
64    source_path: PathBuf,
65
66    /// Document type
67    doc_type: DocumentType,
68
69    /// Page count
70    page_count: usize,
71
72    /// Rendered page images
73    pages: Vec<PageImage>,
74
75    /// Extracted text (with layout preservation)
76    extracted_text: Option<String>,
77
78    /// Document metadata
79    metadata: DocumentMetadata,
80}
81
82/// Document type classification
83#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
84pub enum DocumentType {
85    Pdf,
86    Docx,
87    Doc,
88    Xlsx,
89    Xls,
90    Csv,
91    Txt,
92    Rtf,
93    Image,
94    Unknown,
95}
96
97impl DocumentType {
98    /// Detect document type from file extension
99    fn from_path(path: &Path) -> Self {
100        match path.extension().and_then(|e| e.to_str()) {
101            Some("pdf") => DocumentType::Pdf,
102            Some("docx") => DocumentType::Docx,
103            Some("doc") => DocumentType::Doc,
104            Some("xlsx") => DocumentType::Xlsx,
105            Some("xls") => DocumentType::Xls,
106            Some("csv") => DocumentType::Csv,
107            Some("txt") => DocumentType::Txt,
108            Some("rtf") => DocumentType::Rtf,
109            Some("png") | Some("jpg") | Some("jpeg") | Some("gif") | Some("bmp") | Some("tiff") => DocumentType::Image,
110            _ => DocumentType::Unknown,
111        }
112    }
113
114    /// Check if this document type is supported for vision processing
115    pub fn supports_vision_processing(&self) -> bool {
116        matches!(
117            self,
118            DocumentType::Pdf
119                | DocumentType::Docx
120                | DocumentType::Doc
121                | DocumentType::Xlsx
122                | DocumentType::Xls
123                | DocumentType::Image
124        )
125    }
126}
127
128/// Single page image data
129#[derive(Debug, Clone, Serialize, Deserialize)]
130pub struct PageImage {
131    /// Page number (1-indexed)
132    page_number: usize,
133
134    /// Image file path
135    image_path: PathBuf,
136
137    /// Image dimensions
138    dimensions: ImageDimensions,
139
140    /// Page text content (if OCR enabled)
141    text_content: Option<String>,
142}
143
144/// Image dimensions
145#[derive(Debug, Clone, Serialize, Deserialize)]
146pub struct ImageDimensions {
147    width: u32,
148    height: u32,
149}
150
151/// Document metadata
152#[derive(Debug, Clone, Serialize, Deserialize)]
153pub struct DocumentMetadata {
154    title: Option<String>,
155    author: Option<String>,
156    created_date: Option<String>,
157    modified_date: Option<String>,
158    file_size: u64,
159    page_count: Option<usize>,
160}
161
162/// Main document processor
163pub struct DocumentProcessor {
164    config: DocumentProcessorConfig,
165    temp_dir: PathBuf,
166}
167
168impl DocumentProcessor {
169    /// Create new document processor
170    fn new(config: DocumentProcessorConfig) -> Result<Self> {
171        let temp_dir = std::env::temp_dir().join("vtcode-document-processor");
172        ensure_dir_exists_sync(&temp_dir)?;
173
174        Ok(Self { config, temp_dir })
175    }
176
177    /// Process a document for vision model analysis
178    async fn process_document(&self, document_path: &Path) -> Result<ProcessedDocument> {
179        if !self.config.enabled {
180            return Err(anyhow!("Document processing is disabled"));
181        }
182
183        if !document_path.exists() {
184            return Err(anyhow!("Document not found: {}", document_path.display()));
185        }
186
187        let doc_type = DocumentType::from_path(document_path);
188        info!("Processing document: {} (type: {:?})", document_path.display(), doc_type);
189
190        match doc_type {
191            DocumentType::Pdf => self.process_pdf(document_path).await,
192            DocumentType::Docx | DocumentType::Doc => self.process_word_document(document_path).await,
193            DocumentType::Xlsx | DocumentType::Xls | DocumentType::Csv => self.process_spreadsheet(document_path).await,
194            DocumentType::Image => self.process_image(document_path).await,
195            other => {
196                warn!("Unsupported document type: {:?}", other);
197                Err(anyhow!("Unsupported document type: {other:?}"))
198            }
199        }
200    }
201
202    /// Process PDF document
203    async fn process_pdf(&self, pdf_path: &Path) -> Result<ProcessedDocument> {
204        debug!("Processing PDF: {}", pdf_path.display());
205
206        // For now, return a placeholder implementation
207        // In a full implementation, this would:
208        // 1. Use a PDF rendering library to convert pages to images
209        // 2. Optionally run OCR on each page
210        // 3. Extract metadata
211
212        let metadata = self.extract_file_metadata(pdf_path)?;
213
214        Ok(ProcessedDocument {
215            source_path: pdf_path.to_path_buf(),
216            doc_type: DocumentType::Pdf,
217            page_count: 1,        // Placeholder
218            pages: vec![],        // Placeholder - would contain actual rendered pages
219            extracted_text: None, // Placeholder - would contain OCR text if enabled
220            metadata,
221        })
222    }
223
224    /// Process Word document
225    async fn process_word_document(&self, doc_path: &Path) -> Result<ProcessedDocument> {
226        debug!("Processing Word document: {}", doc_path.display());
227
228        let metadata = self.extract_file_metadata(doc_path)?;
229
230        Ok(ProcessedDocument {
231            source_path: doc_path.to_path_buf(),
232            doc_type: DocumentType::Docx,
233            page_count: 1, // Placeholder
234            pages: vec![],
235            extracted_text: None,
236            metadata,
237        })
238    }
239
240    /// Process spreadsheet
241    async fn process_spreadsheet(&self, spreadsheet_path: &Path) -> Result<ProcessedDocument> {
242        debug!("Processing spreadsheet: {}", spreadsheet_path.display());
243
244        let metadata = self.extract_file_metadata(spreadsheet_path)?;
245        let doc_type = DocumentType::from_path(spreadsheet_path);
246
247        Ok(ProcessedDocument {
248            source_path: spreadsheet_path.to_path_buf(),
249            doc_type,
250            page_count: 1, // Spreadsheets are typically single "sheet"
251            pages: vec![],
252            extracted_text: None,
253            metadata,
254        })
255    }
256
257    /// Process image file
258    async fn process_image(&self, image_path: &Path) -> Result<ProcessedDocument> {
259        debug!("Processing image: {}", image_path.display());
260
261        let metadata = self.extract_file_metadata(image_path)?;
262
263        Ok(ProcessedDocument {
264            source_path: image_path.to_path_buf(),
265            doc_type: DocumentType::Image,
266            page_count: 1,
267            pages: vec![PageImage {
268                page_number: 1,
269                image_path: image_path.to_path_buf(),
270                dimensions: ImageDimensions { width: 0, height: 0 }, // Would detect actual dimensions
271                text_content: None,
272            }],
273            extracted_text: None,
274            metadata,
275        })
276    }
277
278    /// Extract basic file metadata
279    fn extract_file_metadata(&self, path: &Path) -> Result<DocumentMetadata> {
280        let metadata = std::fs::metadata(path)?;
281
282        Ok(DocumentMetadata {
283            title: None,
284            author: None,
285            created_date: None,
286            modified_date: None,
287            file_size: metadata.len(),
288            page_count: None,
289        })
290    }
291
292    /// Generate a prompt for vision model analysis
293    pub fn generate_vision_prompt(&self, processed: &ProcessedDocument, query: &str) -> Result<String> {
294        let mut prompt = String::new();
295
296        prompt.push_str(&format!("Document: {}\n", processed.source_path.display()));
297        prompt.push_str(&format!("Type: {:?}\n", processed.doc_type));
298        prompt.push_str(&format!("Pages: {}\n\n", processed.page_count));
299
300        if let Some(text) = &processed.extracted_text {
301            prompt.push_str("Extracted Text:\n");
302            prompt.push_str(text);
303            prompt.push_str("\n\n");
304        }
305
306        prompt.push_str("Analyze the document images and provide: ");
307        prompt.push_str("\n1. A summary of the content");
308        prompt.push_str("\n2. Key insights or findings");
309        prompt.push_str("\n3. Answers to specific questions");
310        prompt.push_str(&format!("\n\nSpecific query: {query}\n"));
311
312        Ok(prompt)
313    }
314
315    /// Clean up temporary files
316    fn cleanup(&self) -> Result<()> {
317        if self.temp_dir.exists() {
318            std::fs::remove_dir_all(&self.temp_dir)?;
319            debug!("Cleaned up temporary directory: {}", self.temp_dir.display());
320        }
321        Ok(())
322    }
323}
324
325impl Drop for DocumentProcessor {
326    fn drop(&mut self) {
327        // Attempt to clean up on drop
328        let _ = self.cleanup();
329    }
330}
331
332#[cfg(test)]
333mod tests {
334    use super::*;
335
336    #[test]
337    fn test_document_type_detection() {
338        assert_eq!(DocumentType::from_path(Path::new("test.pdf")), DocumentType::Pdf);
339        assert_eq!(DocumentType::from_path(Path::new("test.docx")), DocumentType::Docx);
340        assert_eq!(DocumentType::from_path(Path::new("test.xlsx")), DocumentType::Xlsx);
341        assert_eq!(DocumentType::from_path(Path::new("test.png")), DocumentType::Image);
342        assert_eq!(DocumentType::from_path(Path::new("test.unknown")), DocumentType::Unknown);
343    }
344
345    #[test]
346    fn test_document_processor_creation() {
347        let config = DocumentProcessorConfig::default();
348        let processor = DocumentProcessor::new(config).unwrap();
349        assert!(processor.temp_dir.exists());
350    }
351
352    #[tokio::test]
353    async fn test_process_nonexistent_document() {
354        let config = DocumentProcessorConfig::default();
355        let processor = DocumentProcessor::new(config).unwrap();
356
357        let result = processor.process_document(Path::new("/nonexistent/document.pdf")).await;
358        result.unwrap_err();
359    }
360}